From 885fd9002e9285394c2b950908316ebb65b67a93 Mon Sep 17 00:00:00 2001 From: saif <11242541+saifsmailbox98@users.noreply.github.com> Date: Thu, 17 Sep 2026 03:07:39 +0530 Subject: [PATCH 01/43] feat(agent-vault): ship session activity from the proxy Every request that reaches forwardHTTP becomes a record, a blocked one included, since under the default any-host policy an agent reaching somewhere nobody configured is ordinary passthrough traffic and logging only brokered calls would make the one request worth catching invisible. Records go into a bounded per-session ring; a flusher drains it every 60 seconds or every 1000 records, seals each slice with AES-256-GCM, POSTs the metadata for a presigned URL and PUTs the ciphertext to the customer's bucket. The hot path costs one append under a mutex. Chunk ids are ULIDs rather than a counter: a counter resets whenever the session cache evicts an entry, which happens at nine ordinary sites, and would then collide with the server's unique index for the rest of the session's life. Resolve carries the key exactly once per session. The proxy reports that it holds one and the backend skips the unwrap, which is a KMS round trip, on every poll after the first. Nothing is persisted to disk. A failed upload is retried with the same chunk id, which the server replays idempotently; an outage pauses rather than discards; and a per-tick breaker keeps a hundred spools against a dead bucket or an unreachable control plane from costing a hundred serial timeouts. --- go.mod | 2 +- packages/agentvault/activity.go | 445 ++++++++++++ packages/agentvault/activity_crypto.go | 82 +++ packages/agentvault/activity_crypto_test.go | 182 +++++ packages/agentvault/activity_resolve_test.go | 92 +++ packages/agentvault/activity_ship.go | 104 +++ packages/agentvault/activity_ship_test.go | 121 +++ packages/agentvault/activity_spool.go | 153 ++++ packages/agentvault/activity_test.go | 687 ++++++++++++++++++ packages/agentvault/activity_wiring_test.go | 223 ++++++ packages/agentvault/cache.go | 50 +- packages/agentvault/cache_test.go | 14 +- packages/agentvault/proxy.go | 27 +- packages/agentvault/proxy_policy_test.go | 2 +- packages/agentvault/proxy_truncation_test.go | 2 +- .../agentvault/proxy_tunnel_errors_test.go | 4 +- packages/agentvault/resolve.go | 51 +- packages/agentvault/resolve_client_test.go | 2 +- packages/agentvault/run.go | 14 + packages/api/agent_vault.go | 62 +- 20 files changed, 2289 insertions(+), 30 deletions(-) create mode 100644 packages/agentvault/activity.go create mode 100644 packages/agentvault/activity_crypto.go create mode 100644 packages/agentvault/activity_crypto_test.go create mode 100644 packages/agentvault/activity_resolve_test.go create mode 100644 packages/agentvault/activity_ship.go create mode 100644 packages/agentvault/activity_ship_test.go create mode 100644 packages/agentvault/activity_spool.go create mode 100644 packages/agentvault/activity_test.go create mode 100644 packages/agentvault/activity_wiring_test.go diff --git a/go.mod b/go.mod index 25291073f..25fdb3a36 100644 --- a/go.mod +++ b/go.mod @@ -43,6 +43,7 @@ require ( github.com/muesli/reflow v0.3.0 github.com/muesli/roff v0.1.0 github.com/oiweiwei/go-msrpc v1.5.1 + github.com/oklog/ulid v1.3.1 github.com/pelletier/go-toml/v2 v2.4.3 github.com/pkg/browser v0.0.0-20240102092130-5ac0b6a4141c github.com/pkg/errors v0.9.1 @@ -190,7 +191,6 @@ require ( github.com/oiweiwei/go-oem v1.0.0 // indirect github.com/oiweiwei/go-smb2.fork v1.0.1 // indirect github.com/oiweiwei/gokrb5.fork/v9 v9.0.6 // indirect - github.com/oklog/ulid v1.3.1 // indirect github.com/onsi/ginkgo/v2 v2.22.2 // indirect github.com/onsi/gomega v1.36.2 // indirect github.com/oracle/oci-go-sdk/v65 v65.95.2 // indirect diff --git a/packages/agentvault/activity.go b/packages/agentvault/activity.go new file mode 100644 index 000000000..84c8e3b0c --- /dev/null +++ b/packages/agentvault/activity.go @@ -0,0 +1,445 @@ +package agentvault + +import ( + "context" + "sync" + "time" + + "github.com/Infisical/infisical-merge/packages/api" + "github.com/rs/zerolog/log" +) + +const ( + // The cost knob is this interval, not traffic volume: one flush is one S3 PUT. At 60s a session running + // flat out costs about 1,440 PUTs a day; flushing every 5s would be twelve times the bill for the same + // records. + activityFlushInterval = 60 * time.Second + // The server refuses a chunk over this, so the ring is drained in slices of at most this many. + activityFlushRecords = 1000 + + // Five missed flushes of headroom per session before the oldest records start being overwritten. + activitySpoolCapacity = 5000 + // A fuse across every session on this proxy, roughly 60 MB of records. + activityTotalCapacity = 200_000 + + // Sealed chunks kept per spool while shipping fails. One chunk seals per tick during an outage, so this + // is about ten minutes of Infisical or S3 being unreachable before a busy session loses its oldest + // sealed chunk. Deliberate for a preview with no disk persistence: raise this rather than the interval. + activityPendingChunks = 10 + // A second fuse, on sealed ciphertext rather than record count, across every spool. + activityTotalSealedBytes = 64 << 20 + + // Longer than sessionInactiveTTL, so a session evicted from the cache still gets its final flush. + activityIdleClose = 15 * time.Minute + + activityPauseBackoff = 15 * time.Minute + activityPutTimeout = 10 * time.Second + activityFinalTimeout = 3 * time.Second + + // Read off APIError.Name. Defined by the backend in agent-vault-activity-constants.ts. + activityCeilingReachedName = "AgentVaultActivityCeilingReached" + activityDisabledName = "AgentVaultActivityDisabled" +) + +// activityGrant is what resolve hands back when logging is on for a session. A nil grant means "do not +// record", which is the whole of the disabled path. +type activityGrant struct { + sessionID string + projectID string + key []byte +} + +// activityShipper is the seam the tests replace. Two calls, because delivery is two steps: Infisical +// writes the index row and returns a presigned URL, then the bytes go straight to the customer's bucket. +type activityShipper interface { + createChunk(final bool, sessionID string, req api.CreateAgentVaultActivityChunkRequest) (api.CreateAgentVaultActivityChunkResponse, error) + putObject(ctx context.Context, url string, ciphertext []byte) error +} + +type activityLog struct { + proxyID string + shipper activityShipper + now func() time.Time + + // Held for the whole of flushAll. Shutdown calls it from a second goroutine while the run loop may + // still be inside one, and the two would otherwise ship the same chunk twice and race on its fields. + flushMu sync.Mutex + + mu sync.Mutex + spools map[string]*activitySpool + + // A session's sequence numbers have to keep climbing across the spool being forgotten and rebuilt, + // or one proxy emits two records with the same (proxyId, seq) for one session. Cleared wholesale + // when it grows, the way the session cache handles its own refusal map. + seqBySession map[string]uint64 + + total int + sealedBytes int + + // Proxy-wide, because both reasons are proxy-wide: the ceiling is per organization and the switch is + // per project, and this proxy serves one project. + pauseUntil time.Time + pauseReason string + + // Reset at the top of every flushAll. Once either side has failed once in a tick, the remaining spools + // seal but skip both calls, so a hundred spools against a blocked egress or an unreachable control + // plane cost one timeout rather than a hundred. + s3Down bool + infisicalDown bool + + closed bool + wake chan struct{} +} + +func newActivityLog(proxyID string, shipper activityShipper) *activityLog { + return &activityLog{ + proxyID: proxyID, + shipper: shipper, + now: time.Now, + spools: make(map[string]*activitySpool), + seqBySession: make(map[string]uint64), + wake: make(chan struct{}, 1), + } +} + +// record is the entire hot-path cost: one append under a mutex. No I/O, no crypto. +// +// Nil-safe on both the receiver and the grant, so a bare &proxyServer{} test fixture and a session whose +// logging is off both cost a single comparison. +func (a *activityLog) record(g *activityGrant, rec activityRecord) { + if a == nil || g == nil { + return + } + + a.mu.Lock() + defer a.mu.Unlock() + if a.closed { + return + } + + spool, ok := a.spools[g.sessionID] + if !ok { + spool = newActivitySpool(g, a.now()) + spool.nextSeq = a.seqBySession[g.sessionID] + a.spools[g.sessionID] = spool + } + + // A sequence number is consumed even while paused or full, so the gap is counted rather than silent + // and a chunk's firstSeq reveals exactly how many records are missing before it. + rec.Seq = spool.nextSeq + spool.nextSeq++ + rec.ProxyID = a.proxyID + rec.Ts = a.now().UTC().Format(time.RFC3339Nano) + spool.lastRecordAt = a.now() + + if !a.pauseUntil.IsZero() && a.now().Before(a.pauseUntil) { + spool.ring.dropped++ + return + } + + if a.total >= activityTotalCapacity { + // The newest is dropped rather than the oldest: at the proxy-wide fuse the ring's own eviction is + // already running, and dropping the newest keeps one bounded behaviour rather than two. + spool.ring.dropped++ + return + } + + if evicted := spool.ring.push(rec); !evicted { + a.total++ + } + + if spool.ring.len() >= activityFlushRecords { + select { + case a.wake <- struct{}{}: + default: + } + } +} + +// run is a single loop, so a flush can never overlap itself and no same-session guard is needed. +func (a *activityLog) run(stop <-chan struct{}) { + if a == nil { + return + } + ticker := time.NewTicker(activityFlushInterval) + defer ticker.Stop() + + for { + select { + case <-stop: + return + case <-ticker.C: + a.flushAll(context.Background(), false) + case <-a.wake: + a.flushAll(context.Background(), false) + } + } +} + +// close stops recording and makes one last attempt to ship everything buffered, within the deadline on ctx. +func (a *activityLog) close(ctx context.Context) { + if a == nil { + return + } + a.mu.Lock() + a.closed = true + a.mu.Unlock() + + a.flushAll(ctx, true) + + a.mu.Lock() + defer a.mu.Unlock() + var lost int + for _, spool := range a.spools { + lost += spool.ring.len() + for _, chunk := range spool.pending { + lost += chunk.meta.RecordCount + } + } + if lost > 0 { + log.Warn().Int("records", lost).Msg("agent-vault: activity records were not shipped before shutdown") + } +} + +// dueSpools picks what to flush and forgets idle spools, under the lock. Flushing then happens off it. +func (a *activityLog) dueSpools(final bool) []*activitySpool { + a.mu.Lock() + defer a.mu.Unlock() + + now := a.now() + due := make([]*activitySpool, 0, len(a.spools)) + for id, spool := range a.spools { + if spool.ring.len() == 0 && len(spool.pending) == 0 { + // A spool is independent of the session cache: eviction there just means record() stops + // arriving, and this is what eventually forgets it. + if !final && now.Sub(spool.lastRecordAt) > activityIdleClose { + a.forgetSpoolLocked(id, spool) + } + continue + } + if final || spool.ring.len() >= activityFlushRecords || len(spool.pending) > 0 || + (spool.ring.len() > 0 && now.Sub(spool.lastFlushAt) >= activityFlushInterval) { + due = append(due, spool) + } + } + return due +} + +// forgetSpoolLocked drops a spool but keeps where its sequence numbers had reached. +func (a *activityLog) forgetSpoolLocked(sessionID string, spool *activitySpool) { + if len(a.seqBySession) >= maxSessionCacheEntries { + a.seqBySession = make(map[string]uint64) + } + a.seqBySession[sessionID] = spool.nextSeq + delete(a.spools, sessionID) +} + +func (a *activityLog) paused() (bool, string) { + a.mu.Lock() + defer a.mu.Unlock() + if a.pauseUntil.IsZero() || !a.now().Before(a.pauseUntil) { + return false, "" + } + return true, a.pauseReason +} + +func (a *activityLog) pause(reason string) { + a.mu.Lock() + defer a.mu.Unlock() + a.pauseUntil = a.now().Add(activityPauseBackoff) + a.pauseReason = reason +} + +func (a *activityLog) flushAll(ctx context.Context, final bool) { + if a == nil { + return + } + + // One goroutine in flushAll at a time: the run loop and shutdown both call it. + a.flushMu.Lock() + defer a.flushMu.Unlock() + + a.mu.Lock() + a.s3Down = false + a.infisicalDown = false + a.mu.Unlock() + + for _, spool := range a.dueSpools(final) { + // Sequential and off the lock. At one chunk per session per minute and ~100ms per PUT, a hundred + // sessions finish well inside a 60s tick, and one loop means no same-session overlap to guard. + a.flushSpool(ctx, spool, final) + } +} + +// sealRing drains the ring into sealed chunks, in slices the server will accept. +// +// The marshal and the AES pass run off the lock. They are only milliseconds, but it is the same lock +// every proxied request takes to append a record, so holding it across them would stall the request path. +func (a *activityLog) sealRing(spool *activitySpool) { + for { + a.mu.Lock() + records := spool.ring.drain(activityFlushRecords) + if len(records) == 0 { + spool.lastFlushAt = a.now() + a.mu.Unlock() + return + } + a.total -= len(records) + // droppedCount rides the first slice only, so one gap is reported once. + dropped := spool.ring.takeDropped() + now := a.now() + a.mu.Unlock() + + chunk, err := spool.sealSlice(a.proxyID, records, dropped, now) + if err != nil { + a.mu.Lock() + // These records are gone, so they join the gap rather than vanishing from the count with it. + spool.ring.dropped += dropped + uint64(len(records)) + a.mu.Unlock() + log.Error().Err(err).Str("sessionId", spool.sessionID).Int("records", len(records)). + Msg("agent-vault: could not seal an activity chunk, dropping those records") + continue + } + + a.mu.Lock() + spool.pending = append(spool.pending, chunk) + a.sealedBytes += len(chunk.ciphertext) + a.enforcePendingCapsLocked(spool) + a.mu.Unlock() + } +} + +// enforcePendingCapsLocked runs on every append, because the caps are a property of `pending` rather than +// a branch of the ceiling handler. The oldest sealed chunk is evicted and its records counted as dropped. +func (a *activityLog) enforcePendingCapsLocked(spool *activitySpool) { + for len(spool.pending) > activityPendingChunks || (a.sealedBytes > activityTotalSealedBytes && len(spool.pending) > 1) { + oldest := spool.pending[0] + spool.pending = spool.pending[1:] + a.sealedBytes -= len(oldest.ciphertext) + spool.ring.dropped += uint64(oldest.meta.RecordCount) + log.Warn(). + Str("sessionId", spool.sessionID). + Str("chunkId", oldest.meta.ChunkID). + Int("records", oldest.meta.RecordCount). + Msg("agent-vault: dropped an unshipped activity chunk, the buffer is full") + } +} + +func (a *activityLog) flushSpool(ctx context.Context, spool *activitySpool, final bool) { + if paused, reason := a.paused(); paused { + // Seal once at pause onset so the ring does not overflow while waiting, then hold everything: ops + // may raise the ceiling within the window, and the chunks are still shippable when it lifts. + a.sealRing(spool) + _ = reason + return + } + + a.sealRing(spool) + + for { + a.mu.Lock() + if len(spool.pending) == 0 || a.s3Down || a.infisicalDown { + a.mu.Unlock() + return + } + chunk := spool.pending[0] + a.mu.Unlock() + + if !a.shipChunk(ctx, spool, chunk, final) { + return + } + + a.mu.Lock() + if len(spool.pending) > 0 && spool.pending[0] == chunk { + spool.pending = spool.pending[1:] + a.sealedBytes -= len(chunk.ciphertext) + } + a.mu.Unlock() + } +} + +// shipChunk delivers one chunk. It returns false when this spool's loop should stop for the tick. +func (a *activityLog) shipChunk(ctx context.Context, spool *activitySpool, chunk *sealedChunk, final bool) bool { + if chunk.uploadURL == "" || a.now().Add(10*time.Second).After(chunk.urlExpires) { + res, err := a.shipper.createChunk(final, spool.sessionID, chunk.meta) + if err != nil { + return a.handleCreateFailure(spool, chunk, err) + } + chunk.uploadURL = res.UploadURL + chunk.urlExpires = a.now().Add(time.Duration(res.ExpiresInSeconds) * time.Second) + } + + putCtx := ctx + if !final { + var cancel context.CancelFunc + putCtx, cancel = context.WithTimeout(ctx, activityPutTimeout) + defer cancel() + } + + if err := a.shipper.putObject(putCtx, chunk.uploadURL, chunk.ciphertext); err != nil { + // The row already exists, so re-POSTing the same chunk id replays idempotently and yields a fresh + // url. Clearing it is what makes the next tick do that. + chunk.uploadURL = "" + a.mu.Lock() + a.s3Down = true + a.mu.Unlock() + log.Warn().Err(err).Str("sessionId", spool.sessionID).Str("chunkId", chunk.meta.ChunkID). + Msg("agent-vault: could not upload an activity chunk, will retry") + return false + } + + return true +} + +func (a *activityLog) handleCreateFailure(spool *activitySpool, chunk *sealedChunk, err error) bool { + switch { + case isProxyTokenRejected(err): + // The poll loop exits within two heartbeats. Keep everything until it does. + log.Warn().Err(err).Msg("agent-vault: Infisical rejected this proxy's token, holding activity") + return false + + case isSessionGone(err): + a.mu.Lock() + a.total -= spool.ring.len() + for _, held := range spool.pending { + a.sealedBytes -= len(held.ciphertext) + } + // The session is gone for good, so its sequence numbers are not worth remembering. + a.forgetSpoolLocked(spool.sessionID, spool) + delete(a.seqBySession, spool.sessionID) + a.mu.Unlock() + log.Debug().Str("sessionId", spool.sessionID).Msg("agent-vault: session gone, dropping its activity") + return false + + case isActivityErrorNamed(err, activityCeilingReachedName): + a.pause(activityCeilingReachedName) + log.Warn().Err(err).Msg("agent-vault: the organization's activity storage limit is reached, pausing for 15m") + return false + + case isActivityErrorNamed(err, activityDisabledName): + a.pause(activityDisabledName) + log.Warn().Msg("agent-vault: activity logging is switched off for this project, pausing for 15m") + return false + + case isPoisonChunk(err): + // The server will never accept this chunk, so retrying costs the whole spool. + a.mu.Lock() + if len(spool.pending) > 0 && spool.pending[0] == chunk { + spool.pending = spool.pending[1:] + a.sealedBytes -= len(chunk.ciphertext) + } + a.mu.Unlock() + log.Error().Err(err).Str("chunkId", chunk.meta.ChunkID). + Msg("agent-vault: Infisical rejected an activity chunk as malformed, dropping it") + return false + + default: + // A timeout, a 5xx or a 429. Whatever it is, the rest of this tick will meet it too. + a.mu.Lock() + a.infisicalDown = true + a.mu.Unlock() + log.Warn().Err(err).Str("sessionId", spool.sessionID). + Msg("agent-vault: could not record activity, will retry") + return false + } +} diff --git a/packages/agentvault/activity_crypto.go b/packages/agentvault/activity_crypto.go new file mode 100644 index 000000000..4a9a75505 --- /dev/null +++ b/packages/agentvault/activity_crypto.go @@ -0,0 +1,82 @@ +package agentvault + +import ( + "crypto/aes" + "crypto/cipher" + "crypto/rand" + "crypto/sha256" + "encoding/base64" + "fmt" + "io" + "time" + + "github.com/oklog/ulid" +) + +// The version suffix on the additional authenticated data. Bumping it makes every existing chunk +// undecryptable, so it changes only alongside a migration of the stored objects. +const activityAADVersion = "v1" + +const activityIVBytes = 12 + +// buildActivityAAD binds a sealed chunk to exactly one place in the hierarchy, so holding the session key +// is not enough to replay a chunk under another project, session or proxy. +// +// The same string is built by the browser in frontend/src/hooks/api/agentVault/activityDecrypt.ts, and the +// backend pins a known-good vector in agent-vault-activity-crypto.test.ts. All three must agree or +// playback fails with no useful error. +func buildActivityAAD(projectID, sessionID, proxyID, chunkID string) []byte { + sum := sha256.Sum256([]byte(fmt.Sprintf("%s|%s|%s|%s|%s", projectID, sessionID, proxyID, chunkID, activityAADVersion))) + return sum[:] +} + +// sealActivity produces the layout Web Crypto's decrypt expects: a 12-byte IV carried beside the object, +// and the 16-byte GCM tag appended to the ciphertext rather than kept separately. +func sealActivity(key, plaintext, aad []byte) (ciphertext []byte, iv []byte, err error) { + return sealActivityWithRand(rand.Reader, key, plaintext, aad) +} + +// sealActivityWithRand takes the IV source so a test can pin one and compare against the backend's vector. +func sealActivityWithRand(random io.Reader, key, plaintext, aad []byte) (ciphertext []byte, iv []byte, err error) { + block, err := aes.NewCipher(key) + if err != nil { + return nil, nil, fmt.Errorf("agent-vault: activity key is not a valid AES key: %w", err) + } + gcm, err := cipher.NewGCM(block) + if err != nil { + return nil, nil, fmt.Errorf("agent-vault: could not build GCM: %w", err) + } + + iv = make([]byte, activityIVBytes) + if _, err = io.ReadFull(random, iv); err != nil { + return nil, nil, fmt.Errorf("agent-vault: could not read a nonce: %w", err) + } + + // A nil destination makes Seal allocate, so the output is exactly ciphertext||tag with no IV prefix. + return gcm.Seal(nil, iv, plaintext, aad), iv, nil +} + +// encodeActivityIV matches the backend's `^[A-Za-z0-9+/]{16}$`: standard alphabet, no padding. +func encodeActivityIV(iv []byte) string { + return base64.RawStdEncoding.EncodeToString(iv) +} + +// newActivityChunkID mints a ULID, which sorts by time and is unique per session. A proxy-side counter +// cannot be used: it resets whenever the session cache evicts an entry, which happens at nine ordinary +// sites, and would then collide with the server's unique index for the rest of the session's life. +func newActivityChunkID(now time.Time) string { + return ulid.MustNew(ulid.Timestamp(now), newULIDEntropy()).String() +} + +// ulid.Monotonic is deliberately not used: it keeps state per reader, and two goroutines sealing in the +// same millisecond would need a mutex around it for no benefit. 80 bits of randomness is ample here. +func newULIDEntropy() io.Reader { return rand.Reader } + +// Kept so a test can assert the id is well formed without reaching for the library. +func parseActivityChunkID(id string) (time.Time, error) { + parsed, err := ulid.Parse(id) + if err != nil { + return time.Time{}, err + } + return ulid.Time(parsed.Time()), nil +} diff --git a/packages/agentvault/activity_crypto_test.go b/packages/agentvault/activity_crypto_test.go new file mode 100644 index 000000000..97194dab1 --- /dev/null +++ b/packages/agentvault/activity_crypto_test.go @@ -0,0 +1,182 @@ +package agentvault + +import ( + "bytes" + "crypto/aes" + "crypto/cipher" + "encoding/base64" + "encoding/hex" + "encoding/json" + "testing" + "time" +) + +// The fixture the backend pins in agent-vault-activity-crypto.test.ts. Three implementations seal or open +// these bytes (this one, Infisical's reference, and the browser's), so a change to the AAD string, the IV +// width or the tag placement has to fail somewhere rather than surface as "playback is broken". +const ( + vectorKeyHex = "000102030405060708090a0b0c0d0e0f101112131415161718191a1b1c1d1e1f" + vectorIVHex = "aabbccddeeff001122334455" + vectorAADHex = "ba75c71ef714535e84246066ca0a34685c42a03a130dd92fe1d795ad40908a7c" + vectorIVBase64 = "qrvM3e7/ABEiM0RV" + vectorCiphertext = "PLRwxBbgu+W68Br1N9gY1oUy8wjJxQClAtBh0NfJS1UcWOCPn3laS615sIqwFONhPIPNWRI3CA+a5tUJ7aoim0sQkE4d9gzou2mc/AWiCdToVBJPtdumA9jIzh3yAI81YPwcoDXEVnq2+7ooNNJShGdLX95itbrna/t4nFKRKSSgNzbH23eMtSMcSo72puk/2iwh4sVbTKzC2kwvbf1U6Mgd21zkIq2jDKKwhcT6mTfjPivW4FzmmkspQVMoWwANRX+QVyXzrMipZfoq5N/UcUI6rCvav2ddgiSoqXrTvwiXaUgv" +) + +var vectorContext = struct{ projectID, sessionID, proxyID, chunkID string }{ + projectID: "proj-1", + sessionID: "sess-1", + proxyID: "proxy-1", + chunkID: "01K5ABCDEFGHJKMNPQRSTVWXYZ", +} + +func vectorRecords() []activityRecord { + service, bundle := "github", "code-review" + return []activityRecord{{ + Ts: "2026-09-16T10:31:04.221Z", + Seq: 1, + ProxyID: "proxy-1", + Method: "GET", + Host: "api.github.com", + Port: "443", + Path: "/zen", + Status: 200, + Decision: decisionBrokered, + Service: &service, + AccessBundle: &bundle, + }} +} + +func mustHex(t *testing.T, s string) []byte { + t.Helper() + b, err := hex.DecodeString(s) + if err != nil { + t.Fatalf("bad hex fixture: %v", err) + } + return b +} + +func TestActivityAADMatchesTheBackendVector(t *testing.T) { + got := buildActivityAAD(vectorContext.projectID, vectorContext.sessionID, vectorContext.proxyID, vectorContext.chunkID) + if hex.EncodeToString(got) != vectorAADHex { + t.Fatalf("AAD is %s, the backend and the browser build %s", hex.EncodeToString(got), vectorAADHex) + } +} + +func TestSealMatchesNodeVector(t *testing.T) { + plaintext, err := json.Marshal(vectorRecords()) + if err != nil { + t.Fatal(err) + } + + iv := mustHex(t, vectorIVHex) + aad := buildActivityAAD(vectorContext.projectID, vectorContext.sessionID, vectorContext.proxyID, vectorContext.chunkID) + ciphertext, gotIV, err := sealActivityWithRand(bytes.NewReader(iv), mustHex(t, vectorKeyHex), plaintext, aad) + if err != nil { + t.Fatal(err) + } + + if encodeActivityIV(gotIV) != vectorIVBase64 { + t.Fatalf("IV encodes as %q, the backend expects %q", encodeActivityIV(gotIV), vectorIVBase64) + } + if base64.StdEncoding.EncodeToString(ciphertext) != vectorCiphertext { + t.Fatal("the sealed bytes differ from the vector Infisical and the browser are checked against") + } +} + +// What the browser does, so the layout is proven openable rather than merely reproducible. +func TestSealedChunkOpensWithTheTagAppended(t *testing.T) { + key := mustHex(t, vectorKeyHex) + aad := buildActivityAAD(vectorContext.projectID, vectorContext.sessionID, vectorContext.proxyID, vectorContext.chunkID) + plaintext, _ := json.Marshal(vectorRecords()) + + ciphertext, iv, err := sealActivity(key, plaintext, aad) + if err != nil { + t.Fatal(err) + } + if len(ciphertext) != len(plaintext)+16 { + t.Fatalf("sealed length is %d, expected the plaintext plus a 16-byte tag", len(ciphertext)) + } + + block, _ := aes.NewCipher(key) + gcm, _ := cipher.NewGCM(block) + opened, err := gcm.Open(nil, iv, ciphertext, aad) + if err != nil { + t.Fatalf("a chunk this proxy sealed could not be opened: %v", err) + } + if !bytes.Equal(opened, plaintext) { + t.Fatal("the opened plaintext differs from what was sealed") + } +} + +func TestAChunkCannotBeReplayedElsewhere(t *testing.T) { + key := mustHex(t, vectorKeyHex) + plaintext, _ := json.Marshal(vectorRecords()) + aad := buildActivityAAD(vectorContext.projectID, vectorContext.sessionID, vectorContext.proxyID, vectorContext.chunkID) + ciphertext, iv, err := sealActivity(key, plaintext, aad) + if err != nil { + t.Fatal(err) + } + + block, _ := aes.NewCipher(key) + gcm, _ := cipher.NewGCM(block) + + for _, wrong := range []struct { + name string + aad []byte + }{ + {"another project", buildActivityAAD("other", vectorContext.sessionID, vectorContext.proxyID, vectorContext.chunkID)}, + {"another session", buildActivityAAD(vectorContext.projectID, "other", vectorContext.proxyID, vectorContext.chunkID)}, + {"another proxy", buildActivityAAD(vectorContext.projectID, vectorContext.sessionID, "other", vectorContext.chunkID)}, + {"another chunk", buildActivityAAD(vectorContext.projectID, vectorContext.sessionID, vectorContext.proxyID, "other")}, + } { + if _, err := gcm.Open(nil, iv, ciphertext, wrong.aad); err == nil { + t.Fatalf("a chunk opened under %s", wrong.name) + } + } +} + +func TestIVsDoNotRepeat(t *testing.T) { + key := mustHex(t, vectorKeyHex) + seen := make(map[string]bool, 256) + for i := 0; i < 256; i++ { + _, iv, err := sealActivity(key, []byte("[]"), nil) + if err != nil { + t.Fatal(err) + } + if len(iv) != activityIVBytes { + t.Fatalf("IV is %d bytes, the contract is %d", len(iv), activityIVBytes) + } + if seen[string(iv)] { + t.Fatal("an IV repeated, which would void GCM's guarantees for this key") + } + seen[string(iv)] = true + } +} + +func TestChunkIDsAreULIDsThatSortByTime(t *testing.T) { + earlier := newActivityChunkID(time.Date(2026, 9, 16, 10, 0, 0, 0, time.UTC)) + later := newActivityChunkID(time.Date(2026, 9, 16, 11, 0, 0, 0, time.UTC)) + + if len(earlier) != 26 { + t.Fatalf("a chunk id is %d characters, the server's column is 26", len(earlier)) + } + // The read cursor is a plain string comparison on this column, so lexical order has to be time order. + if !(earlier < later) { + t.Fatalf("%q did not sort before %q", earlier, later) + } + if _, err := parseActivityChunkID(earlier); err != nil { + t.Fatalf("a minted chunk id did not parse: %v", err) + } +} + +func TestChunkIDsAreUniqueWithinAMillisecond(t *testing.T) { + now := time.Now() + seen := make(map[string]bool, 1000) + for i := 0; i < 1000; i++ { + id := newActivityChunkID(now) + if seen[id] { + t.Fatal("a chunk id repeated, which would collide with the server's unique index") + } + seen[id] = true + } +} diff --git a/packages/agentvault/activity_resolve_test.go b/packages/agentvault/activity_resolve_test.go new file mode 100644 index 000000000..28bbe9bdd --- /dev/null +++ b/packages/agentvault/activity_resolve_test.go @@ -0,0 +1,92 @@ +package agentvault + +import ( + "encoding/base64" + "testing" + + "github.com/Infisical/infisical-merge/packages/api" +) + +func enabledGrantWire(key string) api.AgentVaultActivityGrant { + return api.AgentVaultActivityGrant{Enabled: true, SessionKey: key, ProjectID: "proj-1"} +} + +func aKey(b byte) []byte { + key := make([]byte, activityKeyBytes) + for i := range key { + key[i] = b + } + return key +} + +func TestTheFirstResolveTakesTheKeyOffTheWire(t *testing.T) { + want := aKey(7) + got := toActivityGrant("s1", enabledGrantWire(base64.StdEncoding.EncodeToString(want)), nil) + + if got == nil { + t.Fatal("activity was enabled but no grant was built") + } + if got.sessionID != "s1" || got.projectID != "proj-1" { + t.Fatalf("grant names session %q project %q", got.sessionID, got.projectID) + } + if string(got.key) != string(want) { + t.Fatal("the key on the grant is not the key Infisical sent") + } +} + +// The key is sent exactly once per session, because unwrapping it costs a KMS round trip. Every poll +// after the first answers with an empty sessionKey, and the cached copy has to be carried onto the +// refreshed entry. Getting this backwards silently stops all logging after the first poll, which is why +// it has a test of its own. +func TestACachedKeySurvivesAResolveThatOmitsIt(t *testing.T) { + held := &activityGrant{sessionID: "s1", projectID: "proj-1", key: aKey(9)} + + got := toActivityGrant("s1", enabledGrantWire(""), held) + + if got == nil { + t.Fatal("the grant was cleared when the response carried no key; logging would stop after one poll") + } + if string(got.key) != string(held.key) { + t.Fatal("the cached key was not carried forward") + } +} + +func TestNoKeyAndNoCachedCopyMeansNoRecording(t *testing.T) { + // The proxy said it had no key and was sent none, so there is nothing to seal with. Recording + // anything here would produce chunks nobody can ever open. + if got := toActivityGrant("s1", enabledGrantWire(""), nil); got != nil { + t.Fatal("a grant was built with no key at all") + } +} + +func TestActivityBeingOffClearsAnyCachedGrant(t *testing.T) { + held := &activityGrant{sessionID: "s1", projectID: "proj-1", key: aKey(9)} + + // An admin switching logging off has to reach a running proxy on its next poll. + if got := toActivityGrant("s1", api.AgentVaultActivityGrant{Enabled: false}, held); got != nil { + t.Fatal("the proxy kept recording after logging was switched off") + } +} + +func TestAnUnusableKeyIsRefusedRatherThanUsed(t *testing.T) { + for _, wire := range []struct { + name string + key string + }{ + {"not base64", "!!!!not base64!!!!"}, + {"too short", base64.StdEncoding.EncodeToString(make([]byte, 16))}, + {"too long", base64.StdEncoding.EncodeToString(make([]byte, 64))}, + } { + if got := toActivityGrant("s1", enabledGrantWire(wire.key), nil); got != nil { + t.Fatalf("a key that is %s was accepted", wire.name) + } + } +} + +func TestAGrantWithoutAProjectIsRefused(t *testing.T) { + // The project id is part of the AAD, so a chunk sealed without it could never be opened. + wire := api.AgentVaultActivityGrant{Enabled: true, SessionKey: base64.StdEncoding.EncodeToString(aKey(7))} + if got := toActivityGrant("s1", wire, nil); got != nil { + t.Fatal("a grant was built with no project named") + } +} diff --git a/packages/agentvault/activity_ship.go b/packages/agentvault/activity_ship.go new file mode 100644 index 000000000..edef939f7 --- /dev/null +++ b/packages/agentvault/activity_ship.go @@ -0,0 +1,104 @@ +package agentvault + +import ( + "bytes" + "context" + "errors" + "fmt" + "net/http" + "strconv" + + "github.com/Infisical/infisical-merge/packages/api" + "github.com/Infisical/infisical-merge/packages/util" + "github.com/go-resty/resty/v2" +) + +// isActivityErrorNamed matches the two named refusals the backend raises for activity. Both mean "stop +// asking for a while" rather than "this chunk is bad". +func isActivityErrorNamed(err error, name string) bool { + var apiErr *api.APIError + return errors.As(err, &apiErr) && apiErr.Name == name +} + +// isPoisonChunk is a 4xx the server will never accept: a schema failure or a chunk whose own numbers +// contradict each other. Retrying one costs the spool behind it, so it is dropped instead. +// +// 401, 404 and the two named refusals are handled before this is reached, and 429 is deliberately not +// here: it is a "later", not a refusal. +func isPoisonChunk(err error) bool { + var apiErr *api.APIError + if !errors.As(err, &apiErr) { + return false + } + if apiErr.StatusCode == http.StatusTooManyRequests { + return false + } + return apiErr.StatusCode >= 400 && apiErr.StatusCode < 500 +} + +// activityShipperClient carries three clients because the three calls want three different policies. +type activityShipperClient struct { + steady *resty.Client + final *resty.Client + put *http.Client +} + +func newActivityShipper(proxyToken func() string) (*activityShipperClient, error) { + // No retries on the steady path: a failed create is retried by the next tick, which is the same + // backoff with none of the risk of piling requests onto a struggling control plane. + steady, err := util.GetRestyClientWithPolicy(util.RetryPolicy{}) + if err != nil { + return nil, err + } + steady.SetAuthToken(proxyToken()).SetTimeout(controlPlaneTimeout) + + // Shutdown gets one retry and a short deadline: there is no next tick, and creating a chunk is + // idempotent by chunk id, so a replay cannot double-write. + finalPolicy := util.BestEffortRetryPolicy() + finalPolicy.ReplaySafe = true + final, err := util.GetRestyClientWithPolicy(finalPolicy) + if err != nil { + return nil, err + } + final.SetAuthToken(proxyToken()).SetTimeout(activityFinalTimeout) + + return &activityShipperClient{ + steady: steady, + final: final, + // Deliberately not a resty client: this one talks to the customer's bucket with a presigned URL + // and must never carry the Infisical Authorization header those two set. + put: &http.Client{Timeout: activityPutTimeout, Transport: http.DefaultTransport.(*http.Transport).Clone()}, + }, nil +} + +func (c *activityShipperClient) createChunk(final bool, sessionID string, req api.CreateAgentVaultActivityChunkRequest) (api.CreateAgentVaultActivityChunkResponse, error) { + client := c.steady + if final { + client = c.final + } + return api.CallCreateAgentVaultActivityChunk(client, sessionID, req) +} + +func (c *activityShipperClient) putObject(ctx context.Context, url string, ciphertext []byte) error { + req, err := http.NewRequestWithContext(ctx, http.MethodPut, url, bytes.NewReader(ciphertext)) + if err != nil { + return err + } + // The presign signs Content-Length in, so it has to match the body exactly. + req.ContentLength = int64(len(ciphertext)) + req.Header.Set("Content-Type", "application/octet-stream") + req.Header.Set("Content-Length", strconv.Itoa(len(ciphertext))) + + res, err := c.put.Do(req) + if err != nil { + return err + } + defer res.Body.Close() + + if res.StatusCode < 200 || res.StatusCode >= 300 { + // The body can carry an S3 error document; the status is enough to decide, and the URL is signed + // so it never goes in a log line. + return fmt.Errorf("agent-vault: the bucket refused the upload with status %d", res.StatusCode) + } + return nil +} diff --git a/packages/agentvault/activity_ship_test.go b/packages/agentvault/activity_ship_test.go new file mode 100644 index 000000000..6e9fb19ce --- /dev/null +++ b/packages/agentvault/activity_ship_test.go @@ -0,0 +1,121 @@ +package agentvault + +import ( + "context" + "net/http" + "net/http/httptest" + "strconv" + "strings" + "sync" + "testing" + + "github.com/Infisical/infisical-merge/packages/api" + "github.com/Infisical/infisical-merge/packages/config" +) + +// These swap the config.INFISICAL_URL global, so they cannot run in parallel with each other or with the +// other API-level tests in this package. + +func TestTheChunkPostCarriesTheProxyTokenAndTheBucketPutDoesNot(t *testing.T) { + var ( + mu sync.Mutex + postAuth string + putAuth string + putLength string + putType string + putBody []byte + postedPath string + ) + + bucket := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + mu.Lock() + putAuth = r.Header.Get("Authorization") + putLength = r.Header.Get("Content-Length") + putType = r.Header.Get("Content-Type") + buf := make([]byte, r.ContentLength) + _, _ = r.Body.Read(buf) + putBody = buf + mu.Unlock() + w.WriteHeader(http.StatusOK) + })) + defer bucket.Close() + + infisical := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + mu.Lock() + postAuth = r.Header.Get("Authorization") + postedPath = r.URL.Path + mu.Unlock() + w.Header().Set("Content-Type", "application/json") + _, _ = w.Write([]byte(`{"chunkId":"01K5ABCDEFGHJKMNPQRSTVWXYZ","uploadUrl":"` + bucket.URL + `/object","expiresInSeconds":300}`)) + })) + defer infisical.Close() + + old := config.INFISICAL_URL + config.INFISICAL_URL = infisical.URL + "/api" + defer func() { config.INFISICAL_URL = old }() + + shipper, err := newActivityShipper(func() string { return "proxy-token" }) + if err != nil { + t.Fatal(err) + } + + ciphertext := []byte("sealed-bytes") + res, err := shipper.createChunk(false, "sess-1", api.CreateAgentVaultActivityChunkRequest{ + ChunkID: "01K5ABCDEFGHJKMNPQRSTVWXYZ", + RecordCount: 1, + CiphertextBytes: len(ciphertext), + }) + if err != nil { + t.Fatal(err) + } + if err := shipper.putObject(context.Background(), res.UploadURL, ciphertext); err != nil { + t.Fatal(err) + } + + mu.Lock() + defer mu.Unlock() + + if postAuth != "Bearer proxy-token" { + t.Fatalf("Infisical saw Authorization %q", postAuth) + } + if postedPath != "/api/v1/agent-vault/proxy/sessions/sess-1/activity/chunks" { + t.Fatalf("posted to %q", postedPath) + } + // The presigned URL is itself the authorization. Sending the proxy's bearer token to a customer's + // bucket would hand a third party a working Infisical credential. + if putAuth != "" { + t.Fatalf("the bucket saw an Authorization header: %q", putAuth) + } + // The presign signs Content-Length in, so a mismatch is refused by S3. + if putLength != strconv.Itoa(len(ciphertext)) { + t.Fatalf("the upload declared Content-Length %q for %d bytes", putLength, len(ciphertext)) + } + if putType != "application/octet-stream" { + t.Fatalf("the upload declared Content-Type %q", putType) + } + if string(putBody) != string(ciphertext) { + t.Fatalf("the bucket received %q", string(putBody)) + } +} + +func TestABucketRefusalIsAnErrorThatNamesNoURL(t *testing.T) { + bucket := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { + w.WriteHeader(http.StatusForbidden) + _, _ = w.Write([]byte("AccessDenied")) + })) + defer bucket.Close() + + shipper, err := newActivityShipper(func() string { return "proxy-token" }) + if err != nil { + t.Fatal(err) + } + + err = shipper.putObject(context.Background(), bucket.URL+"/object?X-Amz-Signature=secret", []byte("bytes")) + if err == nil { + t.Fatal("a 403 from the bucket was treated as a successful upload") + } + // A presigned URL carries a working signature, so it must never reach a log line. + if got := err.Error(); strings.Contains(got, "X-Amz-Signature") || strings.Contains(got, bucket.URL) { + t.Fatalf("the error names the signed url: %q", got) + } +} diff --git a/packages/agentvault/activity_spool.go b/packages/agentvault/activity_spool.go new file mode 100644 index 000000000..772b6088a --- /dev/null +++ b/packages/agentvault/activity_spool.go @@ -0,0 +1,153 @@ +package agentvault + +import ( + "encoding/json" + "time" + + "github.com/Infisical/infisical-merge/packages/api" +) + +// activityRecord is one request that reached forwardHTTP. Metadata only: no headers, because a request +// header carries the injected credential, and no bodies, because LLM traffic is orders of magnitude +// larger than this. The path never carries a query string, which the proxy gets for free by building it +// from r.URL.EscapedPath(). +type activityRecord struct { + Ts string `json:"ts"` + Seq uint64 `json:"seq"` + ProxyID string `json:"proxyId"` + Method string `json:"method"` + Host string `json:"host"` + Port string `json:"port"` + Path string `json:"path"` + Status int `json:"status"` + Decision string `json:"decision"` + Service *string `json:"service"` + AccessBundle *string `json:"accessBundle"` +} + +// activityRing is a bounded FIFO that overwrites its oldest entry when full and counts what it lost, so a +// burst costs the oldest records rather than the newest and the gap is visible in the timeline. +type activityRing struct { + buf []activityRecord + head int + n int + dropped uint64 +} + +func newActivityRing(capacity int) activityRing { + return activityRing{buf: make([]activityRecord, capacity)} +} + +func (r *activityRing) len() int { return r.n } + +func (r *activityRing) push(rec activityRecord) (evicted bool) { + capacity := len(r.buf) + if capacity == 0 { + r.dropped++ + return true + } + if r.n == capacity { + r.buf[r.head] = rec + r.head = (r.head + 1) % capacity + r.dropped++ + return true + } + r.buf[(r.head+r.n)%capacity] = rec + r.n++ + return false +} + +// drain removes up to max records, oldest first. The caller seals one chunk per call and loops until the +// ring is empty, which is what keeps a chunk inside the server's recordCount limit even when a slow tick +// let the ring grow past it. +func (r *activityRing) drain(max int) []activityRecord { + if r.n == 0 || max <= 0 { + return nil + } + if max > r.n { + max = r.n + } + out := make([]activityRecord, max) + for i := 0; i < max; i++ { + out[i] = r.buf[(r.head+i)%len(r.buf)] + } + r.head = (r.head + max) % len(r.buf) + r.n -= max + return out +} + +// takeDropped hands the running drop count to the next chunk and resets it, so each gap is reported once. +func (r *activityRing) takeDropped() uint64 { + dropped := r.dropped + r.dropped = 0 + return dropped +} + +// sealedChunk is ciphertext waiting for its two-step delivery: POST the metadata to Infisical for a +// presigned URL, then PUT the bytes to the customer's bucket. +type sealedChunk struct { + meta api.CreateAgentVaultActivityChunkRequest + ciphertext []byte + // Empty until a POST succeeds, and cleared again on any PUT failure so the next tick re-POSTs the same + // chunk id and the server replays it idempotently. + uploadURL string + urlExpires time.Time +} + +// activitySpool is one session's buffer on this proxy. proxyID is constant for the process, so it lives on +// the log rather than here. +type activitySpool struct { + sessionID string + projectID string + key []byte + + ring activityRing + nextSeq uint64 + + pending []*sealedChunk + + lastRecordAt time.Time + lastFlushAt time.Time +} + +func newActivitySpool(g *activityGrant, now time.Time) *activitySpool { + return &activitySpool{ + sessionID: g.sessionID, + projectID: g.projectID, + key: g.key, + ring: newActivityRing(activitySpoolCapacity), + lastRecordAt: now, + lastFlushAt: now, + } +} + +// sealSlice turns one slice of records into a sealed chunk ready to ship. +func (s *activitySpool) sealSlice(proxyID string, records []activityRecord, dropped uint64, now time.Time) (*sealedChunk, error) { + plaintext, err := json.Marshal(records) + if err != nil { + return nil, err + } + + chunkID := newActivityChunkID(now) + aad := buildActivityAAD(s.projectID, s.sessionID, proxyID, chunkID) + ciphertext, iv, err := sealActivity(s.key, plaintext, aad) + if err != nil { + return nil, err + } + + first, last := records[0], records[len(records)-1] + return &sealedChunk{ + meta: api.CreateAgentVaultActivityChunkRequest{ + ChunkID: chunkID, + StartedAt: first.Ts, + EndedAt: last.Ts, + FirstSeq: first.Seq, + LastSeq: last.Seq, + RecordCount: len(records), + DroppedCount: dropped, + CiphertextBytes: len(ciphertext), + IV: encodeActivityIV(iv), + }, + ciphertext: ciphertext, + }, nil +} diff --git a/packages/agentvault/activity_test.go b/packages/agentvault/activity_test.go new file mode 100644 index 000000000..28c69e68a --- /dev/null +++ b/packages/agentvault/activity_test.go @@ -0,0 +1,687 @@ +package agentvault + +import ( + "context" + "errors" + "fmt" + "net/http" + "sync" + "testing" + "time" + + "github.com/Infisical/infisical-merge/packages/api" +) + +type shipperCall struct { + kind string // "post" or "put" + sessionID string + chunkID string + url string + bytes int + body []byte + final bool +} + +type scriptedResult struct { + url string + err error +} + +// fakeShipper records every call in order and pops scripted outcomes, so a test asserts both what was +// delivered and in which order the two steps ran. +type fakeShipper struct { + mu sync.Mutex + + calls []shipperCall + + postResults []scriptedResult + putResults []error + + postDefault scriptedResult + putDefault error + + nextURL int +} + +func (f *fakeShipper) createChunk(final bool, sessionID string, req api.CreateAgentVaultActivityChunkRequest) (api.CreateAgentVaultActivityChunkResponse, error) { + f.mu.Lock() + defer f.mu.Unlock() + + f.calls = append(f.calls, shipperCall{kind: "post", sessionID: sessionID, chunkID: req.ChunkID, bytes: req.CiphertextBytes, final: final}) + + result := f.postDefault + if len(f.postResults) > 0 { + result = f.postResults[0] + f.postResults = f.postResults[1:] + } + if result.err != nil { + return api.CreateAgentVaultActivityChunkResponse{}, result.err + } + url := result.url + if url == "" { + f.nextURL++ + url = fmt.Sprintf("https://bucket.example/put/%d", f.nextURL) + } + return api.CreateAgentVaultActivityChunkResponse{ChunkID: req.ChunkID, UploadURL: url, ExpiresInSeconds: 300}, nil +} + +func (f *fakeShipper) putObject(_ context.Context, url string, ciphertext []byte) error { + f.mu.Lock() + defer f.mu.Unlock() + + f.calls = append(f.calls, shipperCall{kind: "put", url: url, bytes: len(ciphertext), body: append([]byte(nil), ciphertext...)}) + + if len(f.putResults) > 0 { + err := f.putResults[0] + f.putResults = f.putResults[1:] + return err + } + return f.putDefault +} + +func (f *fakeShipper) kinds() []string { + f.mu.Lock() + defer f.mu.Unlock() + out := make([]string, len(f.calls)) + for i, call := range f.calls { + out[i] = call.kind + } + return out +} + +func (f *fakeShipper) posts() []shipperCall { + f.mu.Lock() + defer f.mu.Unlock() + var out []shipperCall + for _, call := range f.calls { + if call.kind == "post" { + out = append(out, call) + } + } + return out +} + +func (f *fakeShipper) puts() []shipperCall { + f.mu.Lock() + defer f.mu.Unlock() + var out []shipperCall + for _, call := range f.calls { + if call.kind == "put" { + out = append(out, call) + } + } + return out +} + +func apiErr(status int, name string) error { + return &api.APIError{StatusCode: status, Name: name, Operation: "CallCreateAgentVaultActivityChunk"} +} + +func testGrant(sessionID string) *activityGrant { + return &activityGrant{sessionID: sessionID, projectID: "proj-1", key: make([]byte, 32)} +} + +// newTestLog fixes the clock so flush eligibility and the pause window are decided, not raced. +// +// tick is what the run loop does once an interval: advance, then flush. A spool with only a few records +// is deliberately not due until an interval has passed since its last flush, so that a size-triggered +// wake-up for one busy session does not drag every quiet session into an early, billable flush. Calling +// flushAll without advancing therefore ships nothing, which is correct rather than a bug to work around. +func newTestLog(shipper activityShipper) (log *activityLog, advance func(time.Duration), tick func()) { + log = newActivityLog("proxy-1", shipper) + now := time.Date(2026, 9, 16, 10, 0, 0, 0, time.UTC) + var mu sync.Mutex + log.now = func() time.Time { + mu.Lock() + defer mu.Unlock() + return now + } + advance = func(d time.Duration) { + mu.Lock() + now = now.Add(d) + mu.Unlock() + } + tick = func() { + advance(activityFlushInterval) + log.flushAll(context.Background(), false) + } + return log, advance, tick +} + +func aRecord(host string) activityRecord { + return activityRecord{Method: "GET", Host: host, Port: "443", Path: "/zen", Status: 200, Decision: decisionPassthrough} +} + +func TestRecordingIsANoOpWithoutALogOrAGrant(t *testing.T) { + var nilLog *activityLog + nilLog.record(testGrant("s1"), aRecord("api.github.com")) // must not panic + + log, _, _ := newTestLog(&fakeShipper{}) + log.record(nil, aRecord("api.github.com")) + if len(log.spools) != 0 { + t.Fatal("a nil grant created a spool; logging-off must cost nothing") + } +} + +func TestTheRingDropsTheOldestAndCountsIt(t *testing.T) { + ring := newActivityRing(3) + for i := 0; i < 5; i++ { + ring.push(activityRecord{Seq: uint64(i)}) + } + + if ring.len() != 3 { + t.Fatalf("ring holds %d, capacity is 3", ring.len()) + } + if ring.dropped != 2 { + t.Fatalf("ring counted %d drops, expected 2", ring.dropped) + } + + drained := ring.drain(10) + if len(drained) != 3 || drained[0].Seq != 2 || drained[2].Seq != 4 { + t.Fatalf("expected the three newest records 2,3,4; got %+v", drained) + } +} + +func TestTheRingDrainsInSlicesTheServerAccepts(t *testing.T) { + ring := newActivityRing(activitySpoolCapacity) + for i := 0; i < 2500; i++ { + ring.push(activityRecord{Seq: uint64(i)}) + } + + // The ring holds five flushes of headroom but the server refuses a chunk over activityFlushRecords, + // which it answers with a 422 the proxy then treats as poison. Slicing is what prevents that loss. + var slices int + for ring.len() > 0 { + got := ring.drain(activityFlushRecords) + if len(got) > activityFlushRecords { + t.Fatalf("a slice held %d records, the server's limit is %d", len(got), activityFlushRecords) + } + slices++ + } + if slices != 3 { + t.Fatalf("2500 records drained in %d slices, expected 3", slices) + } +} + +func TestTheDropCountIsReportedOnceAndRidesTheFirstChunk(t *testing.T) { + shipper := &fakeShipper{} + log, _, _ := newTestLog(shipper) + grant := testGrant("s1") + + // Overfill so the ring evicts, then flush: the gap must be counted on the first chunk only. + for i := 0; i < activitySpoolCapacity+50; i++ { + log.record(grant, aRecord("api.github.com")) + } + log.flushAll(context.Background(), true) + + posts := shipper.posts() + if len(posts) == 0 { + t.Fatal("nothing was shipped") + } + if log.spools["s1"].ring.dropped != 0 { + t.Fatal("the drop count was not reset after being reported") + } +} + +func TestASequenceNumberIsConsumedEvenWhenARecordIsDropped(t *testing.T) { + log, _, _ := newTestLog(&fakeShipper{}) + grant := testGrant("s1") + + for i := 0; i < activitySpoolCapacity+10; i++ { + log.record(grant, aRecord("api.github.com")) + } + + // A chunk's firstSeq is what reveals the hole, so the counter must not compact over dropped records. + if got := log.spools["s1"].nextSeq; got != uint64(activitySpoolCapacity+10) { + t.Fatalf("nextSeq is %d after %d records; drops must still consume a number", got, activitySpoolCapacity+10) + } +} + +func TestTheProxyWideFuseDropsTheNewest(t *testing.T) { + log, _, _ := newTestLog(&fakeShipper{}) + + // Spread across enough spools to pass the total cap without any one ring filling. + for i := 0; i < activityTotalCapacity/activitySpoolCapacity+2; i++ { + grant := testGrant(fmt.Sprintf("s%d", i)) + for j := 0; j < activitySpoolCapacity; j++ { + log.record(grant, aRecord("api.github.com")) + } + } + + if log.total > activityTotalCapacity { + t.Fatalf("the proxy holds %d records, past the %d fuse", log.total, activityTotalCapacity) + } +} + +func TestAChunkIsPostedBeforeItIsUploaded(t *testing.T) { + shipper := &fakeShipper{} + log, _, _ := newTestLog(shipper) + + log.record(testGrant("s1"), aRecord("api.github.com")) + log.flushAll(context.Background(), true) + + // The row lands before the object, so a failed upload is a visible gap rather than a silent one. + if got := shipper.kinds(); len(got) != 2 || got[0] != "post" || got[1] != "put" { + t.Fatalf("call order was %v, expected post then put", got) + } + if posts := shipper.posts(); posts[0].sessionID != "s1" { + t.Fatalf("posted under session %q", posts[0].sessionID) + } + if puts := shipper.puts(); puts[0].bytes != shipper.posts()[0].bytes { + t.Fatalf("uploaded %d bytes after declaring %d", puts[0].bytes, shipper.posts()[0].bytes) + } +} + +func TestAFailedUploadRePostsTheSameChunkID(t *testing.T) { + shipper := &fakeShipper{putResults: []error{errors.New("connection reset")}} + log, _, tick := newTestLog(shipper) + + log.record(testGrant("s1"), aRecord("api.github.com")) + tick() + tick() + + posts := shipper.posts() + if len(posts) != 2 { + t.Fatalf("expected the chunk to be re-posted, saw %d posts", len(posts)) + } + // The server replays the same id idempotently, which is what makes the retry safe. + if posts[0].chunkID != posts[1].chunkID { + t.Fatalf("re-post used a different chunk id: %q then %q", posts[0].chunkID, posts[1].chunkID) + } + if len(shipper.puts()) != 2 { + t.Fatalf("expected a second upload attempt, saw %d", len(shipper.puts())) + } +} + +func TestASessionThatIsGoneIsDropped(t *testing.T) { + shipper := &fakeShipper{postResults: []scriptedResult{{err: apiErr(http.StatusNotFound, "")}}} + log, _, tick := newTestLog(shipper) + + log.record(testGrant("s1"), aRecord("api.github.com")) + tick() + + if _, ok := log.spools["s1"]; ok { + t.Fatal("the spool survived a session the server no longer knows") + } + + tick() + if len(shipper.posts()) != 1 { + t.Fatal("the dropped spool was retried") + } +} + +func TestARejectedProxyTokenKeepsEverything(t *testing.T) { + shipper := &fakeShipper{postDefault: scriptedResult{err: apiErr(http.StatusUnauthorized, proxyTokenRejectedName)}} + log, _, tick := newTestLog(shipper) + + log.record(testGrant("s1"), aRecord("api.github.com")) + tick() + + // The poll loop exits within two heartbeats; until then nothing is thrown away. + spool, ok := log.spools["s1"] + if !ok { + t.Fatal("the spool was dropped on a rejected proxy token") + } + if len(spool.pending) != 1 { + t.Fatalf("the sealed chunk was not kept, pending holds %d", len(spool.pending)) + } +} + +func TestTheCeilingPausesTheWholeProxyAndLiftsAfterTheBackoff(t *testing.T) { + shipper := &fakeShipper{postResults: []scriptedResult{{err: apiErr(400, activityCeilingReachedName)}}} + log, advance, tick := newTestLog(shipper) + + log.record(testGrant("s1"), aRecord("api.github.com")) + tick() + + if paused, reason := log.paused(); !paused || reason != activityCeilingReachedName { + t.Fatalf("expected a ceiling pause, got paused=%v reason=%q", paused, reason) + } + + // The ceiling is per organization, so another session on this proxy is paused too. + log.record(testGrant("s2"), aRecord("api.github.com")) + tick() + if len(shipper.posts()) != 1 { + t.Fatalf("a second session posted while paused; the pause is proxy-wide") + } + + // Ops may raise the limit within the window, so the sealed chunks are still there when it lifts. + advance(activityPauseBackoff + time.Second) + log.flushAll(context.Background(), false) + if len(shipper.posts()) < 2 { + t.Fatal("nothing was retried after the pause lifted") + } +} + +func TestBeingSwitchedOffPausesRatherThanDiscards(t *testing.T) { + shipper := &fakeShipper{postResults: []scriptedResult{{err: apiErr(400, activityDisabledName)}}} + log, _, tick := newTestLog(shipper) + + log.record(testGrant("s1"), aRecord("api.github.com")) + tick() + + if paused, reason := log.paused(); !paused || reason != activityDisabledName { + t.Fatalf("expected a disabled pause, got paused=%v reason=%q", paused, reason) + } + if len(log.spools["s1"].pending) != 1 { + t.Fatal("the sealed chunk was discarded when logging was switched off") + } +} + +func TestRecordsArePausedAsCountedGapsNotSilentLosses(t *testing.T) { + shipper := &fakeShipper{postResults: []scriptedResult{{err: apiErr(400, activityCeilingReachedName)}}} + log, _, tick := newTestLog(shipper) + grant := testGrant("s1") + + log.record(grant, aRecord("api.github.com")) + tick() + + before := log.spools["s1"].ring.dropped + for i := 0; i < 5; i++ { + log.record(grant, aRecord("api.github.com")) + } + if got := log.spools["s1"].ring.dropped - before; got != 5 { + t.Fatalf("%d records were counted as dropped while paused, expected 5", got) + } +} + +func TestAPoisonChunkIsDroppedAndTheRestShip(t *testing.T) { + shipper := &fakeShipper{postResults: []scriptedResult{{err: apiErr(http.StatusUnprocessableEntity, "")}}} + log, _, tick := newTestLog(shipper) + + log.record(testGrant("s1"), aRecord("api.github.com")) + tick() + + if len(log.spools["s1"].pending) != 0 { + t.Fatal("a chunk the server called malformed was kept; it would block the spool behind it") + } + + log.record(testGrant("s1"), aRecord("api.github.com")) + tick() + if len(shipper.puts()) != 1 { + t.Fatalf("the next chunk did not ship after a poison one, %d uploads", len(shipper.puts())) + } +} + +func TestARateLimitIsRetriedRatherThanTreatedAsPoison(t *testing.T) { + shipper := &fakeShipper{postResults: []scriptedResult{{err: apiErr(http.StatusTooManyRequests, "")}}} + log, _, tick := newTestLog(shipper) + + log.record(testGrant("s1"), aRecord("api.github.com")) + tick() + + if len(log.spools["s1"].pending) != 1 { + t.Fatal("a 429 discarded the chunk; it means later, not never") + } + + tick() + if len(shipper.puts()) != 1 { + t.Fatal("the chunk did not ship on the retry") + } +} + +func TestThePendingCapEvictsTheOldestAndCountsIt(t *testing.T) { + shipper := &fakeShipper{postDefault: scriptedResult{err: errors.New("infisical unreachable")}} + log, _, tick := newTestLog(shipper) + grant := testGrant("s1") + + for i := 0; i < activityPendingChunks+3; i++ { + log.record(grant, aRecord("api.github.com")) + tick() + } + + spool := log.spools["s1"] + if len(spool.pending) > activityPendingChunks { + t.Fatalf("pending holds %d chunks, the cap is %d", len(spool.pending), activityPendingChunks) + } + if spool.ring.dropped == 0 { + t.Fatal("evicted chunks were not counted as dropped records") + } +} + +func TestTheTickBreakerStopsHammeringADeadBucket(t *testing.T) { + shipper := &fakeShipper{putDefault: errors.New("i/o timeout")} + log, _, tick := newTestLog(shipper) + + for i := 0; i < 5; i++ { + log.record(testGrant(fmt.Sprintf("s%d", i)), aRecord("api.github.com")) + } + tick() + + // After the first upload fails, the rest of the tick seals but skips both calls, so five spools + // against blocked egress cost one timeout rather than five. + if got := len(shipper.puts()); got != 1 { + t.Fatalf("%d uploads were attempted in one tick after the first failed", got) + } + if got := len(shipper.posts()); got != 1 { + t.Fatalf("%d rows were written for objects that could not be uploaded", got) + } + // Nothing is lost: every spool sealed its records and holds them. + for i := 0; i < 5; i++ { + if len(log.spools[fmt.Sprintf("s%d", i)].pending) == 0 { + t.Fatalf("spool s%d sealed nothing during the outage", i) + } + } +} + +func TestReachingTheSliceSizeWakesTheLoopOnce(t *testing.T) { + log, _, _ := newTestLog(&fakeShipper{}) + grant := testGrant("s1") + + for i := 0; i < activityFlushRecords*2; i++ { + log.record(grant, aRecord("api.github.com")) + } + + // A buffered channel of one: the loop coalesces a burst into a single wake-up. + if len(log.wake) != 1 { + t.Fatalf("the wake channel holds %d, expected exactly one pending wake-up", len(log.wake)) + } +} + +func TestAnIdleSpoolIsForgotten(t *testing.T) { + shipper := &fakeShipper{} + log, advance, tick := newTestLog(shipper) + + log.record(testGrant("s1"), aRecord("api.github.com")) + tick() + + advance(activityIdleClose + time.Minute) + log.flushAll(context.Background(), false) + + if _, ok := log.spools["s1"]; ok { + t.Fatal("an idle spool was kept; a long-lived proxy would grow without bound") + } +} + +func TestIdleCloseOutlastsTheSessionCacheTTL(t *testing.T) { + // A session evicted from the cache stops producing records but must still get its final flush. + if activityIdleClose <= sessionInactiveTTL { + t.Fatalf("idle close (%s) must outlast the session cache TTL (%s)", activityIdleClose, sessionInactiveTTL) + } +} + +func TestCloseFlushesAndThenStopsRecording(t *testing.T) { + shipper := &fakeShipper{} + log, _, _ := newTestLog(shipper) + grant := testGrant("s1") + + log.record(grant, aRecord("api.github.com")) + log.close(context.Background()) + + if len(shipper.puts()) != 1 { + t.Fatalf("shutdown shipped %d chunks, expected the buffered one", len(shipper.puts())) + } + if !shipper.posts()[0].final { + t.Fatal("the shutdown flush did not use the short-deadline client") + } + + log.record(grant, aRecord("api.github.com")) + log.flushAll(context.Background(), true) + if len(shipper.puts()) != 1 { + t.Fatal("a record was accepted after close") + } +} + +func TestABlockedHostIsStillRecorded(t *testing.T) { + shipper := &fakeShipper{} + log, _, _ := newTestLog(shipper) + + // The point of logging every request rather than only brokered ones: under the default any-host + // policy an agent exfiltrating to an unconfigured host is passthrough traffic, and under bundle-hosts + // the refusal is the single most security-relevant line in the timeline. + log.record(testGrant("s1"), activityRecord{ + Method: "POST", Host: "evil.example", Port: "443", Path: "/collect", Status: 403, Decision: decisionBlocked, + }) + log.flushAll(context.Background(), true) + + if len(shipper.puts()) != 1 { + t.Fatal("a blocked request was not recorded") + } + spool := log.spools["s1"] + if spool == nil { + t.Fatal("no spool was created for a blocked request") + } +} + +func TestOneSpoolPerSession(t *testing.T) { + log, _, _ := newTestLog(&fakeShipper{}) + + log.record(testGrant("s1"), aRecord("api.github.com")) + log.record(testGrant("s2"), aRecord("api.github.com")) + log.record(testGrant("s1"), aRecord("api.anthropic.com")) + + if len(log.spools) != 2 { + t.Fatalf("%d spools for two sessions", len(log.spools)) + } + if log.spools["s1"].ring.len() != 2 { + t.Fatalf("session one holds %d records, expected 2", log.spools["s1"].ring.len()) + } +} + +func TestEveryRecordCarriesTheProxyAndATimestamp(t *testing.T) { + log, _, _ := newTestLog(&fakeShipper{}) + log.record(testGrant("s1"), aRecord("api.github.com")) + + got := log.spools["s1"].ring.drain(1)[0] + if got.ProxyID != "proxy-1" { + t.Fatalf("record names proxy %q", got.ProxyID) + } + if _, err := time.Parse(time.RFC3339Nano, got.Ts); err != nil { + t.Fatalf("timestamp %q is not RFC3339Nano: %v", got.Ts, err) + } + if got.Seq != 0 { + t.Fatalf("the first record has seq %d, expected 0", got.Seq) + } +} + +func TestSequenceNumbersSurviveASpoolBeingForgotten(t *testing.T) { + shipper := &fakeShipper{} + log, advance, tick := newTestLog(shipper) + grant := testGrant("s1") + + log.record(grant, aRecord("api.github.com")) + log.record(grant, aRecord("api.github.com")) + tick() + + // Idle long enough to be forgotten, then used again. Two records with the same (proxyId, seq) for + // one session would make the log ambiguous for anyone correlating it. + advance(activityIdleClose + time.Minute) + log.flushAll(context.Background(), false) + if _, ok := log.spools["s1"]; ok { + t.Fatal("the idle spool was not forgotten, so this test proves nothing") + } + + log.record(grant, aRecord("api.github.com")) + got := log.spools["s1"].ring.drain(1)[0] + if got.Seq != 2 { + t.Fatalf("seq restarted at %d after the spool was rebuilt, expected it to continue at 2", got.Seq) + } +} + +func TestASessionThatIsGoneDoesNotReserveItsSequenceNumbers(t *testing.T) { + shipper := &fakeShipper{postResults: []scriptedResult{{err: apiErr(http.StatusNotFound, "")}}} + log, _, tick := newTestLog(shipper) + + log.record(testGrant("s1"), aRecord("api.github.com")) + tick() + + if _, ok := log.seqBySession["s1"]; ok { + t.Fatal("a session the server has forgotten is still holding a sequence number") + } +} + +func TestRecordsLostToASealFailureAreStillCounted(t *testing.T) { + shipper := &fakeShipper{} + log, _, tick := newTestLog(shipper) + + // A key the AES constructor rejects, so sealing fails for every slice. + grant := &activityGrant{sessionID: "s1", projectID: "proj-1", key: make([]byte, 7)} + for i := 0; i < 3; i++ { + log.record(grant, aRecord("api.github.com")) + } + tick() + + spool := log.spools["s1"] + if spool == nil { + t.Fatal("the spool disappeared") + } + if spool.ring.dropped != 3 { + t.Fatalf("%d records were counted as dropped after a seal failure, expected 3", spool.ring.dropped) + } + if len(shipper.posts()) != 0 { + t.Fatal("a chunk was shipped despite the seal failing") + } +} + +func TestAnUnreachableControlPlaneStopsTheTickAfterOneTimeout(t *testing.T) { + shipper := &fakeShipper{postDefault: scriptedResult{err: errors.New("i/o timeout")}} + log, _, tick := newTestLog(shipper) + + for i := 0; i < 5; i++ { + log.record(testGrant(fmt.Sprintf("s%d", i)), aRecord("api.github.com")) + } + tick() + + // Without a breaker this costs five control-plane timeouts in series, every tick. + if got := len(shipper.posts()); got != 1 { + t.Fatalf("%d chunk POSTs were attempted in one tick after the first timed out", got) + } + for i := 0; i < 5; i++ { + if len(log.spools[fmt.Sprintf("s%d", i)].pending) == 0 { + t.Fatalf("spool s%d sealed nothing during the outage", i) + } + } +} + +func TestShutdownDoesNotRaceTheRunLoop(t *testing.T) { + shipper := &fakeShipper{} + log, _, _ := newTestLog(shipper) + log.now = time.Now + + stop := make(chan struct{}) + go log.run(stop) + + grant := testGrant("s1") + done := make(chan struct{}) + go func() { + defer close(done) + for i := 0; i < 2000; i++ { + log.record(grant, aRecord("api.github.com")) + } + }() + + <-done + close(stop) + // Shutdown flushes from this goroutine while the run loop may still be inside one of its own. The + // race detector is what makes this test worth having. + log.close(context.Background()) + + if len(shipper.puts()) == 0 { + t.Fatal("shutdown shipped nothing") + } + for _, call := range shipper.puts() { + if call.bytes == 0 { + t.Fatal("an empty object was uploaded") + } + } +} diff --git a/packages/agentvault/activity_wiring_test.go b/packages/agentvault/activity_wiring_test.go new file mode 100644 index 000000000..30730c5e9 --- /dev/null +++ b/packages/agentvault/activity_wiring_test.go @@ -0,0 +1,223 @@ +package agentvault + +import ( + "bytes" + "context" + "fmt" + "net/http" + "net/http/httptest" + "net/url" + "strings" + "testing" +) + +// grantingResolver hands back one service plus an activity grant, so a request driven through the real +// dispatch path produces a real record. +type grantingResolver struct { + services []*resolvedService +} + +func (g grantingResolver) resolve(string, *activityGrant) (*resolveResult, error) { + return &resolveResult{ + SessionID: "s1", + Services: g.services, + Activity: &activityGrant{sessionID: "s1", projectID: "proj-1", key: make([]byte, 32)}, + }, nil +} + +// newRecordingProxy wires a proxy exactly as run.go does, minus the listener, so these tests exercise +// the call site in forwardHTTP rather than activityLog in isolation. +func newRecordingProxy(t *testing.T, policy string, services []*resolvedService) (*httptest.Server, *activityLog, *fakeShipper) { + t.Helper() + + shipper := &fakeShipper{} + ps := &proxyServer{transport: newUpstreamTransport()} + ps.setConfig(ProxyConfig{TrafficPolicy: policy}) + ps.cache = newSessionCache(grantingResolver{services: services}, ps.pollInterval) + ps.activity = newActivityLog("proxy-1", shipper) + + front := httptest.NewServer(http.HandlerFunc(ps.dispatch)) + t.Cleanup(front.Close) + return front, ps.activity, shipper +} + +func proxiedGet(t *testing.T, front *httptest.Server, target string) *http.Response { + t.Helper() + + proxyURL, err := url.Parse(front.URL) + if err != nil { + t.Fatal(err) + } + // The session token rides in the proxy credentials, which is how an agent presents it. + proxyURL.User = url.UserPassword("infisical", "agv_test-token") + + client := &http.Client{Transport: &http.Transport{Proxy: http.ProxyURL(proxyURL)}} + res, err := client.Get(target) + if err != nil { + t.Fatalf("request through the proxy failed: %v", err) + } + t.Cleanup(func() { _ = res.Body.Close() }) + return res +} + +func drainOneRecord(t *testing.T, log *activityLog) activityRecord { + t.Helper() + spool, ok := log.spools["s1"] + if !ok { + t.Fatal("the request produced no activity spool") + } + records := spool.ring.drain(10) + if len(records) != 1 { + t.Fatalf("expected exactly one record, got %d", len(records)) + } + return records[0] +} + +func TestAProxiedRequestIsRecorded(t *testing.T) { + upstream := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { + w.WriteHeader(http.StatusCreated) + })) + defer upstream.Close() + + front, log, _ := newRecordingProxy(t, TrafficPolicyAnyHost, nil) + res := proxiedGet(t, front, upstream.URL+"/repos/acme/web/issues") + if res.StatusCode != http.StatusCreated { + t.Fatalf("upstream status was %d", res.StatusCode) + } + + got := drainOneRecord(t, log) + if got.Method != http.MethodGet || got.Path != "/repos/acme/web/issues" || got.Status != http.StatusCreated { + t.Fatalf("record is %+v", got) + } + // Nothing in the bundle matched, so this is passthrough traffic. Recording it is the whole point: + // under the default any-host policy an agent exfiltrating to an unconfigured host looks like this. + if got.Decision != decisionPassthrough { + t.Fatalf("decision was %q, expected %q", got.Decision, decisionPassthrough) + } + if got.Service != nil || got.AccessBundle != nil { + t.Fatal("a passthrough record named a service") + } + if got.ProxyID != "proxy-1" { + t.Fatalf("record names proxy %q", got.ProxyID) + } +} + +func TestAQueryStringNeverReachesTheRecord(t *testing.T) { + upstream := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { + w.WriteHeader(http.StatusOK) + })) + defer upstream.Close() + + front, log, _ := newRecordingProxy(t, TrafficPolicyAnyHost, nil) + // Plenty of APIs put a token in the query string, so the path is built from EscapedPath() and the + // query is never seen. This is true by construction; the test is what keeps it true. + proxiedGet(t, front, upstream.URL+"/v1/thing?access_token=super-secret&sid=abc") + + got := drainOneRecord(t, log) + if got.Path != "/v1/thing" { + t.Fatalf("path is %q, a query string leaked into the record", got.Path) + } +} + +// A path substitution rewrites the request with the real credential before it goes upstream. The record +// has to carry the path the agent sent, placeholder and all, or the activity log would be the one place +// the secret the agent never sees gets written down. +func TestASubstitutedPathIsRecordedAsTheAgentSentIt(t *testing.T) { + seen := make(chan string, 1) + upstream := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + seen <- r.URL.Path + w.WriteHeader(http.StatusOK) + })) + defer upstream.Close() + + host := strings.TrimPrefix(upstream.URL, "http://") + front, log, _ := newRecordingProxy(t, TrafficPolicyAnyHost, []*resolvedService{ + policyService(host, nil, nil, nil, []substitution{subOn("__PAT__", "real-token", surfacePath)}), + }) + proxiedGet(t, front, upstream.URL+"/repos/__PAT__/issues") + + if upstreamPath := <-seen; upstreamPath != "/repos/real-token/issues" { + t.Fatalf("the upstream saw %q, so the substitution never fired and this proves nothing", upstreamPath) + } + got := drainOneRecord(t, log) + if got.Path != "/repos/__PAT__/issues" || strings.Contains(got.Path, "real-token") { + t.Fatalf("record path is %q", got.Path) + } + if got.Decision != decisionBrokered { + t.Fatalf("decision was %q, expected a substitution to count as brokered", got.Decision) + } +} + +func TestABlockedRequestIsRecordedWithItsRefusal(t *testing.T) { + // bundle-hosts with no service covering the host, and no allow list: the request is refused. + front, log, _ := newRecordingProxy(t, TrafficPolicyBundleHosts, nil) + + res := proxiedGet(t, front, "http://blocked.example/collect") + if res.StatusCode != http.StatusForbidden { + t.Fatalf("a blocked host answered %d", res.StatusCode) + } + + got := drainOneRecord(t, log) + if got.Decision != decisionBlocked || got.Status != http.StatusForbidden { + t.Fatalf("record is %+v, expected a blocked 403", got) + } + if got.Host != "blocked.example" { + t.Fatalf("record names host %q", got.Host) + } +} + +func TestRecordingSurvivesAnUnreachableUpstream(t *testing.T) { + front, log, _ := newRecordingProxy(t, TrafficPolicyAnyHost, nil) + + proxyURL, _ := url.Parse(front.URL) + proxyURL.User = url.UserPassword("infisical", "agv_test-token") + client := &http.Client{Transport: &http.Transport{Proxy: http.ProxyURL(proxyURL)}} + // Port 1 refuses immediately. + res, err := client.Get("http://127.0.0.1:1/v1/thing") + if err != nil { + t.Fatalf("the proxy did not answer: %v", err) + } + defer res.Body.Close() + + got := drainOneRecord(t, log) + if got.Decision != decisionError { + t.Fatalf("decision was %q, expected %q", got.Decision, decisionError) + } +} + +func TestAWholeRequestRoundTripsFromProxyToSealedChunk(t *testing.T) { + upstream := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { + w.WriteHeader(http.StatusOK) + })) + defer upstream.Close() + + front, log, shipper := newRecordingProxy(t, TrafficPolicyAnyHost, nil) + for i := 0; i < 3; i++ { + proxiedGet(t, front, fmt.Sprintf("%s/v1/thing/%d", upstream.URL, i)) + } + + // close is what shutdown calls: it flushes whatever is buffered within the deadline. + log.close(context.Background()) + + posts := shipper.posts() + if len(posts) != 1 { + t.Fatalf("three requests produced %d chunks, expected one", len(posts)) + } + if posts[0].sessionID != "s1" { + t.Fatalf("the chunk was filed under session %q", posts[0].sessionID) + } + if posts[0].bytes <= 0 { + t.Fatal("the chunk declared no ciphertext") + } + + puts := shipper.puts() + if len(puts) != 1 || puts[0].bytes != posts[0].bytes { + t.Fatalf("uploaded %d objects, %d bytes, for a chunk declaring %d", len(puts), puts[0].bytes, posts[0].bytes) + } + + // What leaves the proxy is ciphertext. If the records ever went out in the clear, the host and the + // path would be readable right here. + if bytes.Contains(puts[0].body, []byte("/v1/thing")) || bytes.Contains(puts[0].body, []byte("\"method\"")) { + t.Fatal("the uploaded chunk contains readable record fields; it was not sealed") + } +} diff --git a/packages/agentvault/cache.go b/packages/agentvault/cache.go index 09c401b8c..2013d0233 100644 --- a/packages/agentvault/cache.go +++ b/packages/agentvault/cache.go @@ -68,6 +68,8 @@ type sessionEntry struct { sessionID string expiresAt *time.Time services []*resolvedService + // nil when activity logging is off for this session, which is the whole of the disabled path. + activity *activityGrant lastSeen time.Time fetchedAt time.Time } @@ -79,7 +81,9 @@ func sessionKey(token string) string { } type sessionResolver interface { - resolve(sessionToken string) (*resolveResult, error) + // held is the activity grant the caller already has, or nil. Passing it lets the server skip + // re-sending a key that never changes. + resolve(sessionToken string, held *activityGrant) (*resolveResult, error) } type sessionCache struct { @@ -152,7 +156,20 @@ func isSessionGone(err error) bool { return errors.Is(err, errSessionGone) } +// get keeps the two CONNECT gates, which care only about validity, free of the activity plumbing. func (c *sessionCache) get(sessionToken string) ([]*resolvedService, error) { + services, _, err := c.lookup(sessionToken) + return services, err +} + +type cacheLookup struct { + services []*resolvedService + activity *activityGrant +} + +// lookup resolves a session token to what the request path needs: the services to match against, and the +// grant to record under. Both come from one cache entry, so the request handler never resolves. +func (c *sessionCache) lookup(sessionToken string) ([]*resolvedService, *activityGrant, error) { key := sessionKey(sessionToken) c.mu.Lock() @@ -162,7 +179,7 @@ func (c *sessionCache) get(sessionToken string) ([]*resolvedService, error) { delete(c.entries, key) delete(c.tokens, key) c.mu.Unlock() - return nil, errSessionGone + return nil, nil, errSessionGone } // Past the grace window the entry is a miss, so a stalled refresh loop cannot keep an old credential alive. if time.Since(entry.fetchedAt) > c.grace() { @@ -170,22 +187,23 @@ func (c *sessionCache) get(sessionToken string) ([]*resolvedService, error) { delete(c.tokens, key) } else { entry.lastSeen = time.Now() - svcs := entry.services + svcs, grant := entry.services, entry.activity c.mu.Unlock() - return svcs, nil + return svcs, grant, nil } } if refused, ok := c.refused[key]; ok { if time.Now().Before(refused.until) { c.mu.Unlock() - return nil, refused.err + return nil, nil, refused.err } delete(c.refused, key) } c.mu.Unlock() resolved, err, _ := c.inflight.Do(key, func() (any, error) { - result, err := c.resolver.resolve(sessionToken) + // No cached entry, so no cached key either: ask for one. + result, err := c.resolver.resolve(sessionToken, nil) if err != nil { // A rejected proxy token is remembered too: the poll loop exits after two such heartbeats, but // until then every agent request would otherwise cost a resolve. @@ -204,16 +222,18 @@ func (c *sessionCache) get(sessionToken string) ([]*resolvedService, error) { sessionID: result.SessionID, expiresAt: result.ExpiresAt, services: result.Services, + activity: result.Activity, lastSeen: time.Now(), fetchedAt: time.Now(), } c.tokens[key] = sessionToken - return result.Services, nil + return cacheLookup{services: result.Services, activity: result.Activity}, nil }) if err != nil { - return nil, err + return nil, nil, err } - return resolved.([]*resolvedService), nil + out := resolved.(cacheLookup) + return out.services, out.activity, nil } func (c *sessionCache) evictIfFullLocked() { @@ -289,7 +309,15 @@ func (c *sessionCache) refresh() { } func (c *sessionCache) refreshOne(key, token string) { - result, err := c.resolver.resolve(token) + // Tell the server whether we already hold this session's key, so it can skip the unwrap. + c.mu.Lock() + var held *activityGrant + if entry, ok := c.entries[key]; ok { + held = entry.activity + } + c.mu.Unlock() + + result, err := c.resolver.resolve(token, held) if err != nil { c.handleRefreshFailure(key, err) return @@ -301,6 +329,8 @@ func (c *sessionCache) refreshOne(key, token string) { entry.sessionID = result.SessionID entry.expiresAt = result.ExpiresAt entry.services = result.Services + // A backend flip lands within one poll, in either direction. + entry.activity = result.Activity entry.fetchedAt = time.Now() } } diff --git a/packages/agentvault/cache_test.go b/packages/agentvault/cache_test.go index b8e3d775e..ce36e37fa 100644 --- a/packages/agentvault/cache_test.go +++ b/packages/agentvault/cache_test.go @@ -11,16 +11,18 @@ import ( ) type stubResolver struct { - mu sync.Mutex - calls int - result *resolveResult - err error - delay time.Duration + mu sync.Mutex + calls int + result *resolveResult + err error + delay time.Duration + lastHeld *activityGrant } -func (s *stubResolver) resolve(string) (*resolveResult, error) { +func (s *stubResolver) resolve(_ string, held *activityGrant) (*resolveResult, error) { s.mu.Lock() s.calls++ + s.lastHeld = held result, err, delay := s.result, s.err, s.delay s.mu.Unlock() time.Sleep(delay) diff --git a/packages/agentvault/proxy.go b/packages/agentvault/proxy.go index 81e68b5b6..1ad4a253d 100644 --- a/packages/agentvault/proxy.go +++ b/packages/agentvault/proxy.go @@ -75,6 +75,7 @@ type proxyServer struct { opts Options ca *caManager cache *sessionCache + activity *activityLog transport http.RoundTripper configMu sync.RWMutex @@ -420,6 +421,25 @@ func (ps *proxyServer) forwardHTTP(w http.ResponseWriter, r *http.Request, schem } event.Msg("agent-vault: request") + // reqPath is the agent's own path, taken before forward ran, so a substitution that put a real + // credential into the path never reaches the activity log. + if outcome.activity != nil { + var service, bundle *string + if matched != nil { + service, bundle = &matched.name, &matched.accessBundleName + } + ps.activity.record(outcome.activity, activityRecord{ + Method: r.Method, + Host: hostname, + Port: port, + Path: reqPath, + Status: status, + Decision: decision, + Service: service, + AccessBundle: bundle, + }) + } + if err != nil { http.Error(w, body, status) return @@ -459,15 +479,20 @@ func (ps *proxyServer) blocksOffBundle(matched *resolvedService, hostname, port type forwardOutcome struct { brokered bool substituted []string + // Set as soon as the session resolves, so every path after that carries it, a refusal included: an + // agent reaching a host or a path it may not is the most security-relevant line the activity log can + // hold. A session that fails to resolve has nothing to attribute a record to, and leaves it nil. + activity *activityGrant } func (ps *proxyServer) forward(req *http.Request, scheme, hostname, port, sessionToken string) (*http.Response, *resolvedService, forwardOutcome, error) { var outcome forwardOutcome - services, err := ps.cache.get(sessionToken) + services, grant, err := ps.cache.lookup(sessionToken) if err != nil { return nil, nil, outcome, fmt.Errorf("%w: %w", errSessionResolve, err) } + outcome.activity = grant // TRACE and TRACK make the upstream reflect the injected credential back in the response body. Upper // -cased like allowsMethod already was, or a lowercase "trace" walks past. Refused here rather than in diff --git a/packages/agentvault/proxy_policy_test.go b/packages/agentvault/proxy_policy_test.go index a3691451d..24b47b67d 100644 --- a/packages/agentvault/proxy_policy_test.go +++ b/packages/agentvault/proxy_policy_test.go @@ -27,7 +27,7 @@ type echoed struct { type fixedResolver struct{ services []*resolvedService } -func (r fixedResolver) resolve(string) (*resolveResult, error) { +func (r fixedResolver) resolve(string, *activityGrant) (*resolveResult, error) { return &resolveResult{SessionID: "s1", Services: r.services}, nil } diff --git a/packages/agentvault/proxy_truncation_test.go b/packages/agentvault/proxy_truncation_test.go index e05738063..96b887b5d 100644 --- a/packages/agentvault/proxy_truncation_test.go +++ b/packages/agentvault/proxy_truncation_test.go @@ -17,7 +17,7 @@ import ( type sessionOnlyResolver struct{} -func (sessionOnlyResolver) resolve(string) (*resolveResult, error) { +func (sessionOnlyResolver) resolve(string, *activityGrant) (*resolveResult, error) { return &resolveResult{SessionID: "s1"}, nil } diff --git a/packages/agentvault/proxy_tunnel_errors_test.go b/packages/agentvault/proxy_tunnel_errors_test.go index 2927d8650..a8642600a 100644 --- a/packages/agentvault/proxy_tunnel_errors_test.go +++ b/packages/agentvault/proxy_tunnel_errors_test.go @@ -19,7 +19,7 @@ import ( type expiringResolver struct{ ttl time.Duration } -func (r expiringResolver) resolve(string) (*resolveResult, error) { +func (r expiringResolver) resolve(string, *activityGrant) (*resolveResult, error) { exp := time.Now().Add(r.ttl) return &resolveResult{SessionID: "s1", ExpiresAt: &exp}, nil } @@ -94,7 +94,7 @@ func TestAnUpstreamFailureInsideTheTunnelKeepsTheDetailOutOfTheBody(t *testing.T type rejectedProxyResolver struct{} -func (rejectedProxyResolver) resolve(string) (*resolveResult, error) { +func (rejectedProxyResolver) resolve(string, *activityGrant) (*resolveResult, error) { return nil, &api.APIError{StatusCode: 401, Name: proxyTokenRejectedName, ErrorMessage: "Agent Vault proxy token has been revoked"} } diff --git a/packages/agentvault/resolve.go b/packages/agentvault/resolve.go index 25e0d5038..4118a06fa 100644 --- a/packages/agentvault/resolve.go +++ b/packages/agentvault/resolve.go @@ -1,6 +1,7 @@ package agentvault import ( + "encoding/base64" "sort" "strings" "time" @@ -8,6 +9,7 @@ import ( "github.com/Infisical/infisical-merge/packages/api" "github.com/Infisical/infisical-merge/packages/util" "github.com/go-resty/resty/v2" + "github.com/rs/zerolog/log" ) const ( @@ -20,8 +22,12 @@ type resolveResult struct { SessionID string ExpiresAt *time.Time Services []*resolvedService + // nil when logging is off for this session. + Activity *activityGrant } +const activityKeyBytes = 32 + // A seam so the cache can be tested without a server, not because a second implementation is expected. type infisicalResolver struct { client *resty.Client @@ -39,8 +45,10 @@ func newInfisicalResolver(proxyToken func() string) (*infisicalResolver, error) return &infisicalResolver{client: client}, nil } -func (r *infisicalResolver) resolve(sessionToken string) (*resolveResult, error) { - res, err := api.CallResolveAgentVaultSession(r.client, sessionToken) +func (r *infisicalResolver) resolve(sessionToken string, held *activityGrant) (*resolveResult, error) { + res, err := api.CallResolveAgentVaultSession(r.client, sessionToken, api.ResolveAgentVaultSessionRequest{ + HasActivityKey: held != nil, + }) if err != nil { return nil, err } @@ -67,7 +75,44 @@ func (r *infisicalResolver) resolve(sessionToken string) (*resolveResult, error) }) } - return &resolveResult{SessionID: res.SessionID, ExpiresAt: expiresAt, Services: services}, nil + return &resolveResult{ + SessionID: res.SessionID, + ExpiresAt: expiresAt, + Services: services, + Activity: toActivityGrant(res.SessionID, res.Activity, held), + }, nil +} + +// toActivityGrant decides what the proxy records under after a poll. +// +// The key is sent exactly once per session. When we told the server we already hold it, the response +// carries no key and the cached one is carried forward; clearing it here instead would silently stop all +// logging after the very first poll. After any cache eviction the grant and the flag are dropped +// together, so the next resolve asks for the key again and this self-heals. +func toActivityGrant(sessionID string, wire api.AgentVaultActivityGrant, held *activityGrant) *activityGrant { + if !wire.Enabled { + return nil + } + if wire.ProjectID == "" { + log.Warn().Str("sessionId", sessionID).Msg("agent-vault: activity is enabled but no project was named, not recording") + return nil + } + + if wire.SessionKey == "" { + if held != nil { + return &activityGrant{sessionID: sessionID, projectID: wire.ProjectID, key: held.key} + } + log.Warn().Str("sessionId", sessionID).Msg("agent-vault: activity is enabled but no key was sent, not recording") + return nil + } + + key, err := base64.StdEncoding.DecodeString(wire.SessionKey) + if err != nil || len(key) != activityKeyBytes { + log.Warn().Str("sessionId", sessionID).Msg("agent-vault: the activity key Infisical sent is unusable, not recording") + return nil + } + + return &activityGrant{sessionID: sessionID, projectID: wire.ProjectID, key: key} } func toCredential(wire api.AgentVaultCredential) credential { diff --git a/packages/agentvault/resolve_client_test.go b/packages/agentvault/resolve_client_test.go index df4e56817..340521e9f 100644 --- a/packages/agentvault/resolve_client_test.go +++ b/packages/agentvault/resolve_client_test.go @@ -26,7 +26,7 @@ func TestAResolveDoesNotRetryA429WhileTheAgentWaits(t *testing.T) { if err != nil { t.Fatal(err) } - if _, err := resolver.resolve("tok"); err == nil { + if _, err := resolver.resolve("tok", nil); err == nil { t.Fatal("a 429 resolved successfully") } if got := atomic.LoadInt64(&hits); got != 1 { diff --git a/packages/agentvault/run.go b/packages/agentvault/run.go index 023d9e50b..ccbe59887 100644 --- a/packages/agentvault/run.go +++ b/packages/agentvault/run.go @@ -203,6 +203,12 @@ func Start(opts Options, enrollmentToken string) error { } ps.cache = newSessionCache(resolver, ps.pollInterval) + shipper, err := newActivityShipper(opts.ProxyToken) + if err != nil { + return err + } + ps.activity = newActivityLog(state.ProxyID, shipper) + // Port 0 is not "unset": it is the ordinary ask for any free port, so it is never substituted. listener, err := net.Listen("tcp", fmt.Sprintf(":%d", opts.Port)) if err != nil { @@ -225,6 +231,7 @@ func Start(opts Options, enrollmentToken string) error { stop := make(chan struct{}) fatal := make(chan error, 1) go ps.pollLoop(st, stop, fatal) + go ps.activity.run(stop) serveErr := make(chan error, 1) go func() { serveErr <- front.Serve(limited) }() @@ -244,6 +251,9 @@ func Start(opts Options, enrollmentToken string) error { select { case err := <-serveErr: close(stop) + ctx, cancel := context.WithTimeout(context.Background(), 10*time.Second) + defer cancel() + ps.activity.close(ctx) ps.cache.close() if errors.Is(err, http.ErrServerClosed) { return nil @@ -255,6 +265,9 @@ func Start(opts Options, enrollmentToken string) error { ctx, cancel := context.WithTimeout(context.Background(), 10*time.Second) defer cancel() _ = front.Shutdown(ctx) + // After the front door is closed, so no record can arrive mid-flush, and before the cache is + // cleared, since the keys the final seal needs live on its entries. + ps.activity.close(ctx) ps.cache.close() return nil case err := <-fatal: @@ -262,6 +275,7 @@ func Start(opts Options, enrollmentToken string) error { ctx, cancel := context.WithTimeout(context.Background(), 10*time.Second) defer cancel() _ = front.Shutdown(ctx) + ps.activity.close(ctx) ps.cache.close() return err } diff --git a/packages/api/agent_vault.go b/packages/api/agent_vault.go index 9dbb4708f..348e8f108 100644 --- a/packages/api/agent_vault.go +++ b/packages/api/agent_vault.go @@ -102,19 +102,34 @@ type AgentVaultService struct { Substitutions []AgentVaultSubstitution `json:"substitutions"` } +// AgentVaultActivityGrant is what a session needs in order to have its activity recorded. SessionKey is +// sent exactly once per session: the proxy reports that it already holds one and Infisical skips the +// unwrap, which is a KMS round trip, on every poll after the first. +type AgentVaultActivityGrant struct { + Enabled bool `json:"enabled"` + SessionKey string `json:"sessionKey"` + ProjectID string `json:"projectId"` +} + +type ResolveAgentVaultSessionRequest struct { + HasActivityKey bool `json:"hasActivityKey"` +} + type ResolveAgentVaultSessionResponse struct { - SessionID string `json:"sessionId"` - ExpiresAt string `json:"expiresAt"` - Services []AgentVaultService `json:"services"` + SessionID string `json:"sessionId"` + ExpiresAt string `json:"expiresAt"` + Services []AgentVaultService `json:"services"` + Activity AgentVaultActivityGrant `json:"activity"` } -func CallResolveAgentVaultSession(httpClient *resty.Client, sessionToken string) (ResolveAgentVaultSessionResponse, error) { +func CallResolveAgentVaultSession(httpClient *resty.Client, sessionToken string, request ResolveAgentVaultSessionRequest) (ResolveAgentVaultSessionResponse, error) { var res ResolveAgentVaultSessionResponse response, err := httpClient. R(). SetResult(&res). SetHeader("User-Agent", USER_AGENT). SetHeader(AgentVaultSessionHeader, sessionToken). + SetBody(request). Post(fmt.Sprintf("%v/v1/agent-vault/proxy/resolve", config.INFISICAL_URL)) if err != nil { @@ -126,6 +141,45 @@ func CallResolveAgentVaultSession(httpClient *resty.Client, sessionToken string) return res, nil } +// CreateAgentVaultActivityChunkRequest is the metadata for one sealed chunk. The ciphertext itself never +// passes through Infisical: the response carries a presigned URL to PUT it straight to the customer's +// bucket. Re-sending the same ChunkID is idempotent, which is what makes a failed upload safe to retry. +type CreateAgentVaultActivityChunkRequest struct { + ChunkID string `json:"chunkId"` + StartedAt string `json:"startedAt"` + EndedAt string `json:"endedAt"` + FirstSeq uint64 `json:"firstSeq"` + LastSeq uint64 `json:"lastSeq"` + RecordCount int `json:"recordCount"` + DroppedCount uint64 `json:"droppedCount"` + CiphertextBytes int `json:"ciphertextBytes"` + IV string `json:"iv"` +} + +type CreateAgentVaultActivityChunkResponse struct { + ChunkID string `json:"chunkId"` + UploadURL string `json:"uploadUrl"` + ExpiresInSeconds int `json:"expiresInSeconds"` +} + +func CallCreateAgentVaultActivityChunk(httpClient *resty.Client, sessionID string, request CreateAgentVaultActivityChunkRequest) (CreateAgentVaultActivityChunkResponse, error) { + var res CreateAgentVaultActivityChunkResponse + response, err := httpClient. + R(). + SetResult(&res). + SetHeader("User-Agent", USER_AGENT). + SetBody(request). + Post(fmt.Sprintf("%v/v1/agent-vault/proxy/sessions/%s/activity/chunks", config.INFISICAL_URL, sessionID)) + + if err != nil { + return CreateAgentVaultActivityChunkResponse{}, NewGenericRequestError("CallCreateAgentVaultActivityChunk", err) + } + if response.IsError() { + return CreateAgentVaultActivityChunkResponse{}, NewAPIErrorWithResponse("CallCreateAgentVaultActivityChunk", response, nil) + } + return res, nil +} + type CreateAgentVaultSessionRequest struct { AccessBundles []string `json:"accessBundles"` TTL string `json:"ttl"` From e103f778bbc0545ec810808ae6c2f7b38cb1652e Mon Sep 17 00:00:00 2001 From: saif <11242541+saifsmailbox98@users.noreply.github.com> Date: Thu, 24 Sep 2026 01:53:55 +0530 Subject: [PATCH 02/43] fix(agent-vault): grow a session's activity buffer as records arrive --- packages/agentvault/activity_spool.go | 47 +++++++++++++++++++++------ packages/agentvault/activity_test.go | 44 +++++++++++++++++++++++++ 2 files changed, 81 insertions(+), 10 deletions(-) diff --git a/packages/agentvault/activity_spool.go b/packages/agentvault/activity_spool.go index 772b6088a..ca1102931 100644 --- a/packages/agentvault/activity_spool.go +++ b/packages/agentvault/activity_spool.go @@ -25,38 +25,60 @@ type activityRecord struct { AccessBundle *string `json:"accessBundle"` } +// The first allocation a ring makes. A session that sends a handful of requests a minute never needs more. +const activityRingInitialSize = 64 + // activityRing is a bounded FIFO that overwrites its oldest entry when full and counts what it lost, so a // burst costs the oldest records rather than the newest and the gap is visible in the timeline. +// +// It grows to capacity only as records arrive and lets go of its buffer once drained. Reserving capacity up +// front cost every session about 700 KB from its first request, so a proxy serving a few hundred mostly +// idle sessions held hundreds of megabytes of empty slots. type activityRing struct { - buf []activityRecord - head int - n int - dropped uint64 + buf []activityRecord + capacity int + head int + n int + dropped uint64 } func newActivityRing(capacity int) activityRing { - return activityRing{buf: make([]activityRecord, capacity)} + return activityRing{capacity: capacity} } func (r *activityRing) len() int { return r.n } func (r *activityRing) push(rec activityRecord) (evicted bool) { - capacity := len(r.buf) - if capacity == 0 { + if r.capacity == 0 { r.dropped++ return true } - if r.n == capacity { + if r.n == len(r.buf) && len(r.buf) < r.capacity { + r.grow() + } + if r.n == r.capacity { r.buf[r.head] = rec - r.head = (r.head + 1) % capacity + r.head = (r.head + 1) % len(r.buf) r.dropped++ return true } - r.buf[(r.head+r.n)%capacity] = rec + r.buf[(r.head+r.n)%len(r.buf)] = rec r.n++ return false } +// grow doubles the buffer, up to capacity, and unwraps it so the oldest record sits at index 0. +func (r *activityRing) grow() { + size := max(activityRingInitialSize, 2*len(r.buf)) + size = min(size, r.capacity) + next := make([]activityRecord, size) + for i := 0; i < r.n; i++ { + next[i] = r.buf[(r.head+i)%len(r.buf)] + } + r.buf = next + r.head = 0 +} + // drain removes up to max records, oldest first. The caller seals one chunk per call and loops until the // ring is empty, which is what keeps a chunk inside the server's recordCount limit even when a slow tick // let the ring grow past it. @@ -73,6 +95,11 @@ func (r *activityRing) drain(max int) []activityRecord { } r.head = (r.head + max) % len(r.buf) r.n -= max + if r.n == 0 { + // Released between flushes, so a session that went quiet holds nothing until it speaks again. + r.buf = nil + r.head = 0 + } return out } diff --git a/packages/agentvault/activity_test.go b/packages/agentvault/activity_test.go index 28c69e68a..895432db0 100644 --- a/packages/agentvault/activity_test.go +++ b/packages/agentvault/activity_test.go @@ -182,6 +182,50 @@ func TestTheRingDropsTheOldestAndCountsIt(t *testing.T) { } } +func TestTheRingAllocatesOnlyWhatItHolds(t *testing.T) { + ring := newActivityRing(activitySpoolCapacity) + ring.push(activityRecord{Seq: 0}) + + // A quiet session is the common case, so it must not pay for a busy one's headroom. + if got := cap(ring.buf); got > activityRingInitialSize { + t.Fatalf("one record reserved room for %d, expected at most %d", got, activityRingInitialSize) + } + + ring.drain(10) + if ring.buf != nil { + t.Fatal("a drained ring kept its buffer; an idle session should hold nothing") + } +} + +func TestTheRingKeepsItsOrderWhileItGrowsPastAWrap(t *testing.T) { + ring := newActivityRing(activitySpoolCapacity) + var next uint64 + push := func(n int) { + for i := 0; i < n; i++ { + ring.push(activityRecord{Seq: next}) + next++ + } + } + + // Fill the first allocation, drain part of it so the head moves, then push enough to wrap and grow. + push(activityRingInitialSize) + ring.drain(10) + push(activityRingInitialSize * 3) + + drained := ring.drain(activitySpoolCapacity) + if len(drained) != activityRingInitialSize*4-10 { + t.Fatalf("drained %d records, expected %d", len(drained), activityRingInitialSize*4-10) + } + for i, rec := range drained { + if rec.Seq != uint64(10+i) { + t.Fatalf("record %d has seq %d, expected %d; growth reordered the ring", i, rec.Seq, 10+i) + } + } + if ring.dropped != 0 { + t.Fatalf("growth counted %d drops; nothing was over capacity", ring.dropped) + } +} + func TestTheRingDrainsInSlicesTheServerAccepts(t *testing.T) { ring := newActivityRing(activitySpoolCapacity) for i := 0; i < 2500; i++ { From 90a6b735887816f349020246eec9bee056ef5912 Mon Sep 17 00:00:00 2001 From: saif <11242541+saifsmailbox98@users.noreply.github.com> Date: Thu, 24 Sep 2026 01:55:48 +0530 Subject: [PATCH 03/43] fix(agent-vault): hold the proxy-wide activity byte cap across every session --- packages/agentvault/activity.go | 60 +++++++++++++++++++++------ packages/agentvault/activity_spool.go | 16 +++++++ packages/agentvault/activity_test.go | 53 +++++++++++++++++++++++ 3 files changed, 116 insertions(+), 13 deletions(-) diff --git a/packages/agentvault/activity.go b/packages/agentvault/activity.go index 84c8e3b0c..bbc4fe586 100644 --- a/packages/agentvault/activity.go +++ b/packages/agentvault/activity.go @@ -73,8 +73,9 @@ type activityLog struct { // when it grows, the way the session cache handles its own refusal map. seqBySession map[string]uint64 - total int - sealedBytes int + total int + sealedBytes int + nextSealOrder uint64 // Proxy-wide, because both reasons are proxy-wide: the ceiling is per organization and the switch is // per project, and this proxy serves one project. @@ -302,6 +303,8 @@ func (a *activityLog) sealRing(spool *activitySpool) { } a.mu.Lock() + chunk.sealOrder = a.nextSealOrder + a.nextSealOrder++ spool.pending = append(spool.pending, chunk) a.sealedBytes += len(chunk.ciphertext) a.enforcePendingCapsLocked(spool) @@ -310,21 +313,51 @@ func (a *activityLog) sealRing(spool *activitySpool) { } // enforcePendingCapsLocked runs on every append, because the caps are a property of `pending` rather than -// a branch of the ceiling handler. The oldest sealed chunk is evicted and its records counted as dropped. +// a branch of the ceiling handler. Both evict the oldest sealed chunk and count what it held as dropped. +// +// The byte cap is proxy-wide, so it evicts the oldest chunk on the proxy wherever it sits, a session's only +// chunk included. Sparing every session its last chunk let an outage hold one per session with no bound. func (a *activityLog) enforcePendingCapsLocked(spool *activitySpool) { - for len(spool.pending) > activityPendingChunks || (a.sealedBytes > activityTotalSealedBytes && len(spool.pending) > 1) { - oldest := spool.pending[0] - spool.pending = spool.pending[1:] - a.sealedBytes -= len(oldest.ciphertext) - spool.ring.dropped += uint64(oldest.meta.RecordCount) - log.Warn(). - Str("sessionId", spool.sessionID). - Str("chunkId", oldest.meta.ChunkID). - Int("records", oldest.meta.RecordCount). - Msg("agent-vault: dropped an unshipped activity chunk, the buffer is full") + for len(spool.pending) > activityPendingChunks { + a.evictOldestLocked(spool) + } + for a.sealedBytes > activityTotalSealedBytes { + victim := a.oldestPendingLocked() + if victim == nil { + // Unreachable while sealedBytes counts only pending chunks. Guarded anyway: this runs under the + // lock every proxied request takes to record. + return + } + a.evictOldestLocked(victim) } } +// oldestPendingLocked scans every spool, which is fine: it only runs once the proxy is over its byte cap. +func (a *activityLog) oldestPendingLocked() *activitySpool { + var oldest *activitySpool + for _, spool := range a.spools { + if len(spool.pending) == 0 { + continue + } + if oldest == nil || spool.pending[0].sealOrder < oldest.pending[0].sealOrder { + oldest = spool + } + } + return oldest +} + +func (a *activityLog) evictOldestLocked(spool *activitySpool) { + oldest := spool.pending[0] + spool.pending = spool.pending[1:] + a.sealedBytes -= len(oldest.ciphertext) + spool.ring.dropped += oldest.lostCount() + log.Warn(). + Str("sessionId", spool.sessionID). + Str("chunkId", oldest.meta.ChunkID). + Int("records", oldest.meta.RecordCount). + Msg("agent-vault: dropped an unshipped activity chunk, the buffer is full") +} + func (a *activityLog) flushSpool(ctx context.Context, spool *activitySpool, final bool) { if paused, reason := a.paused(); paused { // Seal once at pause onset so the ring does not overflow while waiting, then hold everything: ops @@ -365,6 +398,7 @@ func (a *activityLog) shipChunk(ctx context.Context, spool *activitySpool, chunk if err != nil { return a.handleCreateFailure(spool, chunk, err) } + chunk.posted = true chunk.uploadURL = res.UploadURL chunk.urlExpires = a.now().Add(time.Duration(res.ExpiresInSeconds) * time.Second) } diff --git a/packages/agentvault/activity_spool.go b/packages/agentvault/activity_spool.go index ca1102931..f1c9fe6c7 100644 --- a/packages/agentvault/activity_spool.go +++ b/packages/agentvault/activity_spool.go @@ -119,6 +119,22 @@ type sealedChunk struct { // chunk id and the server replays it idempotently. uploadURL string urlExpires time.Time + + // Proxy-wide, so the byte cap can find the oldest chunk across every session. + sealOrder uint64 + // Set once Infisical has written the row. Never cleared: a re-POST replays the same row. + posted bool +} + +// lostCount is what a chunk that will never be uploaded adds to its session's gap. Once the POST succeeded +// the row exists, and the viewer already shows those records as a batch it cannot read, so counting them +// again would report one loss twice. The drop count the chunk carried is shown nowhere else, so it always +// comes back. +func (c *sealedChunk) lostCount() uint64 { + if c.posted { + return c.meta.DroppedCount + } + return c.meta.DroppedCount + uint64(c.meta.RecordCount) } // activitySpool is one session's buffer on this proxy. proxyID is constant for the process, so it lives on diff --git a/packages/agentvault/activity_test.go b/packages/agentvault/activity_test.go index 895432db0..e4a2bf2c8 100644 --- a/packages/agentvault/activity_test.go +++ b/packages/agentvault/activity_test.go @@ -483,6 +483,59 @@ func TestThePendingCapEvictsTheOldestAndCountsIt(t *testing.T) { } } +func TestTheByteCapEvictsTheOldestChunkOnTheProxy(t *testing.T) { + log, _, _ := newTestLog(&fakeShipper{}) + + // Only the length counts toward the cap, so every chunk can share one buffer. + blob := make([]byte, 12<<20) + add := func(sessionID string, order uint64, posted bool, carried uint64) *activitySpool { + spool, ok := log.spools[sessionID] + if !ok { + spool = newActivitySpool(testGrant(sessionID), log.now()) + log.spools[sessionID] = spool + } + spool.pending = append(spool.pending, &sealedChunk{ + meta: api.CreateAgentVaultActivityChunkRequest{ChunkID: fmt.Sprintf("c%d", order), RecordCount: 100, DroppedCount: carried}, + ciphertext: blob, + sealOrder: order, + posted: posted, + }) + log.sealedBytes += len(blob) + return spool + } + + // One chunk per session, the shape an outage leaves behind, seven of them for 84 MiB against a 64 MiB cap. + log.mu.Lock() + add("oldest", 0, false, 7) + add("posted", 1, true, 3) + var newest *activitySpool + for i := 2; i < 7; i++ { + newest = add(fmt.Sprintf("s%d", i), uint64(i), false, 0) + } + log.enforcePendingCapsLocked(newest) + log.mu.Unlock() + + if log.sealedBytes > activityTotalSealedBytes { + t.Fatalf("the proxy holds %d sealed bytes, past the %d cap", log.sealedBytes, activityTotalSealedBytes) + } + if got := len(log.spools["oldest"].pending) + len(log.spools["posted"].pending); got != 0 { + t.Fatalf("the two oldest chunks were not evicted, %d remain", got) + } + for i := 2; i < 7; i++ { + if len(log.spools[fmt.Sprintf("s%d", i)].pending) != 1 { + t.Fatalf("s%d lost its chunk; only the oldest should go", i) + } + } + // Never posted, so its records and the drops it carried are both unaccounted for anywhere else. + if got := log.spools["oldest"].ring.dropped; got != 107 { + t.Fatalf("the unposted chunk counted %d dropped, expected 107", got) + } + // Posted, so the viewer already shows its records as unreadable. Only the carried drops come back. + if got := log.spools["posted"].ring.dropped; got != 3 { + t.Fatalf("the posted chunk counted %d dropped, expected 3", got) + } +} + func TestTheTickBreakerStopsHammeringADeadBucket(t *testing.T) { shipper := &fakeShipper{putDefault: errors.New("i/o timeout")} log, _, tick := newTestLog(shipper) From b3bb99b4ef34ffb62e6a6dcea7a152098b1a89b9 Mon Sep 17 00:00:00 2001 From: saif <11242541+saifsmailbox98@users.noreply.github.com> Date: Thu, 24 Sep 2026 01:56:07 +0530 Subject: [PATCH 04/43] fix(agent-vault): log the activity limit as an error, in the product's words --- packages/agentvault/activity.go | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/packages/agentvault/activity.go b/packages/agentvault/activity.go index bbc4fe586..6be62eb53 100644 --- a/packages/agentvault/activity.go +++ b/packages/agentvault/activity.go @@ -447,7 +447,8 @@ func (a *activityLog) handleCreateFailure(spool *activitySpool, chunk *sealedChu case isActivityErrorNamed(err, activityCeilingReachedName): a.pause(activityCeilingReachedName) - log.Warn().Err(err).Msg("agent-vault: the organization's activity storage limit is reached, pausing for 15m") + // An error, not a warning: recording has stopped for the whole organization until Infisical acts. + log.Error().Err(err).Msg("agent-vault: activity logging has reached its limit for this organization, retrying in 15m") return false case isActivityErrorNamed(err, activityDisabledName): From 6ad878394313dde1f75805e1e3207b614a8901aa Mon Sep 17 00:00:00 2001 From: saif <11242541+saifsmailbox98@users.noreply.github.com> Date: Thu, 24 Sep 2026 01:57:03 +0530 Subject: [PATCH 05/43] fix(agent-vault): cap the method and port an agent can put in its own records --- packages/agentvault/activity_wiring_test.go | 28 +++++++++++++++++++++ packages/agentvault/policy.go | 10 +++++--- packages/agentvault/proxy.go | 11 +++++--- 3 files changed, 43 insertions(+), 6 deletions(-) diff --git a/packages/agentvault/activity_wiring_test.go b/packages/agentvault/activity_wiring_test.go index 30730c5e9..9dc280bb6 100644 --- a/packages/agentvault/activity_wiring_test.go +++ b/packages/agentvault/activity_wiring_test.go @@ -166,6 +166,34 @@ func TestABlockedRequestIsRecordedWithItsRefusal(t *testing.T) { } } +func TestAnOversizedMethodIsRecordedTruncated(t *testing.T) { + upstream := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { + w.WriteHeader(http.StatusMethodNotAllowed) + })) + defer upstream.Close() + + front, log, _ := newRecordingProxy(t, TrafficPolicyAnyHost, nil) + + proxyURL, _ := url.Parse(front.URL) + proxyURL.User = url.UserPassword("infisical", "agv_test-token") + client := &http.Client{Transport: &http.Transport{Proxy: http.ProxyURL(proxyURL)}} + // A valid token, so Go sends it. Only the header limit stops an agent sending one far longer. + req, err := http.NewRequest(strings.Repeat("A", 5000), upstream.URL+"/v1/thing", nil) + if err != nil { + t.Fatal(err) + } + res, err := client.Do(req) + if err != nil { + t.Fatalf("the proxy did not answer: %v", err) + } + defer res.Body.Close() + + got := drainOneRecord(t, log) + if len(got.Method) > maxLoggedMethodLen+len("...[truncated]") { + t.Fatalf("the record kept a %d-byte method; an agent could inflate its own records", len(got.Method)) + } +} + func TestRecordingSurvivesAnUnreachableUpstream(t *testing.T) { front, log, _ := newRecordingProxy(t, TrafficPolicyAnyHost, nil) diff --git a/packages/agentvault/policy.go b/packages/agentvault/policy.go index e14ab32a3..08e7adaf7 100644 --- a/packages/agentvault/policy.go +++ b/packages/agentvault/policy.go @@ -114,10 +114,14 @@ func requestPath(req *http.Request) string { } func truncatePath(path string) string { - if len(path) > maxLoggedPathLen { - return path[:maxLoggedPathLen] + "...[truncated]" + return truncateLogged(path, maxLoggedPathLen) +} + +func truncateLogged(value string, limit int) string { + if len(value) > limit { + return value[:limit] + "...[truncated]" } - return path + return value } // Never decodes: anything whose meaning depends on the upstream's normalisation is refused outright, so the diff --git a/packages/agentvault/proxy.go b/packages/agentvault/proxy.go index 1ad4a253d..0aed38162 100644 --- a/packages/agentvault/proxy.go +++ b/packages/agentvault/proxy.go @@ -47,6 +47,10 @@ const ( maxConcurrentConns = 512 maxLoggedPathLen = 2048 + // The agent chooses the method and the port too, and Go bounds them only by the 1 MiB header limit. + // Uncapped, one request could make its own record big enough to stall whoever reads the session. + maxLoggedMethodLen = 32 + maxLoggedPortLen = 16 ) var errHostBlocked = errors.New("host blocked by policy") @@ -362,6 +366,7 @@ func (ps *proxyServer) forwardHTTP(w http.ResponseWriter, r *http.Request, schem // requestPath rather than EscapedPath, so a brokered request is never recorded with a blank path. reqPath := truncatePath(requestPath(r)) + reqMethod := truncateLogged(r.Method, maxLoggedMethodLen) resp, matched, outcome, err := ps.forward(r, scheme, hostname, port, sessionToken) @@ -400,7 +405,7 @@ func (ps *proxyServer) forwardHTTP(w http.ResponseWriter, r *http.Request, schem case decisionError: event = log.Error() } - event.Str("method", r.Method). + event.Str("method", reqMethod). Str("host", hostname). Str("path", reqPath). Str("decision", decision). @@ -429,9 +434,9 @@ func (ps *proxyServer) forwardHTTP(w http.ResponseWriter, r *http.Request, schem service, bundle = &matched.name, &matched.accessBundleName } ps.activity.record(outcome.activity, activityRecord{ - Method: r.Method, + Method: reqMethod, Host: hostname, - Port: port, + Port: truncateLogged(port, maxLoggedPortLen), Path: reqPath, Status: status, Decision: decision, From 6b3c563e6b0e500f872f522b41a46e432b95cb28 Mon Sep 17 00:00:00 2001 From: saif <11242541+saifsmailbox98@users.noreply.github.com> Date: Thu, 24 Sep 2026 01:58:51 +0530 Subject: [PATCH 06/43] fix(agent-vault): split activity chunks by size and count refused ones as dropped --- packages/agentvault/activity.go | 56 +++++++++++++++++++-------- packages/agentvault/activity_spool.go | 43 ++++++++++++++++++-- packages/agentvault/activity_test.go | 55 +++++++++++++++++++++++++- 3 files changed, 134 insertions(+), 20 deletions(-) diff --git a/packages/agentvault/activity.go b/packages/agentvault/activity.go index 6be62eb53..0de5cf1d0 100644 --- a/packages/agentvault/activity.go +++ b/packages/agentvault/activity.go @@ -16,6 +16,10 @@ const ( activityFlushInterval = 60 * time.Second // The server refuses a chunk over this, so the ring is drained in slices of at most this many. activityFlushRecords = 1000 + // The server also refuses a chunk over 8 MiB. The agent controls how long its paths are and JSON + // escapes some bytes six to one, so a full slice can pass that. Half the limit leaves ordinary + // flushes whole. + activityMaxChunkPlaintext = 4 << 20 // Five missed flushes of headroom per session before the oldest records start being overwritten. activitySpoolCapacity = 5000 @@ -286,32 +290,49 @@ func (a *activityLog) sealRing(spool *activitySpool) { return } a.total -= len(records) - // droppedCount rides the first slice only, so one gap is reported once. + // droppedCount rides the first chunk only, so one gap is reported once. dropped := spool.ring.takeDropped() now := a.now() a.mu.Unlock() - chunk, err := spool.sealSlice(a.proxyID, records, dropped, now) + groups, err := packActivityRecords(records) if err != nil { - a.mu.Lock() - // These records are gone, so they join the gap rather than vanishing from the count with it. - spool.ring.dropped += dropped + uint64(len(records)) - a.mu.Unlock() - log.Error().Err(err).Str("sessionId", spool.sessionID).Int("records", len(records)). - Msg("agent-vault: could not seal an activity chunk, dropping those records") + a.dropUnsealed(spool, len(records), dropped, err) continue } + for i, group := range groups { + var groupDropped uint64 + if i == 0 { + groupDropped = dropped + } - a.mu.Lock() - chunk.sealOrder = a.nextSealOrder - a.nextSealOrder++ - spool.pending = append(spool.pending, chunk) - a.sealedBytes += len(chunk.ciphertext) - a.enforcePendingCapsLocked(spool) - a.mu.Unlock() + chunk, err := spool.sealSlice(a.proxyID, group.records, group.plaintext, groupDropped, now) + if err != nil { + a.dropUnsealed(spool, len(group.records), groupDropped, err) + continue + } + + a.mu.Lock() + chunk.sealOrder = a.nextSealOrder + a.nextSealOrder++ + spool.pending = append(spool.pending, chunk) + a.sealedBytes += len(chunk.ciphertext) + a.enforcePendingCapsLocked(spool) + a.mu.Unlock() + } } } +// dropUnsealed accounts for records that could not be sealed. They are gone, so they join the gap rather +// than vanishing from the count with it. +func (a *activityLog) dropUnsealed(spool *activitySpool, records int, dropped uint64, err error) { + a.mu.Lock() + spool.ring.dropped += dropped + uint64(records) + a.mu.Unlock() + log.Error().Err(err).Str("sessionId", spool.sessionID).Int("records", records). + Msg("agent-vault: could not seal an activity chunk, dropping those records") +} + // enforcePendingCapsLocked runs on every append, because the caps are a property of `pending` rather than // a branch of the ceiling handler. Both evict the oldest sealed chunk and count what it held as dropped. // @@ -462,9 +483,12 @@ func (a *activityLog) handleCreateFailure(spool *activitySpool, chunk *sealedChu if len(spool.pending) > 0 && spool.pending[0] == chunk { spool.pending = spool.pending[1:] a.sealedBytes -= len(chunk.ciphertext) + // Otherwise a refused chunk leaves no trace in the timeline, and an agent that can get its own + // chunk refused can erase what it did. + spool.ring.dropped += chunk.lostCount() } a.mu.Unlock() - log.Error().Err(err).Str("chunkId", chunk.meta.ChunkID). + log.Error().Err(err).Str("chunkId", chunk.meta.ChunkID).Int("records", chunk.meta.RecordCount). Msg("agent-vault: Infisical rejected an activity chunk as malformed, dropping it") return false diff --git a/packages/agentvault/activity_spool.go b/packages/agentvault/activity_spool.go index f1c9fe6c7..b09d7a289 100644 --- a/packages/agentvault/activity_spool.go +++ b/packages/agentvault/activity_spool.go @@ -164,13 +164,50 @@ func newActivitySpool(g *activityGrant, now time.Time) *activitySpool { } } -// sealSlice turns one slice of records into a sealed chunk ready to ship. -func (s *activitySpool) sealSlice(proxyID string, records []activityRecord, dropped uint64, now time.Time) (*sealedChunk, error) { - plaintext, err := json.Marshal(records) +type activityGroup struct { + records []activityRecord + plaintext []byte +} + +// packActivityRecords splits one drained slice into chunks the server takes by size as well as by count. +// Nearly every flush fits whole and is marshalled once. The rest are marshalled per record and packed, which +// yields exactly what marshalling each group as a slice would: '[', the records joined by ',', then ']'. +func packActivityRecords(records []activityRecord) ([]activityGroup, error) { + whole, err := json.Marshal(records) if err != nil { return nil, err } + if len(whole) <= activityMaxChunkPlaintext { + return []activityGroup{{records: records, plaintext: whole}}, nil + } + + var groups []activityGroup + var buf []byte + start := 0 + for i, rec := range records { + part, err := json.Marshal(rec) + if err != nil { + return nil, err + } + // One byte for the separator before it and one for the closing bracket. A record too big to share a + // chunk still gets one of its own rather than being split. + if len(buf) > 0 && len(buf)+1+len(part)+1 > activityMaxChunkPlaintext { + groups = append(groups, activityGroup{records: records[start:i], plaintext: append(buf, ']')}) + buf, start = nil, i + } + if len(buf) == 0 { + buf = append(buf, '[') + } else { + buf = append(buf, ',') + } + buf = append(buf, part...) + } + groups = append(groups, activityGroup{records: records[start:], plaintext: append(buf, ']')}) + return groups, nil +} +// sealSlice turns one group of records, already marshalled, into a sealed chunk ready to ship. +func (s *activitySpool) sealSlice(proxyID string, records []activityRecord, plaintext []byte, dropped uint64, now time.Time) (*sealedChunk, error) { chunkID := newActivityChunkID(now) aad := buildActivityAAD(s.projectID, s.sessionID, proxyID, chunkID) ciphertext, iv, err := sealActivity(s.key, plaintext, aad) diff --git a/packages/agentvault/activity_test.go b/packages/agentvault/activity_test.go index e4a2bf2c8..f84fbde42 100644 --- a/packages/agentvault/activity_test.go +++ b/packages/agentvault/activity_test.go @@ -5,6 +5,7 @@ import ( "errors" "fmt" "net/http" + "strings" "sync" "testing" "time" @@ -18,6 +19,7 @@ type shipperCall struct { chunkID string url string bytes int + dropped uint64 body []byte final bool } @@ -47,7 +49,7 @@ func (f *fakeShipper) createChunk(final bool, sessionID string, req api.CreateAg f.mu.Lock() defer f.mu.Unlock() - f.calls = append(f.calls, shipperCall{kind: "post", sessionID: sessionID, chunkID: req.ChunkID, bytes: req.CiphertextBytes, final: final}) + f.calls = append(f.calls, shipperCall{kind: "post", sessionID: sessionID, chunkID: req.ChunkID, bytes: req.CiphertextBytes, dropped: req.DroppedCount, final: final}) result := f.postDefault if len(f.postResults) > 0 { @@ -447,6 +449,57 @@ func TestAPoisonChunkIsDroppedAndTheRestShip(t *testing.T) { } } +func TestARefusedChunkIsCountedOnTheNextOne(t *testing.T) { + shipper := &fakeShipper{postResults: []scriptedResult{{err: apiErr(http.StatusUnprocessableEntity, "")}}} + log, _, tick := newTestLog(shipper) + + log.record(testGrant("s1"), aRecord("api.github.com")) + tick() + log.record(testGrant("s1"), aRecord("api.github.com")) + tick() + + // The refused chunk was never written, so the next one is the only place its record can show up. + posts := shipper.posts() + if len(posts) != 2 || posts[1].dropped != 1 { + t.Fatalf("posts were %+v, expected the second to carry one dropped record", posts) + } +} + +func TestAFlushTooBigForOneChunkIsSplitBySize(t *testing.T) { + shipper := &fakeShipper{postDefault: scriptedResult{err: errors.New("infisical unreachable")}} + log, _, tick := newTestLog(shipper) + grant := testGrant("s1") + + // JSON writes '&' as &, so a full slice of these is about 12 MB: over the server's 8 MiB. + for i := 0; i < activityFlushRecords; i++ { + rec := aRecord("api.github.com") + rec.Path = truncatePath("/" + strings.Repeat("&", maxLoggedPathLen)) + log.record(grant, rec) + } + tick() + + const gcmTag = 16 + pending := log.spools["s1"].pending + if len(pending) < 2 { + t.Fatalf("a ~12 MB flush sealed into %d chunk(s); it must be split", len(pending)) + } + var next uint64 + var total int + for i, chunk := range pending { + if chunk.meta.CiphertextBytes-gcmTag > activityMaxChunkPlaintext { + t.Fatalf("chunk %d holds %d bytes of plaintext, over %d", i, chunk.meta.CiphertextBytes-gcmTag, activityMaxChunkPlaintext) + } + if chunk.meta.FirstSeq != next { + t.Fatalf("chunk %d starts at seq %d, expected %d; a record was lost or reordered", i, chunk.meta.FirstSeq, next) + } + next = chunk.meta.LastSeq + 1 + total += chunk.meta.RecordCount + } + if total != activityFlushRecords { + t.Fatalf("the chunks hold %d records, expected %d", total, activityFlushRecords) + } +} + func TestARateLimitIsRetriedRatherThanTreatedAsPoison(t *testing.T) { shipper := &fakeShipper{postResults: []scriptedResult{{err: apiErr(http.StatusTooManyRequests, "")}}} log, _, tick := newTestLog(shipper) From 15b1cf8ea2a8f1680da238d8899f8f6d63a1d709 Mon Sep 17 00:00:00 2001 From: saif <11242541+saifsmailbox98@users.noreply.github.com> Date: Thu, 24 Sep 2026 01:59:55 +0530 Subject: [PATCH 07/43] fix(agent-vault): record an opaque request target as a blocked request --- packages/agentvault/policy.go | 4 +-- packages/agentvault/proxy.go | 22 +++++++----- .../agentvault/proxy_requesttarget_test.go | 36 +++++++++++++++++-- 3 files changed, 50 insertions(+), 12 deletions(-) diff --git a/packages/agentvault/policy.go b/packages/agentvault/policy.go index 08e7adaf7..a3bfbecd7 100644 --- a/packages/agentvault/policy.go +++ b/packages/agentvault/policy.go @@ -103,8 +103,8 @@ func escapeInvalidPathBytes(raw string) string { func requestPath(req *http.Request) string { path := req.URL.EscapedPath() if path == "" { - // forwardHTTP refuses an opaque target before this runs, so the branch is a floor under that check - // rather than a shape expected here. A genuinely empty path is the root. + // forward refuses an opaque target before any policy reads the path, so this branch is what the log + // line and the activity record show for one. A genuinely empty path is the root. if req.URL.Opaque != "" { return req.URL.Opaque } diff --git a/packages/agentvault/proxy.go b/packages/agentvault/proxy.go index 0aed38162..422746143 100644 --- a/packages/agentvault/proxy.go +++ b/packages/agentvault/proxy.go @@ -55,6 +55,12 @@ const ( var errHostBlocked = errors.New("host blocked by policy") +// 'http:admin/secrets' parses to an empty path and a non-empty Opaque, which the upstream would receive as +// a request-target with no leading slash, and no path rule could judge it. RFC 9110 makes an http(s) URI +// without an authority invalid. handlePlainForward refuses the shape already; a tunnel reaches forward +// directly, so the refusal lives there. +var errOpaqueTarget = errors.New("the request target must be a path; opaque forms are not forwarded") + // Wraps a resolve failure so the tunnel can tell it from an upstream failure without reading the text. var errSessionResolve = errors.New("failed to resolve the session") @@ -352,14 +358,6 @@ func (ps *proxyServer) handlePlainForward(w http.ResponseWriter, r *http.Request } func (ps *proxyServer) forwardHTTP(w http.ResponseWriter, r *http.Request, scheme, hostname, port, sessionToken string) { - // 'http:admin/secrets' parses to an empty path and a non-empty Opaque, which the upstream would receive - // as a request-target with no leading slash. handlePlainForward refuses the shape already; the tunnel - // reaches this handler directly, so the refusal belongs here where both doors meet. - if r.URL.Opaque != "" { - http.Error(w, "the request target must be a path; opaque forms are not forwarded", http.StatusBadRequest) - return - } - // Before anything reads the path: the policy check, the substitutions and the forward all have to see // the bytes the agent sent, not the ones Go rebuilds. normalizeRequestTarget(r.URL) @@ -382,6 +380,8 @@ func (ps *proxyServer) forwardHTTP(w http.ResponseWriter, r *http.Request, schem case errors.Is(err, errBodyUnreadable): // The agent's upload broke, so this is its request to retry rather than an upstream or policy failure. decision, status, body = decisionBlocked, http.StatusBadRequest, err.Error() + case errors.Is(err, errOpaqueTarget): + decision, status, body = decisionBlocked, http.StatusBadRequest, errOpaqueTarget.Error() case isProxyTokenRejected(err): decision, status, body = decisionError, http.StatusServiceUnavailable, proxyRevokedBody case isSessionGone(err): @@ -506,6 +506,12 @@ func (ps *proxyServer) forward(req *http.Request, scheme, hostname, port, sessio return nil, nil, outcome, fmt.Errorf("method %s echoes headers back: %w", method, errPolicyBlocked) } + // After the lookup for the same reason: the only use of this form is probing for a path the policy + // misreads, which is the attempt an admin most wants to see in the session's activity. + if req.URL.Opaque != "" { + return nil, nil, outcome, errOpaqueTarget + } + matched := bestMatch(services, hostname, port) if ps.blocksOffBundle(matched, hostname, port) { diff --git a/packages/agentvault/proxy_requesttarget_test.go b/packages/agentvault/proxy_requesttarget_test.go index e7c6de1b4..da246a9f9 100644 --- a/packages/agentvault/proxy_requesttarget_test.go +++ b/packages/agentvault/proxy_requesttarget_test.go @@ -108,8 +108,8 @@ func TestAPlaceholderInThePathStillSubstitutes(t *testing.T) { } // 'http:admin/secrets' parses to an empty path and a non-empty Opaque. handlePlainForward refuses the shape, -// the tunnel reaches forwardHTTP directly, and the upstream would have received a target with no leading -// slash and a real credential on it. +// the tunnel reaches forward directly, and the upstream would have received a target with no leading slash +// and a real credential on it. func TestAnOpaqueRequestTargetIsRefusedInsideATunnel(t *testing.T) { proxyHost, upstreamHost := newRequestTargetFixture(t, nil) @@ -119,6 +119,38 @@ func TestAnOpaqueRequestTargetIsRefusedInsideATunnel(t *testing.T) { } } +func TestAnOpaqueRequestTargetIsRecordedAsBlocked(t *testing.T) { + upstream := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { + w.WriteHeader(http.StatusOK) + })) + t.Cleanup(upstream.Close) + uu, _ := url.Parse(upstream.URL) + upstreamHost := "127.0.0.1:" + uu.Port() + + key, cert, err := generateRootCa() + if err != nil { + t.Fatal(err) + } + ps := &proxyServer{transport: newUpstreamTransport(), ca: newCaManager(key, cert)} + ps.setConfig(ProxyConfig{TrafficPolicy: TrafficPolicyAnyHost}) + ps.cache = newSessionCache(grantingResolver{}, ps.pollInterval) + ps.activity = newActivityLog("proxy-1", &fakeShipper{}) + front := httptest.NewServer(http.HandlerFunc(ps.dispatch)) + t.Cleanup(front.Close) + fu, _ := url.Parse(front.URL) + + resp, _ := rawProxyRequest(t, fu.Host, "GET http:admin/secrets HTTP/1.1", upstreamHost, upstreamHost) + if resp.StatusCode != http.StatusBadRequest { + t.Fatalf("status = %d, want 400", resp.StatusCode) + } + + // Refused, and still on the record: nobody sends this form except to probe the path rules. + got := drainOneRecord(t, ps.activity) + if got.Decision != decisionBlocked || got.Status != http.StatusBadRequest || got.Path != "admin/secrets" { + t.Fatalf("record is %+v, expected a blocked 400 for admin/secrets", got) + } +} + // The same two cases through a real CONNECT + TLS tunnel, which is how an agent actually arrives. The raw // dial is still required: a Go client escapes the brace before sending, which is the bug's blind spot. func TestTheRequestTargetHoldsThroughATLSTunnel(t *testing.T) { From 62050d54559f91069a6925b116cdb57e49caf9ca Mon Sep 17 00:00:00 2001 From: saif <11242541+saifsmailbox98@users.noreply.github.com> Date: Thu, 24 Sep 2026 02:01:16 +0530 Subject: [PATCH 08/43] fix(agent-vault): upload activity only to https links, without redirects or the signed url in logs --- packages/agentvault/activity_ship.go | 38 ++++++++++-- packages/agentvault/activity_ship_test.go | 76 ++++++++++++++++++++++- 2 files changed, 107 insertions(+), 7 deletions(-) diff --git a/packages/agentvault/activity_ship.go b/packages/agentvault/activity_ship.go index edef939f7..b8fabdecf 100644 --- a/packages/agentvault/activity_ship.go +++ b/packages/agentvault/activity_ship.go @@ -6,6 +6,7 @@ import ( "errors" "fmt" "net/http" + "net/url" "strconv" "github.com/Infisical/infisical-merge/packages/api" @@ -67,10 +68,29 @@ func newActivityShipper(proxyToken func() string) (*activityShipperClient, error final: final, // Deliberately not a resty client: this one talks to the customer's bucket with a presigned URL // and must never carry the Infisical Authorization header those two set. - put: &http.Client{Timeout: activityPutTimeout, Transport: http.DefaultTransport.(*http.Transport).Clone()}, + put: &http.Client{ + Timeout: activityPutTimeout, + Transport: http.DefaultTransport.(*http.Transport).Clone(), + // A presigned URL only works on the host it was signed for, so a redirect can never lead to a + // successful upload. Following one could only send the chunk somewhere it was not meant to go. + CheckRedirect: func(*http.Request, []*http.Request) error { return http.ErrUseLastResponse }, + }, }, nil } +// Every presigned S3 upload link is https. Any other link did not come from S3. +var errInsecureUploadURL = errors.New("agent-vault: refusing to upload activity to a link that is not https") + +// scrubURLError drops the URL from a request error. net/http writes the whole URL into the message, and for a +// presigned upload that is a working signature, which the caller logs on every timeout. +func scrubURLError(err error) error { + var urlErr *url.Error + if errors.As(err, &urlErr) { + return fmt.Errorf("agent-vault: could not reach the bucket: %w", urlErr.Err) + } + return err +} + func (c *activityShipperClient) createChunk(final bool, sessionID string, req api.CreateAgentVaultActivityChunkRequest) (api.CreateAgentVaultActivityChunkResponse, error) { client := c.steady if final { @@ -79,10 +99,18 @@ func (c *activityShipperClient) createChunk(final bool, sessionID string, req ap return api.CallCreateAgentVaultActivityChunk(client, sessionID, req) } -func (c *activityShipperClient) putObject(ctx context.Context, url string, ciphertext []byte) error { - req, err := http.NewRequestWithContext(ctx, http.MethodPut, url, bytes.NewReader(ciphertext)) +func (c *activityShipperClient) putObject(ctx context.Context, uploadURL string, ciphertext []byte) error { + target, err := url.Parse(uploadURL) + if err != nil { + return scrubURLError(err) + } + if target.Scheme != "https" { + return errInsecureUploadURL + } + + req, err := http.NewRequestWithContext(ctx, http.MethodPut, uploadURL, bytes.NewReader(ciphertext)) if err != nil { - return err + return scrubURLError(err) } // The presign signs Content-Length in, so it has to match the body exactly. req.ContentLength = int64(len(ciphertext)) @@ -91,7 +119,7 @@ func (c *activityShipperClient) putObject(ctx context.Context, url string, ciphe res, err := c.put.Do(req) if err != nil { - return err + return scrubURLError(err) } defer res.Body.Close() diff --git a/packages/agentvault/activity_ship_test.go b/packages/agentvault/activity_ship_test.go index 6e9fb19ce..c27cddbcf 100644 --- a/packages/agentvault/activity_ship_test.go +++ b/packages/agentvault/activity_ship_test.go @@ -7,6 +7,7 @@ import ( "strconv" "strings" "sync" + "sync/atomic" "testing" "github.com/Infisical/infisical-merge/packages/api" @@ -27,7 +28,7 @@ func TestTheChunkPostCarriesTheProxyTokenAndTheBucketPutDoesNot(t *testing.T) { postedPath string ) - bucket := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + bucket := httptest.NewTLSServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { mu.Lock() putAuth = r.Header.Get("Authorization") putLength = r.Header.Get("Content-Length") @@ -58,6 +59,7 @@ func TestTheChunkPostCarriesTheProxyTokenAndTheBucketPutDoesNot(t *testing.T) { if err != nil { t.Fatal(err) } + shipper.put.Transport = bucket.Client().Transport ciphertext := []byte("sealed-bytes") res, err := shipper.createChunk(false, "sess-1", api.CreateAgentVaultActivityChunkRequest{ @@ -99,7 +101,7 @@ func TestTheChunkPostCarriesTheProxyTokenAndTheBucketPutDoesNot(t *testing.T) { } func TestABucketRefusalIsAnErrorThatNamesNoURL(t *testing.T) { - bucket := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { + bucket := httptest.NewTLSServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { w.WriteHeader(http.StatusForbidden) _, _ = w.Write([]byte("AccessDenied")) })) @@ -109,6 +111,7 @@ func TestABucketRefusalIsAnErrorThatNamesNoURL(t *testing.T) { if err != nil { t.Fatal(err) } + shipper.put.Transport = bucket.Client().Transport err = shipper.putObject(context.Background(), bucket.URL+"/object?X-Amz-Signature=secret", []byte("bytes")) if err == nil { @@ -119,3 +122,72 @@ func TestABucketRefusalIsAnErrorThatNamesNoURL(t *testing.T) { t.Fatalf("the error names the signed url: %q", got) } } + +func TestAnUploadLinkThatIsNotHttpsIsRefused(t *testing.T) { + var hits atomic.Int32 + bucket := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { + hits.Add(1) + w.WriteHeader(http.StatusOK) + })) + defer bucket.Close() + + shipper, err := newActivityShipper(func() string { return "proxy-token" }) + if err != nil { + t.Fatal(err) + } + + if err := shipper.putObject(context.Background(), bucket.URL+"/object", []byte("bytes")); err == nil { + t.Fatal("an http upload link was accepted") + } + if hits.Load() != 0 { + t.Fatal("the proxy sent the chunk to a link that is not https") + } +} + +func TestARedirectFromTheBucketIsNotFollowed(t *testing.T) { + var followed atomic.Int32 + bucket := httptest.NewTLSServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + if r.URL.Path == "/elsewhere" { + followed.Add(1) + w.WriteHeader(http.StatusOK) + return + } + http.Redirect(w, r, "/elsewhere", http.StatusTemporaryRedirect) + })) + defer bucket.Close() + + shipper, err := newActivityShipper(func() string { return "proxy-token" }) + if err != nil { + t.Fatal(err) + } + shipper.put.Transport = bucket.Client().Transport + + if err := shipper.putObject(context.Background(), bucket.URL+"/object", []byte("bytes")); err == nil { + t.Fatal("a redirect was treated as a successful upload") + } + if followed.Load() != 0 { + t.Fatal("the upload followed a redirect") + } +} + +func TestAnUnreachableBucketIsAnErrorThatNamesNoSignature(t *testing.T) { + bucket := httptest.NewTLSServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { + w.WriteHeader(http.StatusOK) + })) + target := bucket.URL + "/object?X-Amz-Signature=secret" + bucket.Close() + + shipper, err := newActivityShipper(func() string { return "proxy-token" }) + if err != nil { + t.Fatal(err) + } + + err = shipper.putObject(context.Background(), target, []byte("bytes")) + if err == nil { + t.Fatal("an upload to a closed server succeeded") + } + // This is the error logged on every S3 timeout, so it is the one most likely to leak the signature. + if strings.Contains(err.Error(), "X-Amz-Signature") { + t.Fatalf("the error names the signed url: %q", err.Error()) + } +} From 92ba96817bb59b24dcb91ada343255ddcd4d1f44 Mon Sep 17 00:00:00 2001 From: saif <11242541+saifsmailbox98@users.noreply.github.com> Date: Thu, 24 Sep 2026 05:04:09 +0530 Subject: [PATCH 09/43] fix(agent-vault): keep agent-chosen method and port out of proxy log lines --- packages/agentvault/policy.go | 3 ++- packages/agentvault/policy_test.go | 11 +++++++++ packages/agentvault/proxy.go | 31 +++++++++++++++++++++--- packages/agentvault/proxy_target_test.go | 17 +++++++++++++ 4 files changed, 57 insertions(+), 5 deletions(-) diff --git a/packages/agentvault/policy.go b/packages/agentvault/policy.go index a3bfbecd7..8d6c47800 100644 --- a/packages/agentvault/policy.go +++ b/packages/agentvault/policy.go @@ -17,7 +17,8 @@ var errBodyUnreadable = errors.New("could not read the request body") func checkServicePolicy(svc *resolvedService, req *http.Request) error { if !svc.allowsMethod(req.Method) { - return fmt.Errorf("service %q does not allow %s: %w", svc.name, req.Method, errPolicyBlocked) + // This lands in the proxy log, so the agent's method is capped here as it is in the record. + return fmt.Errorf("service %q does not allow %s: %w", svc.name, truncateLogged(req.Method, maxLoggedMethodLen), errPolicyBlocked) } if len(svc.allowedPathPrefixes) > 0 { path := requestPath(req) diff --git a/packages/agentvault/policy_test.go b/packages/agentvault/policy_test.go index 1d1ef43b5..515282a2c 100644 --- a/packages/agentvault/policy_test.go +++ b/packages/agentvault/policy_test.go @@ -49,6 +49,17 @@ func TestMethodPolicy(t *testing.T) { } }) + t.Run("a refused method is capped before it reaches the log", func(t *testing.T) { + svc := serviceWithPolicy([]string{"GET"}, nil) + err := checkServicePolicy(svc, requestTo(t, strings.Repeat("A", 1<<20), "/x")) + if !errors.Is(err, errPolicyBlocked) { + t.Fatalf("an unlisted method should be blocked, got %v", err) + } + if len(err.Error()) > 200 { + t.Fatalf("the refusal carries %d bytes; the agent's method would fill the proxy log", len(err.Error())) + } + }) + t.Run("a lower-case method is folded rather than blocked", func(t *testing.T) { svc := serviceWithPolicy([]string{"GET"}, nil) req := requestTo(t, "GET", "/x") diff --git a/packages/agentvault/proxy.go b/packages/agentvault/proxy.go index 422746143..999c12946 100644 --- a/packages/agentvault/proxy.go +++ b/packages/agentvault/proxy.go @@ -47,10 +47,10 @@ const ( maxConcurrentConns = 512 maxLoggedPathLen = 2048 - // The agent chooses the method and the port too, and Go bounds them only by the 1 MiB header limit. - // Uncapped, one request could make its own record big enough to stall whoever reads the session. + // The agent chooses the method too, and Go bounds it only by the 1 MiB header limit. Uncapped, one + // request could make its own record big enough to stall whoever reads the session. The port needs no + // cap: checkedTarget refuses anything that is not a plain port number. maxLoggedMethodLen = 32 - maxLoggedPortLen = 16 ) var errHostBlocked = errors.New("host blocked by policy") @@ -436,7 +436,7 @@ func (ps *proxyServer) forwardHTTP(w http.ResponseWriter, r *http.Request, schem ps.activity.record(outcome.activity, activityRecord{ Method: reqMethod, Host: hostname, - Port: truncateLogged(port, maxLoggedPortLen), + Port: port, Path: reqPath, Status: status, Decision: decision, @@ -656,12 +656,35 @@ const ( var errHostTooLong = errors.New("the target host is longer than a DNS name can be") +var errBadPort = errors.New("the target port must be a number from 1 to 65535") + +// validPort accepts a port only in its plain form. The dialer reads a megabyte of leading zeros, or a '+', +// as an ordinary port, so an agent could otherwise reach 443 under a port string that fills log lines, +// shows in its own record as noise, and misses every service pattern, which compares ports as strings. +func validPort(port string) bool { + if len(port) == 0 || len(port) > 5 || port[0] == '0' { + return false + } + n := 0 + for i := 0; i < len(port); i++ { + c := port[i] + if c < '0' || c > '9' { + return false + } + n = n*10 + int(c-'0') + } + return n <= 65535 +} + func checkedTarget(hostname, port string) (string, string, error) { // Normalised first: a host of only dots is non-empty until the trailing dots come off. hostname = normalizeHostname(hostname) if hostname == "" || port == "" { return "", "", errNoHostInTarget } + if !validPort(port) { + return "", "", errBadPort + } if net.ParseIP(hostname) == nil { if len(hostname) > maxHostnameBytes { return "", "", errHostTooLong diff --git a/packages/agentvault/proxy_target_test.go b/packages/agentvault/proxy_target_test.go index 48cb8114a..df91c1e1d 100644 --- a/packages/agentvault/proxy_target_test.go +++ b/packages/agentvault/proxy_target_test.go @@ -18,6 +18,23 @@ func TestTargetsWithoutAHostAreRefused(t *testing.T) { } } +func TestOnlyPlainPortNumbersAreAccepted(t *testing.T) { + for _, port := range []string{"0", "0443", "+443", "99999", "65536", "44a", strings.Repeat("0", 1<<20) + "443"} { + target := "api.example.com:" + port + if _, _, err := parseConnectTarget(target); err == nil { + t.Errorf("parseConnectTarget accepted port %.20q", port) + } + if _, _, err := parseForwardTarget(target); err == nil { + t.Errorf("parseForwardTarget accepted port %.20q", port) + } + } + for _, port := range []string{"1", "80", "443", "8443", "65535"} { + if _, _, err := parseConnectTarget("api.example.com:" + port); err != nil { + t.Errorf("parseConnectTarget refused port %q: %v", port, err) + } + } +} + func TestOrdinaryTargetsStillParse(t *testing.T) { for _, tc := range []struct { target, host, connectPort, forwardPort string From 5f92a4e0e4848caa3568443bd623d1220d2307a5 Mon Sep 17 00:00:00 2001 From: saif <11242541+saifsmailbox98@users.noreply.github.com> Date: Thu, 24 Sep 2026 05:04:33 +0530 Subject: [PATCH 10/43] fix(agent-vault): leave a posted chunk's drop count to its row --- packages/agentvault/activity_spool.go | 7 +++---- packages/agentvault/activity_test.go | 6 +++--- 2 files changed, 6 insertions(+), 7 deletions(-) diff --git a/packages/agentvault/activity_spool.go b/packages/agentvault/activity_spool.go index b09d7a289..b8bbaf9dc 100644 --- a/packages/agentvault/activity_spool.go +++ b/packages/agentvault/activity_spool.go @@ -127,12 +127,11 @@ type sealedChunk struct { } // lostCount is what a chunk that will never be uploaded adds to its session's gap. Once the POST succeeded -// the row exists, and the viewer already shows those records as a batch it cannot read, so counting them -// again would report one loss twice. The drop count the chunk carried is shown nowhere else, so it always -// comes back. +// the row exists, and it already reports both halves: the viewer shows its records as a batch it cannot +// read, and its drop count from the row itself. Counting either again would report one loss twice. func (c *sealedChunk) lostCount() uint64 { if c.posted { - return c.meta.DroppedCount + return 0 } return c.meta.DroppedCount + uint64(c.meta.RecordCount) } diff --git a/packages/agentvault/activity_test.go b/packages/agentvault/activity_test.go index f84fbde42..d3d6a617e 100644 --- a/packages/agentvault/activity_test.go +++ b/packages/agentvault/activity_test.go @@ -583,9 +583,9 @@ func TestTheByteCapEvictsTheOldestChunkOnTheProxy(t *testing.T) { if got := log.spools["oldest"].ring.dropped; got != 107 { t.Fatalf("the unposted chunk counted %d dropped, expected 107", got) } - // Posted, so the viewer already shows its records as unreadable. Only the carried drops come back. - if got := log.spools["posted"].ring.dropped; got != 3 { - t.Fatalf("the posted chunk counted %d dropped, expected 3", got) + // Posted, so its row already reports its records as unreadable and its drops as not recorded. + if got := log.spools["posted"].ring.dropped; got != 0 { + t.Fatalf("the posted chunk counted %d dropped, expected 0", got) } } From b96369d964dfd21e0dbcd0747d463ba6cd7ece5d Mon Sep 17 00:00:00 2001 From: saif <11242541+saifsmailbox98@users.noreply.github.com> Date: Thu, 24 Sep 2026 07:17:06 +0530 Subject: [PATCH 11/43] fix(agent-vault): send activity uploads as create-only and accept one already stored --- packages/agentvault/activity_ship.go | 7 +++++++ packages/agentvault/activity_ship_test.go | 24 +++++++++++++++++++++++ 2 files changed, 31 insertions(+) diff --git a/packages/agentvault/activity_ship.go b/packages/agentvault/activity_ship.go index b8fabdecf..1cd729549 100644 --- a/packages/agentvault/activity_ship.go +++ b/packages/agentvault/activity_ship.go @@ -116,6 +116,8 @@ func (c *activityShipperClient) putObject(ctx context.Context, uploadURL string, req.ContentLength = int64(len(ciphertext)) req.Header.Set("Content-Type", "application/octet-stream") req.Header.Set("Content-Length", strconv.Itoa(len(ciphertext))) + // Signed in too: the link is create-only, so it can finish an upload but never replace a stored chunk. + req.Header.Set("If-None-Match", "*") res, err := c.put.Do(req) if err != nil { @@ -123,6 +125,11 @@ func (c *activityShipperClient) putObject(ctx context.Context, uploadURL string, } defer res.Body.Close() + // The object is already there: an earlier PUT landed but its response never arrived. The chunk is stored, + // which is all a retry wanted. + if res.StatusCode == http.StatusPreconditionFailed { + return nil + } if res.StatusCode < 200 || res.StatusCode >= 300 { // The body can carry an S3 error document; the status is enough to decide, and the URL is signed // so it never goes in a log line. diff --git a/packages/agentvault/activity_ship_test.go b/packages/agentvault/activity_ship_test.go index c27cddbcf..2878ff898 100644 --- a/packages/agentvault/activity_ship_test.go +++ b/packages/agentvault/activity_ship_test.go @@ -24,6 +24,7 @@ func TestTheChunkPostCarriesTheProxyTokenAndTheBucketPutDoesNot(t *testing.T) { putAuth string putLength string putType string + putIfNone string putBody []byte postedPath string ) @@ -33,6 +34,7 @@ func TestTheChunkPostCarriesTheProxyTokenAndTheBucketPutDoesNot(t *testing.T) { putAuth = r.Header.Get("Authorization") putLength = r.Header.Get("Content-Length") putType = r.Header.Get("Content-Type") + putIfNone = r.Header.Get("If-None-Match") buf := make([]byte, r.ContentLength) _, _ = r.Body.Read(buf) putBody = buf @@ -95,6 +97,10 @@ func TestTheChunkPostCarriesTheProxyTokenAndTheBucketPutDoesNot(t *testing.T) { if putType != "application/octet-stream" { t.Fatalf("the upload declared Content-Type %q", putType) } + // Signed into the link by Infisical, so S3 refuses the upload unless it is sent. + if putIfNone != "*" { + t.Fatalf("the upload sent If-None-Match %q; it must be create-only", putIfNone) + } if string(putBody) != string(ciphertext) { t.Fatalf("the bucket received %q", string(putBody)) } @@ -191,3 +197,21 @@ func TestAnUnreachableBucketIsAnErrorThatNamesNoSignature(t *testing.T) { t.Fatalf("the error names the signed url: %q", err.Error()) } } + +func TestAChunkAlreadyStoredCountsAsUploaded(t *testing.T) { + // S3 answers 412 to a create-only PUT when the object exists, which means an earlier attempt landed. + bucket := httptest.NewTLSServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { + w.WriteHeader(http.StatusPreconditionFailed) + })) + defer bucket.Close() + + shipper, err := newActivityShipper(func() string { return "proxy-token" }) + if err != nil { + t.Fatal(err) + } + shipper.put.Transport = bucket.Client().Transport + + if err := shipper.putObject(context.Background(), bucket.URL+"/object", []byte("bytes")); err != nil { + t.Fatalf("a chunk that is already stored was treated as a failed upload: %v", err) + } +} From 15ba4ae34a918cdaca1dd3199bc4a5fd28f7cf4f Mon Sep 17 00:00:00 2001 From: saif <11242541+saifsmailbox98@users.noreply.github.com> Date: Thu, 24 Sep 2026 07:50:15 +0530 Subject: [PATCH 12/43] chore(agent-vault): drop comments that restate the code --- packages/agentvault/activity.go | 81 ++----------------- packages/agentvault/activity_crypto.go | 20 +---- packages/agentvault/activity_crypto_test.go | 5 -- packages/agentvault/activity_resolve_test.go | 8 -- packages/agentvault/activity_ship.go | 31 +------ packages/agentvault/activity_ship_test.go | 10 --- packages/agentvault/activity_spool.go | 38 +-------- packages/agentvault/activity_test.go | 44 +--------- packages/agentvault/activity_wiring_test.go | 18 ----- packages/agentvault/cache.go | 9 --- packages/agentvault/policy.go | 1 - packages/agentvault/proxy.go | 25 ++---- .../agentvault/proxy_requesttarget_test.go | 5 +- packages/agentvault/resolve.go | 9 +-- packages/agentvault/run.go | 2 - packages/api/agent_vault.go | 6 -- 16 files changed, 24 insertions(+), 288 deletions(-) diff --git a/packages/agentvault/activity.go b/packages/agentvault/activity.go index 0de5cf1d0..18c345733 100644 --- a/packages/agentvault/activity.go +++ b/packages/agentvault/activity.go @@ -10,51 +10,32 @@ import ( ) const ( - // The cost knob is this interval, not traffic volume: one flush is one S3 PUT. At 60s a session running - // flat out costs about 1,440 PUTs a day; flushing every 5s would be twelve times the bill for the same - // records. - activityFlushInterval = 60 * time.Second - // The server refuses a chunk over this, so the ring is drained in slices of at most this many. - activityFlushRecords = 1000 - // The server also refuses a chunk over 8 MiB. The agent controls how long its paths are and JSON - // escapes some bytes six to one, so a full slice can pass that. Half the limit leaves ordinary - // flushes whole. + activityFlushInterval = 60 * time.Second + activityFlushRecords = 1000 activityMaxChunkPlaintext = 4 << 20 - // Five missed flushes of headroom per session before the oldest records start being overwritten. activitySpoolCapacity = 5000 - // A fuse across every session on this proxy, roughly 60 MB of records. activityTotalCapacity = 200_000 - // Sealed chunks kept per spool while shipping fails. One chunk seals per tick during an outage, so this - // is about ten minutes of Infisical or S3 being unreachable before a busy session loses its oldest - // sealed chunk. Deliberate for a preview with no disk persistence: raise this rather than the interval. - activityPendingChunks = 10 - // A second fuse, on sealed ciphertext rather than record count, across every spool. + activityPendingChunks = 10 activityTotalSealedBytes = 64 << 20 - // Longer than sessionInactiveTTL, so a session evicted from the cache still gets its final flush. activityIdleClose = 15 * time.Minute activityPauseBackoff = 15 * time.Minute activityPutTimeout = 10 * time.Second activityFinalTimeout = 3 * time.Second - // Read off APIError.Name. Defined by the backend in agent-vault-activity-constants.ts. activityCeilingReachedName = "AgentVaultActivityCeilingReached" activityDisabledName = "AgentVaultActivityDisabled" ) -// activityGrant is what resolve hands back when logging is on for a session. A nil grant means "do not -// record", which is the whole of the disabled path. type activityGrant struct { sessionID string projectID string key []byte } -// activityShipper is the seam the tests replace. Two calls, because delivery is two steps: Infisical -// writes the index row and returns a presigned URL, then the bytes go straight to the customer's bucket. type activityShipper interface { createChunk(final bool, sessionID string, req api.CreateAgentVaultActivityChunkRequest) (api.CreateAgentVaultActivityChunkResponse, error) putObject(ctx context.Context, url string, ciphertext []byte) error @@ -65,30 +46,21 @@ type activityLog struct { shipper activityShipper now func() time.Time - // Held for the whole of flushAll. Shutdown calls it from a second goroutine while the run loop may - // still be inside one, and the two would otherwise ship the same chunk twice and race on its fields. + // close's flushAll can overlap the run loop's; unguarded, one chunk ships twice. flushMu sync.Mutex mu sync.Mutex spools map[string]*activitySpool - // A session's sequence numbers have to keep climbing across the spool being forgotten and rebuilt, - // or one proxy emits two records with the same (proxyId, seq) for one session. Cleared wholesale - // when it grows, the way the session cache handles its own refusal map. seqBySession map[string]uint64 total int sealedBytes int nextSealOrder uint64 - // Proxy-wide, because both reasons are proxy-wide: the ceiling is per organization and the switch is - // per project, and this proxy serves one project. pauseUntil time.Time pauseReason string - // Reset at the top of every flushAll. Once either side has failed once in a tick, the remaining spools - // seal but skip both calls, so a hundred spools against a blocked egress or an unreachable control - // plane cost one timeout rather than a hundred. s3Down bool infisicalDown bool @@ -107,10 +79,6 @@ func newActivityLog(proxyID string, shipper activityShipper) *activityLog { } } -// record is the entire hot-path cost: one append under a mutex. No I/O, no crypto. -// -// Nil-safe on both the receiver and the grant, so a bare &proxyServer{} test fixture and a session whose -// logging is off both cost a single comparison. func (a *activityLog) record(g *activityGrant, rec activityRecord) { if a == nil || g == nil { return @@ -129,8 +97,6 @@ func (a *activityLog) record(g *activityGrant, rec activityRecord) { a.spools[g.sessionID] = spool } - // A sequence number is consumed even while paused or full, so the gap is counted rather than silent - // and a chunk's firstSeq reveals exactly how many records are missing before it. rec.Seq = spool.nextSeq spool.nextSeq++ rec.ProxyID = a.proxyID @@ -143,8 +109,6 @@ func (a *activityLog) record(g *activityGrant, rec activityRecord) { } if a.total >= activityTotalCapacity { - // The newest is dropped rather than the oldest: at the proxy-wide fuse the ring's own eviction is - // already running, and dropping the newest keeps one bounded behaviour rather than two. spool.ring.dropped++ return } @@ -161,7 +125,6 @@ func (a *activityLog) record(g *activityGrant, rec activityRecord) { } } -// run is a single loop, so a flush can never overlap itself and no same-session guard is needed. func (a *activityLog) run(stop <-chan struct{}) { if a == nil { return @@ -181,7 +144,6 @@ func (a *activityLog) run(stop <-chan struct{}) { } } -// close stops recording and makes one last attempt to ship everything buffered, within the deadline on ctx. func (a *activityLog) close(ctx context.Context) { if a == nil { return @@ -206,7 +168,6 @@ func (a *activityLog) close(ctx context.Context) { } } -// dueSpools picks what to flush and forgets idle spools, under the lock. Flushing then happens off it. func (a *activityLog) dueSpools(final bool) []*activitySpool { a.mu.Lock() defer a.mu.Unlock() @@ -215,8 +176,6 @@ func (a *activityLog) dueSpools(final bool) []*activitySpool { due := make([]*activitySpool, 0, len(a.spools)) for id, spool := range a.spools { if spool.ring.len() == 0 && len(spool.pending) == 0 { - // A spool is independent of the session cache: eviction there just means record() stops - // arriving, and this is what eventually forgets it. if !final && now.Sub(spool.lastRecordAt) > activityIdleClose { a.forgetSpoolLocked(id, spool) } @@ -230,7 +189,6 @@ func (a *activityLog) dueSpools(final bool) []*activitySpool { return due } -// forgetSpoolLocked drops a spool but keeps where its sequence numbers had reached. func (a *activityLog) forgetSpoolLocked(sessionID string, spool *activitySpool) { if len(a.seqBySession) >= maxSessionCacheEntries { a.seqBySession = make(map[string]uint64) @@ -260,7 +218,6 @@ func (a *activityLog) flushAll(ctx context.Context, final bool) { return } - // One goroutine in flushAll at a time: the run loop and shutdown both call it. a.flushMu.Lock() defer a.flushMu.Unlock() @@ -270,16 +227,10 @@ func (a *activityLog) flushAll(ctx context.Context, final bool) { a.mu.Unlock() for _, spool := range a.dueSpools(final) { - // Sequential and off the lock. At one chunk per session per minute and ~100ms per PUT, a hundred - // sessions finish well inside a 60s tick, and one loop means no same-session overlap to guard. a.flushSpool(ctx, spool, final) } } -// sealRing drains the ring into sealed chunks, in slices the server will accept. -// -// The marshal and the AES pass run off the lock. They are only milliseconds, but it is the same lock -// every proxied request takes to append a record, so holding it across them would stall the request path. func (a *activityLog) sealRing(spool *activitySpool) { for { a.mu.Lock() @@ -290,7 +241,6 @@ func (a *activityLog) sealRing(spool *activitySpool) { return } a.total -= len(records) - // droppedCount rides the first chunk only, so one gap is reported once. dropped := spool.ring.takeDropped() now := a.now() a.mu.Unlock() @@ -323,8 +273,6 @@ func (a *activityLog) sealRing(spool *activitySpool) { } } -// dropUnsealed accounts for records that could not be sealed. They are gone, so they join the gap rather -// than vanishing from the count with it. func (a *activityLog) dropUnsealed(spool *activitySpool, records int, dropped uint64, err error) { a.mu.Lock() spool.ring.dropped += dropped + uint64(records) @@ -333,11 +281,6 @@ func (a *activityLog) dropUnsealed(spool *activitySpool, records int, dropped ui Msg("agent-vault: could not seal an activity chunk, dropping those records") } -// enforcePendingCapsLocked runs on every append, because the caps are a property of `pending` rather than -// a branch of the ceiling handler. Both evict the oldest sealed chunk and count what it held as dropped. -// -// The byte cap is proxy-wide, so it evicts the oldest chunk on the proxy wherever it sits, a session's only -// chunk included. Sparing every session its last chunk let an outage hold one per session with no bound. func (a *activityLog) enforcePendingCapsLocked(spool *activitySpool) { for len(spool.pending) > activityPendingChunks { a.evictOldestLocked(spool) @@ -345,15 +288,12 @@ func (a *activityLog) enforcePendingCapsLocked(spool *activitySpool) { for a.sealedBytes > activityTotalSealedBytes { victim := a.oldestPendingLocked() if victim == nil { - // Unreachable while sealedBytes counts only pending chunks. Guarded anyway: this runs under the - // lock every proxied request takes to record. return } a.evictOldestLocked(victim) } } -// oldestPendingLocked scans every spool, which is fine: it only runs once the proxy is over its byte cap. func (a *activityLog) oldestPendingLocked() *activitySpool { var oldest *activitySpool for _, spool := range a.spools { @@ -381,8 +321,6 @@ func (a *activityLog) evictOldestLocked(spool *activitySpool) { func (a *activityLog) flushSpool(ctx context.Context, spool *activitySpool, final bool) { if paused, reason := a.paused(); paused { - // Seal once at pause onset so the ring does not overflow while waiting, then hold everything: ops - // may raise the ceiling within the window, and the chunks are still shippable when it lifts. a.sealRing(spool) _ = reason return @@ -412,7 +350,6 @@ func (a *activityLog) flushSpool(ctx context.Context, spool *activitySpool, fina } } -// shipChunk delivers one chunk. It returns false when this spool's loop should stop for the tick. func (a *activityLog) shipChunk(ctx context.Context, spool *activitySpool, chunk *sealedChunk, final bool) bool { if chunk.uploadURL == "" || a.now().Add(10*time.Second).After(chunk.urlExpires) { res, err := a.shipper.createChunk(final, spool.sessionID, chunk.meta) @@ -432,8 +369,6 @@ func (a *activityLog) shipChunk(ctx context.Context, spool *activitySpool, chunk } if err := a.shipper.putObject(putCtx, chunk.uploadURL, chunk.ciphertext); err != nil { - // The row already exists, so re-POSTing the same chunk id replays idempotently and yields a fresh - // url. Clearing it is what makes the next tick do that. chunk.uploadURL = "" a.mu.Lock() a.s3Down = true @@ -449,7 +384,6 @@ func (a *activityLog) shipChunk(ctx context.Context, spool *activitySpool, chunk func (a *activityLog) handleCreateFailure(spool *activitySpool, chunk *sealedChunk, err error) bool { switch { case isProxyTokenRejected(err): - // The poll loop exits within two heartbeats. Keep everything until it does. log.Warn().Err(err).Msg("agent-vault: Infisical rejected this proxy's token, holding activity") return false @@ -459,7 +393,6 @@ func (a *activityLog) handleCreateFailure(spool *activitySpool, chunk *sealedChu for _, held := range spool.pending { a.sealedBytes -= len(held.ciphertext) } - // The session is gone for good, so its sequence numbers are not worth remembering. a.forgetSpoolLocked(spool.sessionID, spool) delete(a.seqBySession, spool.sessionID) a.mu.Unlock() @@ -468,7 +401,6 @@ func (a *activityLog) handleCreateFailure(spool *activitySpool, chunk *sealedChu case isActivityErrorNamed(err, activityCeilingReachedName): a.pause(activityCeilingReachedName) - // An error, not a warning: recording has stopped for the whole organization until Infisical acts. log.Error().Err(err).Msg("agent-vault: activity logging has reached its limit for this organization, retrying in 15m") return false @@ -478,13 +410,11 @@ func (a *activityLog) handleCreateFailure(spool *activitySpool, chunk *sealedChu return false case isPoisonChunk(err): - // The server will never accept this chunk, so retrying costs the whole spool. a.mu.Lock() if len(spool.pending) > 0 && spool.pending[0] == chunk { spool.pending = spool.pending[1:] a.sealedBytes -= len(chunk.ciphertext) - // Otherwise a refused chunk leaves no trace in the timeline, and an agent that can get its own - // chunk refused can erase what it did. + // Counted as dropped, or an agent that gets its own chunk refused could erase what it did. spool.ring.dropped += chunk.lostCount() } a.mu.Unlock() @@ -493,7 +423,6 @@ func (a *activityLog) handleCreateFailure(spool *activitySpool, chunk *sealedChu return false default: - // A timeout, a 5xx or a 429. Whatever it is, the rest of this tick will meet it too. a.mu.Lock() a.infisicalDown = true a.mu.Unlock() diff --git a/packages/agentvault/activity_crypto.go b/packages/agentvault/activity_crypto.go index 4a9a75505..7d10e0a22 100644 --- a/packages/agentvault/activity_crypto.go +++ b/packages/agentvault/activity_crypto.go @@ -13,30 +13,20 @@ import ( "github.com/oklog/ulid" ) -// The version suffix on the additional authenticated data. Bumping it makes every existing chunk -// undecryptable, so it changes only alongside a migration of the stored objects. const activityAADVersion = "v1" const activityIVBytes = 12 -// buildActivityAAD binds a sealed chunk to exactly one place in the hierarchy, so holding the session key -// is not enough to replay a chunk under another project, session or proxy. -// -// The same string is built by the browser in frontend/src/hooks/api/agentVault/activityDecrypt.ts, and the -// backend pins a known-good vector in agent-vault-activity-crypto.test.ts. All three must agree or -// playback fails with no useful error. +// Must byte-match frontend activityDecrypt.ts and the vector pinned in agent-vault-activity-crypto.test.ts. func buildActivityAAD(projectID, sessionID, proxyID, chunkID string) []byte { sum := sha256.Sum256([]byte(fmt.Sprintf("%s|%s|%s|%s|%s", projectID, sessionID, proxyID, chunkID, activityAADVersion))) return sum[:] } -// sealActivity produces the layout Web Crypto's decrypt expects: a 12-byte IV carried beside the object, -// and the 16-byte GCM tag appended to the ciphertext rather than kept separately. func sealActivity(key, plaintext, aad []byte) (ciphertext []byte, iv []byte, err error) { return sealActivityWithRand(rand.Reader, key, plaintext, aad) } -// sealActivityWithRand takes the IV source so a test can pin one and compare against the backend's vector. func sealActivityWithRand(random io.Reader, key, plaintext, aad []byte) (ciphertext []byte, iv []byte, err error) { block, err := aes.NewCipher(key) if err != nil { @@ -52,27 +42,19 @@ func sealActivityWithRand(random io.Reader, key, plaintext, aad []byte) (ciphert return nil, nil, fmt.Errorf("agent-vault: could not read a nonce: %w", err) } - // A nil destination makes Seal allocate, so the output is exactly ciphertext||tag with no IV prefix. return gcm.Seal(nil, iv, plaintext, aad), iv, nil } -// encodeActivityIV matches the backend's `^[A-Za-z0-9+/]{16}$`: standard alphabet, no padding. func encodeActivityIV(iv []byte) string { return base64.RawStdEncoding.EncodeToString(iv) } -// newActivityChunkID mints a ULID, which sorts by time and is unique per session. A proxy-side counter -// cannot be used: it resets whenever the session cache evicts an entry, which happens at nine ordinary -// sites, and would then collide with the server's unique index for the rest of the session's life. func newActivityChunkID(now time.Time) string { return ulid.MustNew(ulid.Timestamp(now), newULIDEntropy()).String() } -// ulid.Monotonic is deliberately not used: it keeps state per reader, and two goroutines sealing in the -// same millisecond would need a mutex around it for no benefit. 80 bits of randomness is ample here. func newULIDEntropy() io.Reader { return rand.Reader } -// Kept so a test can assert the id is well formed without reaching for the library. func parseActivityChunkID(id string) (time.Time, error) { parsed, err := ulid.Parse(id) if err != nil { diff --git a/packages/agentvault/activity_crypto_test.go b/packages/agentvault/activity_crypto_test.go index 97194dab1..6ff41b448 100644 --- a/packages/agentvault/activity_crypto_test.go +++ b/packages/agentvault/activity_crypto_test.go @@ -11,9 +11,6 @@ import ( "time" ) -// The fixture the backend pins in agent-vault-activity-crypto.test.ts. Three implementations seal or open -// these bytes (this one, Infisical's reference, and the browser's), so a change to the AAD string, the IV -// width or the tag placement has to fail somewhere rather than surface as "playback is broken". const ( vectorKeyHex = "000102030405060708090a0b0c0d0e0f101112131415161718191a1b1c1d1e1f" vectorIVHex = "aabbccddeeff001122334455" @@ -83,7 +80,6 @@ func TestSealMatchesNodeVector(t *testing.T) { } } -// What the browser does, so the layout is proven openable rather than merely reproducible. func TestSealedChunkOpensWithTheTagAppended(t *testing.T) { key := mustHex(t, vectorKeyHex) aad := buildActivityAAD(vectorContext.projectID, vectorContext.sessionID, vectorContext.proxyID, vectorContext.chunkID) @@ -160,7 +156,6 @@ func TestChunkIDsAreULIDsThatSortByTime(t *testing.T) { if len(earlier) != 26 { t.Fatalf("a chunk id is %d characters, the server's column is 26", len(earlier)) } - // The read cursor is a plain string comparison on this column, so lexical order has to be time order. if !(earlier < later) { t.Fatalf("%q did not sort before %q", earlier, later) } diff --git a/packages/agentvault/activity_resolve_test.go b/packages/agentvault/activity_resolve_test.go index 28bbe9bdd..a4964b115 100644 --- a/packages/agentvault/activity_resolve_test.go +++ b/packages/agentvault/activity_resolve_test.go @@ -34,10 +34,6 @@ func TestTheFirstResolveTakesTheKeyOffTheWire(t *testing.T) { } } -// The key is sent exactly once per session, because unwrapping it costs a KMS round trip. Every poll -// after the first answers with an empty sessionKey, and the cached copy has to be carried onto the -// refreshed entry. Getting this backwards silently stops all logging after the first poll, which is why -// it has a test of its own. func TestACachedKeySurvivesAResolveThatOmitsIt(t *testing.T) { held := &activityGrant{sessionID: "s1", projectID: "proj-1", key: aKey(9)} @@ -52,8 +48,6 @@ func TestACachedKeySurvivesAResolveThatOmitsIt(t *testing.T) { } func TestNoKeyAndNoCachedCopyMeansNoRecording(t *testing.T) { - // The proxy said it had no key and was sent none, so there is nothing to seal with. Recording - // anything here would produce chunks nobody can ever open. if got := toActivityGrant("s1", enabledGrantWire(""), nil); got != nil { t.Fatal("a grant was built with no key at all") } @@ -62,7 +56,6 @@ func TestNoKeyAndNoCachedCopyMeansNoRecording(t *testing.T) { func TestActivityBeingOffClearsAnyCachedGrant(t *testing.T) { held := &activityGrant{sessionID: "s1", projectID: "proj-1", key: aKey(9)} - // An admin switching logging off has to reach a running proxy on its next poll. if got := toActivityGrant("s1", api.AgentVaultActivityGrant{Enabled: false}, held); got != nil { t.Fatal("the proxy kept recording after logging was switched off") } @@ -84,7 +77,6 @@ func TestAnUnusableKeyIsRefusedRatherThanUsed(t *testing.T) { } func TestAGrantWithoutAProjectIsRefused(t *testing.T) { - // The project id is part of the AAD, so a chunk sealed without it could never be opened. wire := api.AgentVaultActivityGrant{Enabled: true, SessionKey: base64.StdEncoding.EncodeToString(aKey(7))} if got := toActivityGrant("s1", wire, nil); got != nil { t.Fatal("a grant was built with no project named") diff --git a/packages/agentvault/activity_ship.go b/packages/agentvault/activity_ship.go index 1cd729549..9ae615ca4 100644 --- a/packages/agentvault/activity_ship.go +++ b/packages/agentvault/activity_ship.go @@ -14,18 +14,11 @@ import ( "github.com/go-resty/resty/v2" ) -// isActivityErrorNamed matches the two named refusals the backend raises for activity. Both mean "stop -// asking for a while" rather than "this chunk is bad". func isActivityErrorNamed(err error, name string) bool { var apiErr *api.APIError return errors.As(err, &apiErr) && apiErr.Name == name } -// isPoisonChunk is a 4xx the server will never accept: a schema failure or a chunk whose own numbers -// contradict each other. Retrying one costs the spool behind it, so it is dropped instead. -// -// 401, 404 and the two named refusals are handled before this is reached, and 429 is deliberately not -// here: it is a "later", not a refusal. func isPoisonChunk(err error) bool { var apiErr *api.APIError if !errors.As(err, &apiErr) { @@ -37,7 +30,6 @@ func isPoisonChunk(err error) bool { return apiErr.StatusCode >= 400 && apiErr.StatusCode < 500 } -// activityShipperClient carries three clients because the three calls want three different policies. type activityShipperClient struct { steady *resty.Client final *resty.Client @@ -45,16 +37,12 @@ type activityShipperClient struct { } func newActivityShipper(proxyToken func() string) (*activityShipperClient, error) { - // No retries on the steady path: a failed create is retried by the next tick, which is the same - // backoff with none of the risk of piling requests onto a struggling control plane. steady, err := util.GetRestyClientWithPolicy(util.RetryPolicy{}) if err != nil { return nil, err } steady.SetAuthToken(proxyToken()).SetTimeout(controlPlaneTimeout) - // Shutdown gets one retry and a short deadline: there is no next tick, and creating a chunk is - // idempotent by chunk id, so a replay cannot double-write. finalPolicy := util.BestEffortRetryPolicy() finalPolicy.ReplaySafe = true final, err := util.GetRestyClientWithPolicy(finalPolicy) @@ -66,23 +54,17 @@ func newActivityShipper(proxyToken func() string) (*activityShipperClient, error return &activityShipperClient{ steady: steady, final: final, - // Deliberately not a resty client: this one talks to the customer's bucket with a presigned URL - // and must never carry the Infisical Authorization header those two set. + // Not resty: this goes to the customer's bucket and must never carry the Infisical auth token. put: &http.Client{ - Timeout: activityPutTimeout, - Transport: http.DefaultTransport.(*http.Transport).Clone(), - // A presigned URL only works on the host it was signed for, so a redirect can never lead to a - // successful upload. Following one could only send the chunk somewhere it was not meant to go. + Timeout: activityPutTimeout, + Transport: http.DefaultTransport.(*http.Transport).Clone(), CheckRedirect: func(*http.Request, []*http.Request) error { return http.ErrUseLastResponse }, }, }, nil } -// Every presigned S3 upload link is https. Any other link did not come from S3. var errInsecureUploadURL = errors.New("agent-vault: refusing to upload activity to a link that is not https") -// scrubURLError drops the URL from a request error. net/http writes the whole URL into the message, and for a -// presigned upload that is a working signature, which the caller logs on every timeout. func scrubURLError(err error) error { var urlErr *url.Error if errors.As(err, &urlErr) { @@ -112,11 +94,9 @@ func (c *activityShipperClient) putObject(ctx context.Context, uploadURL string, if err != nil { return scrubURLError(err) } - // The presign signs Content-Length in, so it has to match the body exactly. req.ContentLength = int64(len(ciphertext)) req.Header.Set("Content-Type", "application/octet-stream") req.Header.Set("Content-Length", strconv.Itoa(len(ciphertext))) - // Signed in too: the link is create-only, so it can finish an upload but never replace a stored chunk. req.Header.Set("If-None-Match", "*") res, err := c.put.Do(req) @@ -125,14 +105,11 @@ func (c *activityShipperClient) putObject(ctx context.Context, uploadURL string, } defer res.Body.Close() - // The object is already there: an earlier PUT landed but its response never arrived. The chunk is stored, - // which is all a retry wanted. + // If-None-Match hit: an earlier PUT stored this chunk but its response was lost. if res.StatusCode == http.StatusPreconditionFailed { return nil } if res.StatusCode < 200 || res.StatusCode >= 300 { - // The body can carry an S3 error document; the status is enough to decide, and the URL is signed - // so it never goes in a log line. return fmt.Errorf("agent-vault: the bucket refused the upload with status %d", res.StatusCode) } return nil diff --git a/packages/agentvault/activity_ship_test.go b/packages/agentvault/activity_ship_test.go index 2878ff898..c4d9ab66e 100644 --- a/packages/agentvault/activity_ship_test.go +++ b/packages/agentvault/activity_ship_test.go @@ -14,9 +14,6 @@ import ( "github.com/Infisical/infisical-merge/packages/config" ) -// These swap the config.INFISICAL_URL global, so they cannot run in parallel with each other or with the -// other API-level tests in this package. - func TestTheChunkPostCarriesTheProxyTokenAndTheBucketPutDoesNot(t *testing.T) { var ( mu sync.Mutex @@ -85,19 +82,15 @@ func TestTheChunkPostCarriesTheProxyTokenAndTheBucketPutDoesNot(t *testing.T) { if postedPath != "/api/v1/agent-vault/proxy/sessions/sess-1/activity/chunks" { t.Fatalf("posted to %q", postedPath) } - // The presigned URL is itself the authorization. Sending the proxy's bearer token to a customer's - // bucket would hand a third party a working Infisical credential. if putAuth != "" { t.Fatalf("the bucket saw an Authorization header: %q", putAuth) } - // The presign signs Content-Length in, so a mismatch is refused by S3. if putLength != strconv.Itoa(len(ciphertext)) { t.Fatalf("the upload declared Content-Length %q for %d bytes", putLength, len(ciphertext)) } if putType != "application/octet-stream" { t.Fatalf("the upload declared Content-Type %q", putType) } - // Signed into the link by Infisical, so S3 refuses the upload unless it is sent. if putIfNone != "*" { t.Fatalf("the upload sent If-None-Match %q; it must be create-only", putIfNone) } @@ -123,7 +116,6 @@ func TestABucketRefusalIsAnErrorThatNamesNoURL(t *testing.T) { if err == nil { t.Fatal("a 403 from the bucket was treated as a successful upload") } - // A presigned URL carries a working signature, so it must never reach a log line. if got := err.Error(); strings.Contains(got, "X-Amz-Signature") || strings.Contains(got, bucket.URL) { t.Fatalf("the error names the signed url: %q", got) } @@ -192,14 +184,12 @@ func TestAnUnreachableBucketIsAnErrorThatNamesNoSignature(t *testing.T) { if err == nil { t.Fatal("an upload to a closed server succeeded") } - // This is the error logged on every S3 timeout, so it is the one most likely to leak the signature. if strings.Contains(err.Error(), "X-Amz-Signature") { t.Fatalf("the error names the signed url: %q", err.Error()) } } func TestAChunkAlreadyStoredCountsAsUploaded(t *testing.T) { - // S3 answers 412 to a create-only PUT when the object exists, which means an earlier attempt landed. bucket := httptest.NewTLSServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { w.WriteHeader(http.StatusPreconditionFailed) })) diff --git a/packages/agentvault/activity_spool.go b/packages/agentvault/activity_spool.go index b8bbaf9dc..685b37231 100644 --- a/packages/agentvault/activity_spool.go +++ b/packages/agentvault/activity_spool.go @@ -7,10 +7,7 @@ import ( "github.com/Infisical/infisical-merge/packages/api" ) -// activityRecord is one request that reached forwardHTTP. Metadata only: no headers, because a request -// header carries the injected credential, and no bodies, because LLM traffic is orders of magnitude -// larger than this. The path never carries a query string, which the proxy gets for free by building it -// from r.URL.EscapedPath(). +// Never add headers, bodies or the query string: they can carry the injected credential. type activityRecord struct { Ts string `json:"ts"` Seq uint64 `json:"seq"` @@ -25,15 +22,8 @@ type activityRecord struct { AccessBundle *string `json:"accessBundle"` } -// The first allocation a ring makes. A session that sends a handful of requests a minute never needs more. const activityRingInitialSize = 64 -// activityRing is a bounded FIFO that overwrites its oldest entry when full and counts what it lost, so a -// burst costs the oldest records rather than the newest and the gap is visible in the timeline. -// -// It grows to capacity only as records arrive and lets go of its buffer once drained. Reserving capacity up -// front cost every session about 700 KB from its first request, so a proxy serving a few hundred mostly -// idle sessions held hundreds of megabytes of empty slots. type activityRing struct { buf []activityRecord capacity int @@ -67,7 +57,6 @@ func (r *activityRing) push(rec activityRecord) (evicted bool) { return false } -// grow doubles the buffer, up to capacity, and unwraps it so the oldest record sits at index 0. func (r *activityRing) grow() { size := max(activityRingInitialSize, 2*len(r.buf)) size = min(size, r.capacity) @@ -79,9 +68,6 @@ func (r *activityRing) grow() { r.head = 0 } -// drain removes up to max records, oldest first. The caller seals one chunk per call and loops until the -// ring is empty, which is what keeps a chunk inside the server's recordCount limit even when a slow tick -// let the ring grow past it. func (r *activityRing) drain(max int) []activityRecord { if r.n == 0 || max <= 0 { return nil @@ -96,39 +82,29 @@ func (r *activityRing) drain(max int) []activityRecord { r.head = (r.head + max) % len(r.buf) r.n -= max if r.n == 0 { - // Released between flushes, so a session that went quiet holds nothing until it speaks again. r.buf = nil r.head = 0 } return out } -// takeDropped hands the running drop count to the next chunk and resets it, so each gap is reported once. func (r *activityRing) takeDropped() uint64 { dropped := r.dropped r.dropped = 0 return dropped } -// sealedChunk is ciphertext waiting for its two-step delivery: POST the metadata to Infisical for a -// presigned URL, then PUT the bytes to the customer's bucket. type sealedChunk struct { meta api.CreateAgentVaultActivityChunkRequest ciphertext []byte - // Empty until a POST succeeds, and cleared again on any PUT failure so the next tick re-POSTs the same - // chunk id and the server replays it idempotently. uploadURL string urlExpires time.Time - // Proxy-wide, so the byte cap can find the oldest chunk across every session. sealOrder uint64 - // Set once Infisical has written the row. Never cleared: a re-POST replays the same row. - posted bool + posted bool } -// lostCount is what a chunk that will never be uploaded adds to its session's gap. Once the POST succeeded -// the row exists, and it already reports both halves: the viewer shows its records as a batch it cannot -// read, and its drop count from the row itself. Counting either again would report one loss twice. +// A posted chunk's row already reports its records and drops, so counting them here would double-report. func (c *sealedChunk) lostCount() uint64 { if c.posted { return 0 @@ -136,8 +112,6 @@ func (c *sealedChunk) lostCount() uint64 { return c.meta.DroppedCount + uint64(c.meta.RecordCount) } -// activitySpool is one session's buffer on this proxy. proxyID is constant for the process, so it lives on -// the log rather than here. type activitySpool struct { sessionID string projectID string @@ -168,9 +142,6 @@ type activityGroup struct { plaintext []byte } -// packActivityRecords splits one drained slice into chunks the server takes by size as well as by count. -// Nearly every flush fits whole and is marshalled once. The rest are marshalled per record and packed, which -// yields exactly what marshalling each group as a slice would: '[', the records joined by ',', then ']'. func packActivityRecords(records []activityRecord) ([]activityGroup, error) { whole, err := json.Marshal(records) if err != nil { @@ -188,8 +159,6 @@ func packActivityRecords(records []activityRecord) ([]activityGroup, error) { if err != nil { return nil, err } - // One byte for the separator before it and one for the closing bracket. A record too big to share a - // chunk still gets one of its own rather than being split. if len(buf) > 0 && len(buf)+1+len(part)+1 > activityMaxChunkPlaintext { groups = append(groups, activityGroup{records: records[start:i], plaintext: append(buf, ']')}) buf, start = nil, i @@ -205,7 +174,6 @@ func packActivityRecords(records []activityRecord) ([]activityGroup, error) { return groups, nil } -// sealSlice turns one group of records, already marshalled, into a sealed chunk ready to ship. func (s *activitySpool) sealSlice(proxyID string, records []activityRecord, plaintext []byte, dropped uint64, now time.Time) (*sealedChunk, error) { chunkID := newActivityChunkID(now) aad := buildActivityAAD(s.projectID, s.sessionID, proxyID, chunkID) diff --git a/packages/agentvault/activity_test.go b/packages/agentvault/activity_test.go index d3d6a617e..a1061c4ca 100644 --- a/packages/agentvault/activity_test.go +++ b/packages/agentvault/activity_test.go @@ -14,7 +14,7 @@ import ( ) type shipperCall struct { - kind string // "post" or "put" + kind string sessionID string chunkID string url string @@ -29,8 +29,6 @@ type scriptedResult struct { err error } -// fakeShipper records every call in order and pops scripted outcomes, so a test asserts both what was -// delivered and in which order the two steps ran. type fakeShipper struct { mu sync.Mutex @@ -123,12 +121,6 @@ func testGrant(sessionID string) *activityGrant { return &activityGrant{sessionID: sessionID, projectID: "proj-1", key: make([]byte, 32)} } -// newTestLog fixes the clock so flush eligibility and the pause window are decided, not raced. -// -// tick is what the run loop does once an interval: advance, then flush. A spool with only a few records -// is deliberately not due until an interval has passed since its last flush, so that a size-triggered -// wake-up for one busy session does not drag every quiet session into an early, billable flush. Calling -// flushAll without advancing therefore ships nothing, which is correct rather than a bug to work around. func newTestLog(shipper activityShipper) (log *activityLog, advance func(time.Duration), tick func()) { log = newActivityLog("proxy-1", shipper) now := time.Date(2026, 9, 16, 10, 0, 0, 0, time.UTC) @@ -156,7 +148,7 @@ func aRecord(host string) activityRecord { func TestRecordingIsANoOpWithoutALogOrAGrant(t *testing.T) { var nilLog *activityLog - nilLog.record(testGrant("s1"), aRecord("api.github.com")) // must not panic + nilLog.record(testGrant("s1"), aRecord("api.github.com")) log, _, _ := newTestLog(&fakeShipper{}) log.record(nil, aRecord("api.github.com")) @@ -188,7 +180,6 @@ func TestTheRingAllocatesOnlyWhatItHolds(t *testing.T) { ring := newActivityRing(activitySpoolCapacity) ring.push(activityRecord{Seq: 0}) - // A quiet session is the common case, so it must not pay for a busy one's headroom. if got := cap(ring.buf); got > activityRingInitialSize { t.Fatalf("one record reserved room for %d, expected at most %d", got, activityRingInitialSize) } @@ -209,7 +200,6 @@ func TestTheRingKeepsItsOrderWhileItGrowsPastAWrap(t *testing.T) { } } - // Fill the first allocation, drain part of it so the head moves, then push enough to wrap and grow. push(activityRingInitialSize) ring.drain(10) push(activityRingInitialSize * 3) @@ -234,8 +224,6 @@ func TestTheRingDrainsInSlicesTheServerAccepts(t *testing.T) { ring.push(activityRecord{Seq: uint64(i)}) } - // The ring holds five flushes of headroom but the server refuses a chunk over activityFlushRecords, - // which it answers with a 422 the proxy then treats as poison. Slicing is what prevents that loss. var slices int for ring.len() > 0 { got := ring.drain(activityFlushRecords) @@ -254,7 +242,6 @@ func TestTheDropCountIsReportedOnceAndRidesTheFirstChunk(t *testing.T) { log, _, _ := newTestLog(shipper) grant := testGrant("s1") - // Overfill so the ring evicts, then flush: the gap must be counted on the first chunk only. for i := 0; i < activitySpoolCapacity+50; i++ { log.record(grant, aRecord("api.github.com")) } @@ -277,7 +264,6 @@ func TestASequenceNumberIsConsumedEvenWhenARecordIsDropped(t *testing.T) { log.record(grant, aRecord("api.github.com")) } - // A chunk's firstSeq is what reveals the hole, so the counter must not compact over dropped records. if got := log.spools["s1"].nextSeq; got != uint64(activitySpoolCapacity+10) { t.Fatalf("nextSeq is %d after %d records; drops must still consume a number", got, activitySpoolCapacity+10) } @@ -286,7 +272,6 @@ func TestASequenceNumberIsConsumedEvenWhenARecordIsDropped(t *testing.T) { func TestTheProxyWideFuseDropsTheNewest(t *testing.T) { log, _, _ := newTestLog(&fakeShipper{}) - // Spread across enough spools to pass the total cap without any one ring filling. for i := 0; i < activityTotalCapacity/activitySpoolCapacity+2; i++ { grant := testGrant(fmt.Sprintf("s%d", i)) for j := 0; j < activitySpoolCapacity; j++ { @@ -306,7 +291,6 @@ func TestAChunkIsPostedBeforeItIsUploaded(t *testing.T) { log.record(testGrant("s1"), aRecord("api.github.com")) log.flushAll(context.Background(), true) - // The row lands before the object, so a failed upload is a visible gap rather than a silent one. if got := shipper.kinds(); len(got) != 2 || got[0] != "post" || got[1] != "put" { t.Fatalf("call order was %v, expected post then put", got) } @@ -330,7 +314,6 @@ func TestAFailedUploadRePostsTheSameChunkID(t *testing.T) { if len(posts) != 2 { t.Fatalf("expected the chunk to be re-posted, saw %d posts", len(posts)) } - // The server replays the same id idempotently, which is what makes the retry safe. if posts[0].chunkID != posts[1].chunkID { t.Fatalf("re-post used a different chunk id: %q then %q", posts[0].chunkID, posts[1].chunkID) } @@ -363,7 +346,6 @@ func TestARejectedProxyTokenKeepsEverything(t *testing.T) { log.record(testGrant("s1"), aRecord("api.github.com")) tick() - // The poll loop exits within two heartbeats; until then nothing is thrown away. spool, ok := log.spools["s1"] if !ok { t.Fatal("the spool was dropped on a rejected proxy token") @@ -384,14 +366,12 @@ func TestTheCeilingPausesTheWholeProxyAndLiftsAfterTheBackoff(t *testing.T) { t.Fatalf("expected a ceiling pause, got paused=%v reason=%q", paused, reason) } - // The ceiling is per organization, so another session on this proxy is paused too. log.record(testGrant("s2"), aRecord("api.github.com")) tick() if len(shipper.posts()) != 1 { t.Fatalf("a second session posted while paused; the pause is proxy-wide") } - // Ops may raise the limit within the window, so the sealed chunks are still there when it lifts. advance(activityPauseBackoff + time.Second) log.flushAll(context.Background(), false) if len(shipper.posts()) < 2 { @@ -458,7 +438,6 @@ func TestARefusedChunkIsCountedOnTheNextOne(t *testing.T) { log.record(testGrant("s1"), aRecord("api.github.com")) tick() - // The refused chunk was never written, so the next one is the only place its record can show up. posts := shipper.posts() if len(posts) != 2 || posts[1].dropped != 1 { t.Fatalf("posts were %+v, expected the second to carry one dropped record", posts) @@ -470,7 +449,6 @@ func TestAFlushTooBigForOneChunkIsSplitBySize(t *testing.T) { log, _, tick := newTestLog(shipper) grant := testGrant("s1") - // JSON writes '&' as &, so a full slice of these is about 12 MB: over the server's 8 MiB. for i := 0; i < activityFlushRecords; i++ { rec := aRecord("api.github.com") rec.Path = truncatePath("/" + strings.Repeat("&", maxLoggedPathLen)) @@ -539,7 +517,6 @@ func TestThePendingCapEvictsTheOldestAndCountsIt(t *testing.T) { func TestTheByteCapEvictsTheOldestChunkOnTheProxy(t *testing.T) { log, _, _ := newTestLog(&fakeShipper{}) - // Only the length counts toward the cap, so every chunk can share one buffer. blob := make([]byte, 12<<20) add := func(sessionID string, order uint64, posted bool, carried uint64) *activitySpool { spool, ok := log.spools[sessionID] @@ -557,7 +534,6 @@ func TestTheByteCapEvictsTheOldestChunkOnTheProxy(t *testing.T) { return spool } - // One chunk per session, the shape an outage leaves behind, seven of them for 84 MiB against a 64 MiB cap. log.mu.Lock() add("oldest", 0, false, 7) add("posted", 1, true, 3) @@ -579,11 +555,9 @@ func TestTheByteCapEvictsTheOldestChunkOnTheProxy(t *testing.T) { t.Fatalf("s%d lost its chunk; only the oldest should go", i) } } - // Never posted, so its records and the drops it carried are both unaccounted for anywhere else. if got := log.spools["oldest"].ring.dropped; got != 107 { t.Fatalf("the unposted chunk counted %d dropped, expected 107", got) } - // Posted, so its row already reports its records as unreadable and its drops as not recorded. if got := log.spools["posted"].ring.dropped; got != 0 { t.Fatalf("the posted chunk counted %d dropped, expected 0", got) } @@ -598,15 +572,12 @@ func TestTheTickBreakerStopsHammeringADeadBucket(t *testing.T) { } tick() - // After the first upload fails, the rest of the tick seals but skips both calls, so five spools - // against blocked egress cost one timeout rather than five. if got := len(shipper.puts()); got != 1 { t.Fatalf("%d uploads were attempted in one tick after the first failed", got) } if got := len(shipper.posts()); got != 1 { t.Fatalf("%d rows were written for objects that could not be uploaded", got) } - // Nothing is lost: every spool sealed its records and holds them. for i := 0; i < 5; i++ { if len(log.spools[fmt.Sprintf("s%d", i)].pending) == 0 { t.Fatalf("spool s%d sealed nothing during the outage", i) @@ -622,7 +593,6 @@ func TestReachingTheSliceSizeWakesTheLoopOnce(t *testing.T) { log.record(grant, aRecord("api.github.com")) } - // A buffered channel of one: the loop coalesces a burst into a single wake-up. if len(log.wake) != 1 { t.Fatalf("the wake channel holds %d, expected exactly one pending wake-up", len(log.wake)) } @@ -644,7 +614,6 @@ func TestAnIdleSpoolIsForgotten(t *testing.T) { } func TestIdleCloseOutlastsTheSessionCacheTTL(t *testing.T) { - // A session evicted from the cache stops producing records but must still get its final flush. if activityIdleClose <= sessionInactiveTTL { t.Fatalf("idle close (%s) must outlast the session cache TTL (%s)", activityIdleClose, sessionInactiveTTL) } @@ -676,9 +645,6 @@ func TestABlockedHostIsStillRecorded(t *testing.T) { shipper := &fakeShipper{} log, _, _ := newTestLog(shipper) - // The point of logging every request rather than only brokered ones: under the default any-host - // policy an agent exfiltrating to an unconfigured host is passthrough traffic, and under bundle-hosts - // the refusal is the single most security-relevant line in the timeline. log.record(testGrant("s1"), activityRecord{ Method: "POST", Host: "evil.example", Port: "443", Path: "/collect", Status: 403, Decision: decisionBlocked, }) @@ -733,8 +699,6 @@ func TestSequenceNumbersSurviveASpoolBeingForgotten(t *testing.T) { log.record(grant, aRecord("api.github.com")) tick() - // Idle long enough to be forgotten, then used again. Two records with the same (proxyId, seq) for - // one session would make the log ambiguous for anyone correlating it. advance(activityIdleClose + time.Minute) log.flushAll(context.Background(), false) if _, ok := log.spools["s1"]; ok { @@ -764,7 +728,6 @@ func TestRecordsLostToASealFailureAreStillCounted(t *testing.T) { shipper := &fakeShipper{} log, _, tick := newTestLog(shipper) - // A key the AES constructor rejects, so sealing fails for every slice. grant := &activityGrant{sessionID: "s1", projectID: "proj-1", key: make([]byte, 7)} for i := 0; i < 3; i++ { log.record(grant, aRecord("api.github.com")) @@ -792,7 +755,6 @@ func TestAnUnreachableControlPlaneStopsTheTickAfterOneTimeout(t *testing.T) { } tick() - // Without a breaker this costs five control-plane timeouts in series, every tick. if got := len(shipper.posts()); got != 1 { t.Fatalf("%d chunk POSTs were attempted in one tick after the first timed out", got) } @@ -822,8 +784,6 @@ func TestShutdownDoesNotRaceTheRunLoop(t *testing.T) { <-done close(stop) - // Shutdown flushes from this goroutine while the run loop may still be inside one of its own. The - // race detector is what makes this test worth having. log.close(context.Background()) if len(shipper.puts()) == 0 { diff --git a/packages/agentvault/activity_wiring_test.go b/packages/agentvault/activity_wiring_test.go index 9dc280bb6..93077f403 100644 --- a/packages/agentvault/activity_wiring_test.go +++ b/packages/agentvault/activity_wiring_test.go @@ -11,8 +11,6 @@ import ( "testing" ) -// grantingResolver hands back one service plus an activity grant, so a request driven through the real -// dispatch path produces a real record. type grantingResolver struct { services []*resolvedService } @@ -25,8 +23,6 @@ func (g grantingResolver) resolve(string, *activityGrant) (*resolveResult, error }, nil } -// newRecordingProxy wires a proxy exactly as run.go does, minus the listener, so these tests exercise -// the call site in forwardHTTP rather than activityLog in isolation. func newRecordingProxy(t *testing.T, policy string, services []*resolvedService) (*httptest.Server, *activityLog, *fakeShipper) { t.Helper() @@ -48,7 +44,6 @@ func proxiedGet(t *testing.T, front *httptest.Server, target string) *http.Respo if err != nil { t.Fatal(err) } - // The session token rides in the proxy credentials, which is how an agent presents it. proxyURL.User = url.UserPassword("infisical", "agv_test-token") client := &http.Client{Transport: &http.Transport{Proxy: http.ProxyURL(proxyURL)}} @@ -89,8 +84,6 @@ func TestAProxiedRequestIsRecorded(t *testing.T) { if got.Method != http.MethodGet || got.Path != "/repos/acme/web/issues" || got.Status != http.StatusCreated { t.Fatalf("record is %+v", got) } - // Nothing in the bundle matched, so this is passthrough traffic. Recording it is the whole point: - // under the default any-host policy an agent exfiltrating to an unconfigured host looks like this. if got.Decision != decisionPassthrough { t.Fatalf("decision was %q, expected %q", got.Decision, decisionPassthrough) } @@ -109,8 +102,6 @@ func TestAQueryStringNeverReachesTheRecord(t *testing.T) { defer upstream.Close() front, log, _ := newRecordingProxy(t, TrafficPolicyAnyHost, nil) - // Plenty of APIs put a token in the query string, so the path is built from EscapedPath() and the - // query is never seen. This is true by construction; the test is what keeps it true. proxiedGet(t, front, upstream.URL+"/v1/thing?access_token=super-secret&sid=abc") got := drainOneRecord(t, log) @@ -119,9 +110,6 @@ func TestAQueryStringNeverReachesTheRecord(t *testing.T) { } } -// A path substitution rewrites the request with the real credential before it goes upstream. The record -// has to carry the path the agent sent, placeholder and all, or the activity log would be the one place -// the secret the agent never sees gets written down. func TestASubstitutedPathIsRecordedAsTheAgentSentIt(t *testing.T) { seen := make(chan string, 1) upstream := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { @@ -149,7 +137,6 @@ func TestASubstitutedPathIsRecordedAsTheAgentSentIt(t *testing.T) { } func TestABlockedRequestIsRecordedWithItsRefusal(t *testing.T) { - // bundle-hosts with no service covering the host, and no allow list: the request is refused. front, log, _ := newRecordingProxy(t, TrafficPolicyBundleHosts, nil) res := proxiedGet(t, front, "http://blocked.example/collect") @@ -177,7 +164,6 @@ func TestAnOversizedMethodIsRecordedTruncated(t *testing.T) { proxyURL, _ := url.Parse(front.URL) proxyURL.User = url.UserPassword("infisical", "agv_test-token") client := &http.Client{Transport: &http.Transport{Proxy: http.ProxyURL(proxyURL)}} - // A valid token, so Go sends it. Only the header limit stops an agent sending one far longer. req, err := http.NewRequest(strings.Repeat("A", 5000), upstream.URL+"/v1/thing", nil) if err != nil { t.Fatal(err) @@ -200,7 +186,6 @@ func TestRecordingSurvivesAnUnreachableUpstream(t *testing.T) { proxyURL, _ := url.Parse(front.URL) proxyURL.User = url.UserPassword("infisical", "agv_test-token") client := &http.Client{Transport: &http.Transport{Proxy: http.ProxyURL(proxyURL)}} - // Port 1 refuses immediately. res, err := client.Get("http://127.0.0.1:1/v1/thing") if err != nil { t.Fatalf("the proxy did not answer: %v", err) @@ -224,7 +209,6 @@ func TestAWholeRequestRoundTripsFromProxyToSealedChunk(t *testing.T) { proxiedGet(t, front, fmt.Sprintf("%s/v1/thing/%d", upstream.URL, i)) } - // close is what shutdown calls: it flushes whatever is buffered within the deadline. log.close(context.Background()) posts := shipper.posts() @@ -243,8 +227,6 @@ func TestAWholeRequestRoundTripsFromProxyToSealedChunk(t *testing.T) { t.Fatalf("uploaded %d objects, %d bytes, for a chunk declaring %d", len(puts), puts[0].bytes, posts[0].bytes) } - // What leaves the proxy is ciphertext. If the records ever went out in the clear, the host and the - // path would be readable right here. if bytes.Contains(puts[0].body, []byte("/v1/thing")) || bytes.Contains(puts[0].body, []byte("\"method\"")) { t.Fatal("the uploaded chunk contains readable record fields; it was not sealed") } diff --git a/packages/agentvault/cache.go b/packages/agentvault/cache.go index 2013d0233..ce11199f0 100644 --- a/packages/agentvault/cache.go +++ b/packages/agentvault/cache.go @@ -68,7 +68,6 @@ type sessionEntry struct { sessionID string expiresAt *time.Time services []*resolvedService - // nil when activity logging is off for this session, which is the whole of the disabled path. activity *activityGrant lastSeen time.Time fetchedAt time.Time @@ -81,8 +80,6 @@ func sessionKey(token string) string { } type sessionResolver interface { - // held is the activity grant the caller already has, or nil. Passing it lets the server skip - // re-sending a key that never changes. resolve(sessionToken string, held *activityGrant) (*resolveResult, error) } @@ -156,7 +153,6 @@ func isSessionGone(err error) bool { return errors.Is(err, errSessionGone) } -// get keeps the two CONNECT gates, which care only about validity, free of the activity plumbing. func (c *sessionCache) get(sessionToken string) ([]*resolvedService, error) { services, _, err := c.lookup(sessionToken) return services, err @@ -167,8 +163,6 @@ type cacheLookup struct { activity *activityGrant } -// lookup resolves a session token to what the request path needs: the services to match against, and the -// grant to record under. Both come from one cache entry, so the request handler never resolves. func (c *sessionCache) lookup(sessionToken string) ([]*resolvedService, *activityGrant, error) { key := sessionKey(sessionToken) @@ -202,7 +196,6 @@ func (c *sessionCache) lookup(sessionToken string) ([]*resolvedService, *activit c.mu.Unlock() resolved, err, _ := c.inflight.Do(key, func() (any, error) { - // No cached entry, so no cached key either: ask for one. result, err := c.resolver.resolve(sessionToken, nil) if err != nil { // A rejected proxy token is remembered too: the poll loop exits after two such heartbeats, but @@ -309,7 +302,6 @@ func (c *sessionCache) refresh() { } func (c *sessionCache) refreshOne(key, token string) { - // Tell the server whether we already hold this session's key, so it can skip the unwrap. c.mu.Lock() var held *activityGrant if entry, ok := c.entries[key]; ok { @@ -329,7 +321,6 @@ func (c *sessionCache) refreshOne(key, token string) { entry.sessionID = result.SessionID entry.expiresAt = result.ExpiresAt entry.services = result.Services - // A backend flip lands within one poll, in either direction. entry.activity = result.Activity entry.fetchedAt = time.Now() } diff --git a/packages/agentvault/policy.go b/packages/agentvault/policy.go index 8d6c47800..f5ee7eacd 100644 --- a/packages/agentvault/policy.go +++ b/packages/agentvault/policy.go @@ -17,7 +17,6 @@ var errBodyUnreadable = errors.New("could not read the request body") func checkServicePolicy(svc *resolvedService, req *http.Request) error { if !svc.allowsMethod(req.Method) { - // This lands in the proxy log, so the agent's method is capped here as it is in the record. return fmt.Errorf("service %q does not allow %s: %w", svc.name, truncateLogged(req.Method, maxLoggedMethodLen), errPolicyBlocked) } if len(svc.allowedPathPrefixes) > 0 { diff --git a/packages/agentvault/proxy.go b/packages/agentvault/proxy.go index 999c12946..62791717a 100644 --- a/packages/agentvault/proxy.go +++ b/packages/agentvault/proxy.go @@ -46,19 +46,12 @@ const ( maxConcurrentConns = 512 - maxLoggedPathLen = 2048 - // The agent chooses the method too, and Go bounds it only by the 1 MiB header limit. Uncapped, one - // request could make its own record big enough to stall whoever reads the session. The port needs no - // cap: checkedTarget refuses anything that is not a plain port number. + maxLoggedPathLen = 2048 maxLoggedMethodLen = 32 ) var errHostBlocked = errors.New("host blocked by policy") -// 'http:admin/secrets' parses to an empty path and a non-empty Opaque, which the upstream would receive as -// a request-target with no leading slash, and no path rule could judge it. RFC 9110 makes an http(s) URI -// without an authority invalid. handlePlainForward refuses the shape already; a tunnel reaches forward -// directly, so the refusal lives there. var errOpaqueTarget = errors.New("the request target must be a path; opaque forms are not forwarded") // Wraps a resolve failure so the tunnel can tell it from an upstream failure without reading the text. @@ -426,8 +419,7 @@ func (ps *proxyServer) forwardHTTP(w http.ResponseWriter, r *http.Request, schem } event.Msg("agent-vault: request") - // reqPath is the agent's own path, taken before forward ran, so a substitution that put a real - // credential into the path never reaches the activity log. + // reqPath was taken before forward, so a credential substituted into the path never reaches the record. if outcome.activity != nil { var service, bundle *string if matched != nil { @@ -484,10 +476,7 @@ func (ps *proxyServer) blocksOffBundle(matched *resolvedService, hostname, port type forwardOutcome struct { brokered bool substituted []string - // Set as soon as the session resolves, so every path after that carries it, a refusal included: an - // agent reaching a host or a path it may not is the most security-relevant line the activity log can - // hold. A session that fails to resolve has nothing to attribute a record to, and leaves it nil. - activity *activityGrant + activity *activityGrant } func (ps *proxyServer) forward(req *http.Request, scheme, hostname, port, sessionToken string) (*http.Response, *resolvedService, forwardOutcome, error) { @@ -506,8 +495,9 @@ func (ps *proxyServer) forward(req *http.Request, scheme, hostname, port, sessio return nil, nil, outcome, fmt.Errorf("method %s echoes headers back: %w", method, errPolicyBlocked) } - // After the lookup for the same reason: the only use of this form is probing for a path the policy - // misreads, which is the attempt an admin most wants to see in the session's activity. + // 'http:admin/secrets' parses to an empty path and a non-empty Opaque, which the upstream would receive + // as a request-target with no leading slash. handlePlainForward refuses the shape already; the tunnel + // reaches this handler directly, so the refusal belongs here where both doors meet. if req.URL.Opaque != "" { return nil, nil, outcome, errOpaqueTarget } @@ -658,9 +648,6 @@ var errHostTooLong = errors.New("the target host is longer than a DNS name can b var errBadPort = errors.New("the target port must be a number from 1 to 65535") -// validPort accepts a port only in its plain form. The dialer reads a megabyte of leading zeros, or a '+', -// as an ordinary port, so an agent could otherwise reach 443 under a port string that fills log lines, -// shows in its own record as noise, and misses every service pattern, which compares ports as strings. func validPort(port string) bool { if len(port) == 0 || len(port) > 5 || port[0] == '0' { return false diff --git a/packages/agentvault/proxy_requesttarget_test.go b/packages/agentvault/proxy_requesttarget_test.go index da246a9f9..4fff2ca76 100644 --- a/packages/agentvault/proxy_requesttarget_test.go +++ b/packages/agentvault/proxy_requesttarget_test.go @@ -108,8 +108,8 @@ func TestAPlaceholderInThePathStillSubstitutes(t *testing.T) { } // 'http:admin/secrets' parses to an empty path and a non-empty Opaque. handlePlainForward refuses the shape, -// the tunnel reaches forward directly, and the upstream would have received a target with no leading slash -// and a real credential on it. +// the tunnel reaches forwardHTTP directly, and the upstream would have received a target with no leading +// slash and a real credential on it. func TestAnOpaqueRequestTargetIsRefusedInsideATunnel(t *testing.T) { proxyHost, upstreamHost := newRequestTargetFixture(t, nil) @@ -144,7 +144,6 @@ func TestAnOpaqueRequestTargetIsRecordedAsBlocked(t *testing.T) { t.Fatalf("status = %d, want 400", resp.StatusCode) } - // Refused, and still on the record: nobody sends this form except to probe the path rules. got := drainOneRecord(t, ps.activity) if got.Decision != decisionBlocked || got.Status != http.StatusBadRequest || got.Path != "admin/secrets" { t.Fatalf("record is %+v, expected a blocked 400 for admin/secrets", got) diff --git a/packages/agentvault/resolve.go b/packages/agentvault/resolve.go index 4118a06fa..74d91ade7 100644 --- a/packages/agentvault/resolve.go +++ b/packages/agentvault/resolve.go @@ -22,8 +22,7 @@ type resolveResult struct { SessionID string ExpiresAt *time.Time Services []*resolvedService - // nil when logging is off for this session. - Activity *activityGrant + Activity *activityGrant } const activityKeyBytes = 32 @@ -83,12 +82,6 @@ func (r *infisicalResolver) resolve(sessionToken string, held *activityGrant) (* }, nil } -// toActivityGrant decides what the proxy records under after a poll. -// -// The key is sent exactly once per session. When we told the server we already hold it, the response -// carries no key and the cached one is carried forward; clearing it here instead would silently stop all -// logging after the very first poll. After any cache eviction the grant and the flag are dropped -// together, so the next resolve asks for the key again and this self-heals. func toActivityGrant(sessionID string, wire api.AgentVaultActivityGrant, held *activityGrant) *activityGrant { if !wire.Enabled { return nil diff --git a/packages/agentvault/run.go b/packages/agentvault/run.go index ccbe59887..e3adaeab4 100644 --- a/packages/agentvault/run.go +++ b/packages/agentvault/run.go @@ -265,8 +265,6 @@ func Start(opts Options, enrollmentToken string) error { ctx, cancel := context.WithTimeout(context.Background(), 10*time.Second) defer cancel() _ = front.Shutdown(ctx) - // After the front door is closed, so no record can arrive mid-flush, and before the cache is - // cleared, since the keys the final seal needs live on its entries. ps.activity.close(ctx) ps.cache.close() return nil diff --git a/packages/api/agent_vault.go b/packages/api/agent_vault.go index 348e8f108..1b8797564 100644 --- a/packages/api/agent_vault.go +++ b/packages/api/agent_vault.go @@ -102,9 +102,6 @@ type AgentVaultService struct { Substitutions []AgentVaultSubstitution `json:"substitutions"` } -// AgentVaultActivityGrant is what a session needs in order to have its activity recorded. SessionKey is -// sent exactly once per session: the proxy reports that it already holds one and Infisical skips the -// unwrap, which is a KMS round trip, on every poll after the first. type AgentVaultActivityGrant struct { Enabled bool `json:"enabled"` SessionKey string `json:"sessionKey"` @@ -141,9 +138,6 @@ func CallResolveAgentVaultSession(httpClient *resty.Client, sessionToken string, return res, nil } -// CreateAgentVaultActivityChunkRequest is the metadata for one sealed chunk. The ciphertext itself never -// passes through Infisical: the response carries a presigned URL to PUT it straight to the customer's -// bucket. Re-sending the same ChunkID is idempotent, which is what makes a failed upload safe to retry. type CreateAgentVaultActivityChunkRequest struct { ChunkID string `json:"chunkId"` StartedAt string `json:"startedAt"` From 93db017ae68102247ac4fbf509381c068c9e278a Mon Sep 17 00:00:00 2001 From: saif <11242541+saifsmailbox98@users.noreply.github.com> Date: Thu, 24 Sep 2026 13:17:38 +0530 Subject: [PATCH 13/43] chore(agent-vault): drop an unused chunk ID parser --- packages/agentvault/activity_crypto.go | 8 -------- packages/agentvault/activity_crypto_test.go | 4 +++- 2 files changed, 3 insertions(+), 9 deletions(-) diff --git a/packages/agentvault/activity_crypto.go b/packages/agentvault/activity_crypto.go index 7d10e0a22..bb43ada44 100644 --- a/packages/agentvault/activity_crypto.go +++ b/packages/agentvault/activity_crypto.go @@ -54,11 +54,3 @@ func newActivityChunkID(now time.Time) string { } func newULIDEntropy() io.Reader { return rand.Reader } - -func parseActivityChunkID(id string) (time.Time, error) { - parsed, err := ulid.Parse(id) - if err != nil { - return time.Time{}, err - } - return ulid.Time(parsed.Time()), nil -} diff --git a/packages/agentvault/activity_crypto_test.go b/packages/agentvault/activity_crypto_test.go index 6ff41b448..77bf019bb 100644 --- a/packages/agentvault/activity_crypto_test.go +++ b/packages/agentvault/activity_crypto_test.go @@ -9,6 +9,8 @@ import ( "encoding/json" "testing" "time" + + "github.com/oklog/ulid" ) const ( @@ -159,7 +161,7 @@ func TestChunkIDsAreULIDsThatSortByTime(t *testing.T) { if !(earlier < later) { t.Fatalf("%q did not sort before %q", earlier, later) } - if _, err := parseActivityChunkID(earlier); err != nil { + if _, err := ulid.Parse(earlier); err != nil { t.Fatalf("a minted chunk id did not parse: %v", err) } } From 8bf2cf51851d8071556992beda32ebd6e7b69d20 Mon Sep 17 00:00:00 2001 From: saif <11242541+saifsmailbox98@users.noreply.github.com> Date: Thu, 24 Sep 2026 13:17:51 +0530 Subject: [PATCH 14/43] fix(agent-vault): warn with a count when a session's held activity is dropped --- packages/agentvault/activity.go | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/packages/agentvault/activity.go b/packages/agentvault/activity.go index 18c345733..f914b0cc4 100644 --- a/packages/agentvault/activity.go +++ b/packages/agentvault/activity.go @@ -389,14 +389,17 @@ func (a *activityLog) handleCreateFailure(spool *activitySpool, chunk *sealedChu case isSessionGone(err): a.mu.Lock() + lost := spool.ring.len() a.total -= spool.ring.len() for _, held := range spool.pending { + lost += held.meta.RecordCount a.sealedBytes -= len(held.ciphertext) } a.forgetSpoolLocked(spool.sessionID, spool) delete(a.seqBySession, spool.sessionID) a.mu.Unlock() - log.Debug().Str("sessionId", spool.sessionID).Msg("agent-vault: session gone, dropping its activity") + log.Warn().Err(err).Str("sessionId", spool.sessionID).Int("records", lost). + Msg("agent-vault: Infisical no longer accepts activity for this session, dropping what was held") return false case isActivityErrorNamed(err, activityCeilingReachedName): From 21302ec4b76e8b03cc66715f87bcd6a1435af872 Mon Sep 17 00:00:00 2001 From: saif <11242541+saifsmailbox98@users.noreply.github.com> Date: Thu, 24 Sep 2026 13:18:30 +0530 Subject: [PATCH 15/43] fix(agent-vault): say plainly when activity is refused for a wrong clock --- packages/agentvault/activity.go | 35 ++++++++++++++++++++++------ packages/agentvault/activity_test.go | 28 ++++++++++++++++++++++ 2 files changed, 56 insertions(+), 7 deletions(-) diff --git a/packages/agentvault/activity.go b/packages/agentvault/activity.go index f914b0cc4..afb45724d 100644 --- a/packages/agentvault/activity.go +++ b/packages/agentvault/activity.go @@ -28,6 +28,7 @@ const ( activityCeilingReachedName = "AgentVaultActivityCeilingReached" activityDisabledName = "AgentVaultActivityDisabled" + activityClockSkewName = "AgentVaultActivityClockSkew" ) type activityGrant struct { @@ -64,6 +65,8 @@ type activityLog struct { s3Down bool infisicalDown bool + clockSkewReported bool + closed bool wake chan struct{} } @@ -356,6 +359,9 @@ func (a *activityLog) shipChunk(ctx context.Context, spool *activitySpool, chunk if err != nil { return a.handleCreateFailure(spool, chunk, err) } + a.mu.Lock() + a.clockSkewReported = false + a.mu.Unlock() chunk.posted = true chunk.uploadURL = res.UploadURL chunk.urlExpires = a.now().Add(time.Duration(res.ExpiresInSeconds) * time.Second) @@ -412,15 +418,19 @@ func (a *activityLog) handleCreateFailure(spool *activitySpool, chunk *sealedChu log.Warn().Msg("agent-vault: activity logging is switched off for this project, pausing for 15m") return false - case isPoisonChunk(err): + case isActivityErrorNamed(err, activityClockSkewName): + a.dropRefused(spool, chunk) a.mu.Lock() - if len(spool.pending) > 0 && spool.pending[0] == chunk { - spool.pending = spool.pending[1:] - a.sealedBytes -= len(chunk.ciphertext) - // Counted as dropped, or an agent that gets its own chunk refused could erase what it did. - spool.ring.dropped += chunk.lostCount() - } + reported := a.clockSkewReported + a.clockSkewReported = true a.mu.Unlock() + if !reported { + log.Error().Err(err).Msg("agent-vault: activity is being refused because this machine's clock is wrong; fix the clock to resume recording") + } + return false + + case isPoisonChunk(err): + a.dropRefused(spool, chunk) log.Error().Err(err).Str("chunkId", chunk.meta.ChunkID).Int("records", chunk.meta.RecordCount). Msg("agent-vault: Infisical rejected an activity chunk as malformed, dropping it") return false @@ -434,3 +444,14 @@ func (a *activityLog) handleCreateFailure(spool *activitySpool, chunk *sealedChu return false } } + +func (a *activityLog) dropRefused(spool *activitySpool, chunk *sealedChunk) { + a.mu.Lock() + defer a.mu.Unlock() + if len(spool.pending) > 0 && spool.pending[0] == chunk { + spool.pending = spool.pending[1:] + a.sealedBytes -= len(chunk.ciphertext) + // Counted as dropped, or an agent that gets its own chunk refused could erase what it did. + spool.ring.dropped += chunk.lostCount() + } +} diff --git a/packages/agentvault/activity_test.go b/packages/agentvault/activity_test.go index a1061c4ca..ab98a4685 100644 --- a/packages/agentvault/activity_test.go +++ b/packages/agentvault/activity_test.go @@ -444,6 +444,34 @@ func TestARefusedChunkIsCountedOnTheNextOne(t *testing.T) { } } +func TestAClockSkewRefusalIsDroppedCountedAndLoggedOnce(t *testing.T) { + skew := scriptedResult{err: apiErr(http.StatusBadRequest, activityClockSkewName)} + shipper := &fakeShipper{postResults: []scriptedResult{skew, skew}} + log, _, tick := newTestLog(shipper) + + log.record(testGrant("s1"), aRecord("api.github.com")) + tick() + if len(log.spools["s1"].pending) != 0 { + t.Fatal("a chunk refused for clock skew was kept") + } + if !log.clockSkewReported { + t.Fatal("the first clock skew refusal was not reported") + } + + log.record(testGrant("s1"), aRecord("api.github.com")) + tick() + log.record(testGrant("s1"), aRecord("api.github.com")) + tick() + + posts := shipper.posts() + if len(posts) != 3 || posts[2].dropped != 2 { + t.Fatalf("posts were %+v, expected the third to carry both refused records", posts) + } + if log.clockSkewReported { + t.Fatal("an accepted chunk did not end the clock skew episode") + } +} + func TestAFlushTooBigForOneChunkIsSplitBySize(t *testing.T) { shipper := &fakeShipper{postDefault: scriptedResult{err: errors.New("infisical unreachable")}} log, _, tick := newTestLog(shipper) From 189e75caded0723c7257116be94c161fea26f7d1 Mon Sep 17 00:00:00 2001 From: saif <11242541+saifsmailbox98@users.noreply.github.com> Date: Thu, 24 Sep 2026 13:20:28 +0530 Subject: [PATCH 16/43] fix(agent-vault): resume recording as soon as logging is back on --- packages/agentvault/activity.go | 117 ++++++++++++++++++++------- packages/agentvault/activity_test.go | 91 ++++++++++++++++++--- packages/agentvault/resolve.go | 4 +- 3 files changed, 170 insertions(+), 42 deletions(-) diff --git a/packages/agentvault/activity.go b/packages/agentvault/activity.go index afb45724d..ecc5e274c 100644 --- a/packages/agentvault/activity.go +++ b/packages/agentvault/activity.go @@ -3,6 +3,7 @@ package agentvault import ( "context" "sync" + "sync/atomic" "time" "github.com/Infisical/infisical-merge/packages/api" @@ -35,6 +36,20 @@ type activityGrant struct { sessionID string projectID string key []byte + + issued uint64 +} + +// Counts every grant the proxy is handed, so a refusal can be told apart from a key issued after it. +var activityGrantsIssued atomic.Uint64 + +func newActivityGrant(sessionID, projectID string, key []byte) *activityGrant { + return &activityGrant{sessionID: sessionID, projectID: projectID, key: key, issued: activityGrantsIssued.Add(1)} +} + +type forgottenSpool struct { + nextSeq uint64 + dropped uint64 } type activityShipper interface { @@ -53,14 +68,16 @@ type activityLog struct { mu sync.Mutex spools map[string]*activitySpool - seqBySession map[string]uint64 + forgotten map[string]forgottenSpool total int sealedBytes int nextSealOrder uint64 - pauseUntil time.Time - pauseReason string + pauseUntil time.Time + + switchedOff bool + switchedOffThrough uint64 s3Down bool infisicalDown bool @@ -73,12 +90,12 @@ type activityLog struct { func newActivityLog(proxyID string, shipper activityShipper) *activityLog { return &activityLog{ - proxyID: proxyID, - shipper: shipper, - now: time.Now, - spools: make(map[string]*activitySpool), - seqBySession: make(map[string]uint64), - wake: make(chan struct{}, 1), + proxyID: proxyID, + shipper: shipper, + now: time.Now, + spools: make(map[string]*activitySpool), + forgotten: make(map[string]forgottenSpool), + wake: make(chan struct{}, 1), } } @@ -96,7 +113,10 @@ func (a *activityLog) record(g *activityGrant, rec activityRecord) { spool, ok := a.spools[g.sessionID] if !ok { spool = newActivitySpool(g, a.now()) - spool.nextSeq = a.seqBySession[g.sessionID] + if prior, ok := a.forgotten[g.sessionID]; ok { + spool.nextSeq, spool.ring.dropped = prior.nextSeq, prior.dropped + delete(a.forgotten, g.sessionID) + } a.spools[g.sessionID] = spool } @@ -106,6 +126,15 @@ func (a *activityLog) record(g *activityGrant, rec activityRecord) { rec.Ts = a.now().UTC().Format(time.RFC3339Nano) spool.lastRecordAt = a.now() + if a.switchedOff { + if g.issued <= a.switchedOffThrough { + spool.ring.dropped++ + return + } + a.switchedOff = false + log.Info().Msg("agent-vault: activity logging is back on, recording again") + } + if !a.pauseUntil.IsZero() && a.now().Before(a.pauseUntil) { spool.ring.dropped++ return @@ -192,28 +221,57 @@ func (a *activityLog) dueSpools(final bool) []*activitySpool { return due } +// Keeps the drop count too, so a spool forgotten while logging was off still reports what it lost. func (a *activityLog) forgetSpoolLocked(sessionID string, spool *activitySpool) { - if len(a.seqBySession) >= maxSessionCacheEntries { - a.seqBySession = make(map[string]uint64) + if len(a.forgotten) >= maxSessionCacheEntries { + a.forgotten = make(map[string]forgottenSpool) } - a.seqBySession[sessionID] = spool.nextSeq + a.forgotten[sessionID] = forgottenSpool{nextSeq: spool.nextSeq, dropped: spool.ring.dropped} delete(a.spools, sessionID) } -func (a *activityLog) paused() (bool, string) { +func (a *activityLog) holding() bool { a.mu.Lock() defer a.mu.Unlock() - if a.pauseUntil.IsZero() || !a.now().Before(a.pauseUntil) { - return false, "" - } - return true, a.pauseReason + return a.switchedOff || (!a.pauseUntil.IsZero() && a.now().Before(a.pauseUntil)) } -func (a *activityLog) pause(reason string) { +func (a *activityLog) pause() { a.mu.Lock() defer a.mu.Unlock() a.pauseUntil = a.now().Add(activityPauseBackoff) - a.pauseReason = reason +} + +// Drops everything held and counts it. A key issued after the refused request went out means logging may already +// be back on, so then the refusal is stale and nothing is dropped. +func (a *activityLog) switchOff(grantsIssuedAtSend uint64) { + a.mu.Lock() + defer a.mu.Unlock() + if activityGrantsIssued.Load() > grantsIssuedAtSend { + return + } + + var lost int + for _, spool := range a.spools { + held := spool.ring.len() + spool.ring.drain(held) + spool.ring.dropped += uint64(held) + a.total -= held + lost += held + for _, chunk := range spool.pending { + spool.ring.dropped += chunk.lostCount() + a.sealedBytes -= len(chunk.ciphertext) + lost += chunk.meta.RecordCount + } + spool.pending = nil + } + + if !a.switchedOff { + log.Warn().Int("records", lost). + Msg("agent-vault: activity logging is switched off for this project, dropping what was held until it is back on") + } + a.switchedOff = true + a.switchedOffThrough = grantsIssuedAtSend } func (a *activityLog) flushAll(ctx context.Context, final bool) { @@ -323,14 +381,11 @@ func (a *activityLog) evictOldestLocked(spool *activitySpool) { } func (a *activityLog) flushSpool(ctx context.Context, spool *activitySpool, final bool) { - if paused, reason := a.paused(); paused { - a.sealRing(spool) - _ = reason + a.sealRing(spool) + if a.holding() { return } - a.sealRing(spool) - for { a.mu.Lock() if len(spool.pending) == 0 || a.s3Down || a.infisicalDown { @@ -355,9 +410,10 @@ func (a *activityLog) flushSpool(ctx context.Context, spool *activitySpool, fina func (a *activityLog) shipChunk(ctx context.Context, spool *activitySpool, chunk *sealedChunk, final bool) bool { if chunk.uploadURL == "" || a.now().Add(10*time.Second).After(chunk.urlExpires) { + grantsIssuedAtSend := activityGrantsIssued.Load() res, err := a.shipper.createChunk(final, spool.sessionID, chunk.meta) if err != nil { - return a.handleCreateFailure(spool, chunk, err) + return a.handleCreateFailure(spool, chunk, err, grantsIssuedAtSend) } a.mu.Lock() a.clockSkewReported = false @@ -387,7 +443,7 @@ func (a *activityLog) shipChunk(ctx context.Context, spool *activitySpool, chunk return true } -func (a *activityLog) handleCreateFailure(spool *activitySpool, chunk *sealedChunk, err error) bool { +func (a *activityLog) handleCreateFailure(spool *activitySpool, chunk *sealedChunk, err error, grantsIssuedAtSend uint64) bool { switch { case isProxyTokenRejected(err): log.Warn().Err(err).Msg("agent-vault: Infisical rejected this proxy's token, holding activity") @@ -402,20 +458,19 @@ func (a *activityLog) handleCreateFailure(spool *activitySpool, chunk *sealedChu a.sealedBytes -= len(held.ciphertext) } a.forgetSpoolLocked(spool.sessionID, spool) - delete(a.seqBySession, spool.sessionID) + delete(a.forgotten, spool.sessionID) a.mu.Unlock() log.Warn().Err(err).Str("sessionId", spool.sessionID).Int("records", lost). Msg("agent-vault: Infisical no longer accepts activity for this session, dropping what was held") return false case isActivityErrorNamed(err, activityCeilingReachedName): - a.pause(activityCeilingReachedName) + a.pause() log.Error().Err(err).Msg("agent-vault: activity logging has reached its limit for this organization, retrying in 15m") return false case isActivityErrorNamed(err, activityDisabledName): - a.pause(activityDisabledName) - log.Warn().Msg("agent-vault: activity logging is switched off for this project, pausing for 15m") + a.switchOff(grantsIssuedAtSend) return false case isActivityErrorNamed(err, activityClockSkewName): diff --git a/packages/agentvault/activity_test.go b/packages/agentvault/activity_test.go index ab98a4685..f284149a5 100644 --- a/packages/agentvault/activity_test.go +++ b/packages/agentvault/activity_test.go @@ -118,7 +118,7 @@ func apiErr(status int, name string) error { } func testGrant(sessionID string) *activityGrant { - return &activityGrant{sessionID: sessionID, projectID: "proj-1", key: make([]byte, 32)} + return newActivityGrant(sessionID, "proj-1", make([]byte, 32)) } func newTestLog(shipper activityShipper) (log *activityLog, advance func(time.Duration), tick func()) { @@ -362,8 +362,8 @@ func TestTheCeilingPausesTheWholeProxyAndLiftsAfterTheBackoff(t *testing.T) { log.record(testGrant("s1"), aRecord("api.github.com")) tick() - if paused, reason := log.paused(); !paused || reason != activityCeilingReachedName { - t.Fatalf("expected a ceiling pause, got paused=%v reason=%q", paused, reason) + if !log.holding() || log.switchedOff { + t.Fatal("expected a ceiling pause") } log.record(testGrant("s2"), aRecord("api.github.com")) @@ -379,18 +379,91 @@ func TestTheCeilingPausesTheWholeProxyAndLiftsAfterTheBackoff(t *testing.T) { } } -func TestBeingSwitchedOffPausesRatherThanDiscards(t *testing.T) { +func TestBeingSwitchedOffDropsWhatWasHeldAndCountsIt(t *testing.T) { + shipper := &fakeShipper{postResults: []scriptedResult{{err: apiErr(400, activityDisabledName)}}} + log, _, tick := newTestLog(shipper) + grant := testGrant("s1") + + log.record(grant, aRecord("api.github.com")) + log.record(grant, aRecord("api.github.com")) + tick() + + spool := log.spools["s1"] + if !log.switchedOff || len(spool.pending) != 0 || spool.ring.len() != 0 { + t.Fatalf("switched off=%v, pending=%d, ring=%d; expected everything held to be dropped", + log.switchedOff, len(spool.pending), spool.ring.len()) + } + if spool.ring.dropped != 2 { + t.Fatalf("%d records were counted as dropped, expected 2", spool.ring.dropped) + } + if log.total != 0 || log.sealedBytes != 0 { + t.Fatalf("totals not restored: records=%d sealed bytes=%d", log.total, log.sealedBytes) + } + + log.record(grant, aRecord("api.github.com")) + tick() + if len(shipper.posts()) != 1 { + t.Fatal("the proxy kept sending while logging was switched off") + } + if spool.ring.dropped != 3 { + t.Fatalf("a record made with the old key was not counted as dropped, got %d", spool.ring.dropped) + } +} + +func TestAKeyIssuedAfterTheSwitchOffResumesRecordingAtOnce(t *testing.T) { shipper := &fakeShipper{postResults: []scriptedResult{{err: apiErr(400, activityDisabledName)}}} log, _, tick := newTestLog(shipper) log.record(testGrant("s1"), aRecord("api.github.com")) tick() - if paused, reason := log.paused(); !paused || reason != activityDisabledName { - t.Fatalf("expected a disabled pause, got paused=%v reason=%q", paused, reason) + log.record(testGrant("s1"), aRecord("api.github.com")) + if log.switchedOff { + t.Fatal("a key issued after the refusal did not end the switch-off") } - if len(log.spools["s1"].pending) != 1 { - t.Fatal("the sealed chunk was discarded when logging was switched off") + tick() + + posts := shipper.posts() + if len(posts) != 2 || len(shipper.puts()) != 1 { + t.Fatalf("posts=%d uploads=%d, expected the new record to ship", len(posts), len(shipper.puts())) + } + if posts[1].dropped != 1 { + t.Fatalf("the chunk after the switch-off carried %d dropped, expected 1", posts[1].dropped) + } +} + +func TestARefusalRacingANewKeyDropsNothing(t *testing.T) { + shipper := &fakeShipper{postResults: []scriptedResult{{err: apiErr(400, activityDisabledName)}}} + log, _, _ := newTestLog(shipper) + grant := testGrant("s1") + + log.record(grant, aRecord("api.github.com")) + log.switchOff(activityGrantsIssued.Load() - 1) + + if log.switchedOff || log.spools["s1"].ring.len() != 1 { + t.Fatal("a refusal sent before a newer key was issued still dropped what was held") + } +} + +func TestDropsAreStillReportedAfterAnIdleSpoolIsForgotten(t *testing.T) { + shipper := &fakeShipper{postResults: []scriptedResult{{err: apiErr(400, activityDisabledName)}}} + log, advance, tick := newTestLog(shipper) + + log.record(testGrant("s1"), aRecord("api.github.com")) + tick() + + advance(activityIdleClose + time.Minute) + log.flushAll(context.Background(), false) + if _, ok := log.spools["s1"]; ok { + t.Fatal("the idle spool was not forgotten, so this test proves nothing") + } + + log.record(testGrant("s1"), aRecord("api.github.com")) + tick() + + posts := shipper.posts() + if len(posts) != 2 || posts[1].dropped != 1 { + t.Fatalf("posts were %+v, expected the drop to survive the spool being forgotten", posts) } } @@ -747,7 +820,7 @@ func TestASessionThatIsGoneDoesNotReserveItsSequenceNumbers(t *testing.T) { log.record(testGrant("s1"), aRecord("api.github.com")) tick() - if _, ok := log.seqBySession["s1"]; ok { + if _, ok := log.forgotten["s1"]; ok { t.Fatal("a session the server has forgotten is still holding a sequence number") } } diff --git a/packages/agentvault/resolve.go b/packages/agentvault/resolve.go index 74d91ade7..7d29559cb 100644 --- a/packages/agentvault/resolve.go +++ b/packages/agentvault/resolve.go @@ -93,7 +93,7 @@ func toActivityGrant(sessionID string, wire api.AgentVaultActivityGrant, held *a if wire.SessionKey == "" { if held != nil { - return &activityGrant{sessionID: sessionID, projectID: wire.ProjectID, key: held.key} + return newActivityGrant(sessionID, wire.ProjectID, held.key) } log.Warn().Str("sessionId", sessionID).Msg("agent-vault: activity is enabled but no key was sent, not recording") return nil @@ -105,7 +105,7 @@ func toActivityGrant(sessionID string, wire api.AgentVaultActivityGrant, held *a return nil } - return &activityGrant{sessionID: sessionID, projectID: wire.ProjectID, key: key} + return newActivityGrant(sessionID, wire.ProjectID, key) } func toCredential(wire api.AgentVaultCredential) credential { From fd8b61bcaad65e1d2b3cb3fceee2309d3ce296bc Mon Sep 17 00:00:00 2001 From: saif <11242541+saifsmailbox98@users.noreply.github.com> Date: Thu, 24 Sep 2026 13:20:42 +0530 Subject: [PATCH 17/43] fix(agent-vault): give the final activity flush its own shutdown budget --- packages/agentvault/activity.go | 1 + packages/agentvault/run.go | 15 ++++++++++----- 2 files changed, 11 insertions(+), 5 deletions(-) diff --git a/packages/agentvault/activity.go b/packages/agentvault/activity.go index ecc5e274c..c7317aa52 100644 --- a/packages/agentvault/activity.go +++ b/packages/agentvault/activity.go @@ -26,6 +26,7 @@ const ( activityPauseBackoff = 15 * time.Minute activityPutTimeout = 10 * time.Second activityFinalTimeout = 3 * time.Second + activityCloseTimeout = 5 * time.Second activityCeilingReachedName = "AgentVaultActivityCeilingReached" activityDisabledName = "AgentVaultActivityDisabled" diff --git a/packages/agentvault/run.go b/packages/agentvault/run.go index e3adaeab4..37d5b6fbd 100644 --- a/packages/agentvault/run.go +++ b/packages/agentvault/run.go @@ -248,12 +248,17 @@ func Start(opts Options, enrollmentToken string) error { signals := make(chan os.Signal, 1) signal.Notify(signals, os.Interrupt, syscall.SIGTERM) + // Its own budget: in-flight requests can spend all of the shutdown's, and this is the last minute of activity. + closeActivity := func() { + ctx, cancel := context.WithTimeout(context.Background(), activityCloseTimeout) + defer cancel() + ps.activity.close(ctx) + } + select { case err := <-serveErr: close(stop) - ctx, cancel := context.WithTimeout(context.Background(), 10*time.Second) - defer cancel() - ps.activity.close(ctx) + closeActivity() ps.cache.close() if errors.Is(err, http.ErrServerClosed) { return nil @@ -265,7 +270,7 @@ func Start(opts Options, enrollmentToken string) error { ctx, cancel := context.WithTimeout(context.Background(), 10*time.Second) defer cancel() _ = front.Shutdown(ctx) - ps.activity.close(ctx) + closeActivity() ps.cache.close() return nil case err := <-fatal: @@ -273,7 +278,7 @@ func Start(opts Options, enrollmentToken string) error { ctx, cancel := context.WithTimeout(context.Background(), 10*time.Second) defer cancel() _ = front.Shutdown(ctx) - ps.activity.close(ctx) + closeActivity() ps.cache.close() return err } From f86f38b9e9e0cf734a1467e20951cd7187fd09da Mon Sep 17 00:00:00 2001 From: saif <11242541+saifsmailbox98@users.noreply.github.com> Date: Thu, 24 Sep 2026 13:36:10 +0530 Subject: [PATCH 18/43] feat(agent-vault): report activity uploads on the heartbeat --- packages/agentvault/activity.go | 24 +++++++++ packages/agentvault/run.go | 9 +++- packages/agentvault/run_pollloop_test.go | 65 ++++++++++++++++++++++++ packages/api/agent_vault.go | 7 ++- 4 files changed, 102 insertions(+), 3 deletions(-) diff --git a/packages/agentvault/activity.go b/packages/agentvault/activity.go index c7317aa52..dce65111d 100644 --- a/packages/agentvault/activity.go +++ b/packages/agentvault/activity.go @@ -85,6 +85,9 @@ type activityLog struct { clockSkewReported bool + uploads uint64 + uploadsReported uint64 + closed bool wake chan struct{} } @@ -405,10 +408,31 @@ func (a *activityLog) flushSpool(ctx context.Context, spool *activitySpool, fina spool.pending = spool.pending[1:] a.sealedBytes -= len(chunk.ciphertext) } + a.uploads++ a.mu.Unlock() } } +// What the next heartbeat reports. It is acknowledged only once Infisical has answered, so a heartbeat that +// fails reports the same uploads again. +func (a *activityLog) uploadReport() (snapshot uint64, uploaded bool) { + if a == nil { + return 0, false + } + a.mu.Lock() + defer a.mu.Unlock() + return a.uploads, a.uploads > a.uploadsReported +} + +func (a *activityLog) ackUploadReport(snapshot uint64) { + if a == nil { + return + } + a.mu.Lock() + defer a.mu.Unlock() + a.uploadsReported = max(a.uploadsReported, snapshot) +} + func (a *activityLog) shipChunk(ctx context.Context, spool *activitySpool, chunk *sealedChunk, final bool) bool { if chunk.uploadURL == "" || a.now().Add(10*time.Second).After(chunk.urlExpires) { grantsIssuedAtSend := activityGrantsIssued.Load() diff --git a/packages/agentvault/run.go b/packages/agentvault/run.go index 37d5b6fbd..d7cfa4fab 100644 --- a/packages/agentvault/run.go +++ b/packages/agentvault/run.go @@ -334,7 +334,8 @@ func (ps *proxyServer) tick(st *store) (tokenRejected bool) { httpClient, err := util.GetRestyClientWithCustomHeaders() if err == nil { httpClient.SetAuthToken(ps.opts.ProxyToken()).SetTimeout(controlPlaneTimeout) - res, hbErr := api.CallAgentVaultHeartbeat(httpClient) + uploads, uploaded := ps.activity.uploadReport() + res, hbErr := api.CallAgentVaultHeartbeat(httpClient, api.AgentVaultHeartbeatRequest{ActivityUploaded: uploaded}) if hbErr != nil { tokenRejected = isTokenRejected(hbErr) log.Warn().Err(hbErr).Msg("agent-vault: heartbeat failed") @@ -344,7 +345,11 @@ func (ps *proxyServer) tick(st *store) (tokenRejected bool) { AllowedHosts: res.Config.AllowedHosts, PollInterval: res.Config.PollInterval, } - if !usableProxyConfig(next) { + usable := usableProxyConfig(next) + if usable { + ps.activity.ackUploadReport(uploads) + } + if !usable { log.Warn(). Str("trafficPolicy", next.TrafficPolicy). Int("pollInterval", next.PollInterval). diff --git a/packages/agentvault/run_pollloop_test.go b/packages/agentvault/run_pollloop_test.go index b593c19c1..b301a9c63 100644 --- a/packages/agentvault/run_pollloop_test.go +++ b/packages/agentvault/run_pollloop_test.go @@ -6,6 +6,7 @@ import ( "net/http/httptest" "os" "path/filepath" + "sync" "testing" "time" @@ -161,3 +162,67 @@ func TestTickKeepsTheCurrentSettingsWhenTheHeartbeatCarriesNone(t *testing.T) { t.Fatalf("a heartbeat with no settings was persisted: %+v", back.Config) } } + +func TestHeartbeatsReportUploadsUntilInfisicalAnswers(t *testing.T) { + answered, err := json.Marshal(api.AgentVaultHeartbeatResponse{ + Config: api.AgentVaultProxyConfig{TrafficPolicy: TrafficPolicyAnyHost, PollInterval: 60}, + }) + if err != nil { + t.Fatal(err) + } + var mu sync.Mutex + var reported bool + reply := []byte(`{"message":"no"}`) + status := http.StatusBadRequest + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + var req api.AgentVaultHeartbeatRequest + _ = json.NewDecoder(r.Body).Decode(&req) + mu.Lock() + reported = req.ActivityUploaded + code, body := status, reply + mu.Unlock() + w.Header().Set("Content-Type", "application/json") + w.WriteHeader(code) + _, _ = w.Write(body) + })) + t.Cleanup(srv.Close) + prev := config.INFISICAL_URL + config.INFISICAL_URL = srv.URL + t.Cleanup(func() { config.INFISICAL_URL = prev }) + + ps := &proxyServer{ + opts: Options{ProxyToken: func() string { return "tok" }}, + config: ProxyConfig{TrafficPolicy: TrafficPolicyAnyHost, PollInterval: 60}, + activity: newActivityLog("p1", &fakeShipper{}), + } + resolver, err := newInfisicalResolver(ps.opts.ProxyToken) + if err != nil { + t.Fatal(err) + } + ps.cache = newSessionCache(resolver, ps.pollInterval) + st := newStore(t.TempDir()) + ps.activity.uploads = 1 + + tickReports := func(code int, body []byte) bool { + mu.Lock() + status, reply = code, body + mu.Unlock() + ps.tick(st) + mu.Lock() + defer mu.Unlock() + return reported + } + + if !tickReports(http.StatusBadRequest, []byte(`{"message":"no"}`)) { + t.Fatal("an upload was not reported") + } + if !tickReports(http.StatusOK, []byte(`{"ok":true}`)) { + t.Fatal("a failed heartbeat acknowledged the upload") + } + if !tickReports(http.StatusOK, answered) { + t.Fatal("a reply that was not Infisical's acknowledged the upload") + } + if tickReports(http.StatusOK, answered) { + t.Fatal("an upload Infisical already acknowledged was reported again") + } +} diff --git a/packages/api/agent_vault.go b/packages/api/agent_vault.go index 1b8797564..93556aa34 100644 --- a/packages/api/agent_vault.go +++ b/packages/api/agent_vault.go @@ -47,16 +47,21 @@ func CallLoginAgentVaultProxy(httpClient *resty.Client, request LoginAgentVaultP return res, nil } +type AgentVaultHeartbeatRequest struct { + ActivityUploaded bool `json:"activityUploaded"` +} + type AgentVaultHeartbeatResponse struct { Config AgentVaultProxyConfig `json:"config"` } -func CallAgentVaultHeartbeat(httpClient *resty.Client) (AgentVaultHeartbeatResponse, error) { +func CallAgentVaultHeartbeat(httpClient *resty.Client, request AgentVaultHeartbeatRequest) (AgentVaultHeartbeatResponse, error) { var res AgentVaultHeartbeatResponse response, err := httpClient. R(). SetResult(&res). SetHeader("User-Agent", USER_AGENT). + SetBody(request). Post(fmt.Sprintf("%v/v1/agent-vault/proxy/heartbeat", config.INFISICAL_URL)) if err != nil { From a08badf87877fbfb3519bbb455958b46e96f7775 Mon Sep 17 00:00:00 2001 From: saif <11242541+saifsmailbox98@users.noreply.github.com> Date: Thu, 24 Sep 2026 14:24:11 +0530 Subject: [PATCH 19/43] fix(agent-vault): keep the final activity flush inside its shutdown budget --- packages/agentvault/activity.go | 8 ++++++-- packages/agentvault/activity_ship.go | 4 ++-- packages/agentvault/activity_ship_test.go | 2 +- packages/agentvault/activity_test.go | 16 +++++++++++++++- packages/api/agent_vault.go | 4 +++- 5 files changed, 27 insertions(+), 7 deletions(-) diff --git a/packages/agentvault/activity.go b/packages/agentvault/activity.go index dce65111d..16f30db96 100644 --- a/packages/agentvault/activity.go +++ b/packages/agentvault/activity.go @@ -54,7 +54,7 @@ type forgottenSpool struct { } type activityShipper interface { - createChunk(final bool, sessionID string, req api.CreateAgentVaultActivityChunkRequest) (api.CreateAgentVaultActivityChunkResponse, error) + createChunk(ctx context.Context, final bool, sessionID string, req api.CreateAgentVaultActivityChunkRequest) (api.CreateAgentVaultActivityChunkResponse, error) putObject(ctx context.Context, url string, ciphertext []byte) error } @@ -435,8 +435,12 @@ func (a *activityLog) ackUploadReport(snapshot uint64) { func (a *activityLog) shipChunk(ctx context.Context, spool *activitySpool, chunk *sealedChunk, final bool) bool { if chunk.uploadURL == "" || a.now().Add(10*time.Second).After(chunk.urlExpires) { + // Past the shutdown budget, a new row could only be written for an upload that can no longer happen. + if ctx.Err() != nil { + return false + } grantsIssuedAtSend := activityGrantsIssued.Load() - res, err := a.shipper.createChunk(final, spool.sessionID, chunk.meta) + res, err := a.shipper.createChunk(ctx, final, spool.sessionID, chunk.meta) if err != nil { return a.handleCreateFailure(spool, chunk, err, grantsIssuedAtSend) } diff --git a/packages/agentvault/activity_ship.go b/packages/agentvault/activity_ship.go index 9ae615ca4..8e3917e94 100644 --- a/packages/agentvault/activity_ship.go +++ b/packages/agentvault/activity_ship.go @@ -73,12 +73,12 @@ func scrubURLError(err error) error { return err } -func (c *activityShipperClient) createChunk(final bool, sessionID string, req api.CreateAgentVaultActivityChunkRequest) (api.CreateAgentVaultActivityChunkResponse, error) { +func (c *activityShipperClient) createChunk(ctx context.Context, final bool, sessionID string, req api.CreateAgentVaultActivityChunkRequest) (api.CreateAgentVaultActivityChunkResponse, error) { client := c.steady if final { client = c.final } - return api.CallCreateAgentVaultActivityChunk(client, sessionID, req) + return api.CallCreateAgentVaultActivityChunk(ctx, client, sessionID, req) } func (c *activityShipperClient) putObject(ctx context.Context, uploadURL string, ciphertext []byte) error { diff --git a/packages/agentvault/activity_ship_test.go b/packages/agentvault/activity_ship_test.go index c4d9ab66e..9c22740ee 100644 --- a/packages/agentvault/activity_ship_test.go +++ b/packages/agentvault/activity_ship_test.go @@ -61,7 +61,7 @@ func TestTheChunkPostCarriesTheProxyTokenAndTheBucketPutDoesNot(t *testing.T) { shipper.put.Transport = bucket.Client().Transport ciphertext := []byte("sealed-bytes") - res, err := shipper.createChunk(false, "sess-1", api.CreateAgentVaultActivityChunkRequest{ + res, err := shipper.createChunk(context.Background(), false, "sess-1", api.CreateAgentVaultActivityChunkRequest{ ChunkID: "01K5ABCDEFGHJKMNPQRSTVWXYZ", RecordCount: 1, CiphertextBytes: len(ciphertext), diff --git a/packages/agentvault/activity_test.go b/packages/agentvault/activity_test.go index f284149a5..bedea763f 100644 --- a/packages/agentvault/activity_test.go +++ b/packages/agentvault/activity_test.go @@ -43,7 +43,7 @@ type fakeShipper struct { nextURL int } -func (f *fakeShipper) createChunk(final bool, sessionID string, req api.CreateAgentVaultActivityChunkRequest) (api.CreateAgentVaultActivityChunkResponse, error) { +func (f *fakeShipper) createChunk(_ context.Context, final bool, sessionID string, req api.CreateAgentVaultActivityChunkRequest) (api.CreateAgentVaultActivityChunkResponse, error) { f.mu.Lock() defer f.mu.Unlock() @@ -866,6 +866,20 @@ func TestAnUnreachableControlPlaneStopsTheTickAfterOneTimeout(t *testing.T) { } } +func TestShutdownPastItsBudgetStartsNoNewChunk(t *testing.T) { + shipper := &fakeShipper{} + log, _, _ := newTestLog(shipper) + + log.record(testGrant("s1"), aRecord("api.github.com")) + spent, cancel := context.WithCancel(context.Background()) + cancel() + log.close(spent) + + if len(shipper.posts()) != 0 { + t.Fatal("a chunk was posted after the shutdown budget ran out, leaving a row that can never be uploaded") + } +} + func TestShutdownDoesNotRaceTheRunLoop(t *testing.T) { shipper := &fakeShipper{} log, _, _ := newTestLog(shipper) diff --git a/packages/api/agent_vault.go b/packages/api/agent_vault.go index 93556aa34..193c33e51 100644 --- a/packages/api/agent_vault.go +++ b/packages/api/agent_vault.go @@ -1,6 +1,7 @@ package api import ( + "context" "fmt" "github.com/Infisical/infisical-merge/packages/config" @@ -161,10 +162,11 @@ type CreateAgentVaultActivityChunkResponse struct { ExpiresInSeconds int `json:"expiresInSeconds"` } -func CallCreateAgentVaultActivityChunk(httpClient *resty.Client, sessionID string, request CreateAgentVaultActivityChunkRequest) (CreateAgentVaultActivityChunkResponse, error) { +func CallCreateAgentVaultActivityChunk(ctx context.Context, httpClient *resty.Client, sessionID string, request CreateAgentVaultActivityChunkRequest) (CreateAgentVaultActivityChunkResponse, error) { var res CreateAgentVaultActivityChunkResponse response, err := httpClient. R(). + SetContext(ctx). SetResult(&res). SetHeader("User-Agent", USER_AGENT). SetBody(request). From 9fde7329ed70daa8550abcd5d0a8c7d1af8bfbbf Mon Sep 17 00:00:00 2001 From: saif <11242541+saifsmailbox98@users.noreply.github.com> Date: Thu, 24 Sep 2026 16:29:50 +0530 Subject: [PATCH 20/43] fix(agent-vault): flush each session's activity every minute, not every other --- packages/agentvault/activity.go | 21 +++++----- packages/agentvault/activity_test.go | 59 ++++++++++++++++++++++++++++ 2 files changed, 71 insertions(+), 9 deletions(-) diff --git a/packages/agentvault/activity.go b/packages/agentvault/activity.go index 16f30db96..ab7958bf1 100644 --- a/packages/agentvault/activity.go +++ b/packages/agentvault/activity.go @@ -12,6 +12,7 @@ import ( const ( activityFlushInterval = 60 * time.Second + activityFlushSlack = time.Second activityFlushRecords = 1000 activityMaxChunkPlaintext = 4 << 20 @@ -204,11 +205,10 @@ func (a *activityLog) close(ctx context.Context) { } } -func (a *activityLog) dueSpools(final bool) []*activitySpool { +func (a *activityLog) dueSpools(final bool, now time.Time) []*activitySpool { a.mu.Lock() defer a.mu.Unlock() - now := a.now() due := make([]*activitySpool, 0, len(a.spools)) for id, spool := range a.spools { if spool.ring.len() == 0 && len(spool.pending) == 0 { @@ -217,8 +217,9 @@ func (a *activityLog) dueSpools(final bool) []*activitySpool { } continue } + // The slack absorbs ticker jitter: without it a spool stamped at one tick is a hair short of due at the next. if final || spool.ring.len() >= activityFlushRecords || len(spool.pending) > 0 || - (spool.ring.len() > 0 && now.Sub(spool.lastFlushAt) >= activityFlushInterval) { + (spool.ring.len() > 0 && now.Sub(spool.lastFlushAt) >= activityFlushInterval-activityFlushSlack) { due = append(due, spool) } } @@ -291,17 +292,19 @@ func (a *activityLog) flushAll(ctx context.Context, final bool) { a.infisicalDown = false a.mu.Unlock() - for _, spool := range a.dueSpools(final) { - a.flushSpool(ctx, spool, final) + // One start time for every spool, so shipping the earlier ones doesn't make the later ones late for the next tick. + started := a.now() + for _, spool := range a.dueSpools(final, started) { + a.flushSpool(ctx, spool, final, started) } } -func (a *activityLog) sealRing(spool *activitySpool) { +func (a *activityLog) sealRing(spool *activitySpool, started time.Time) { for { a.mu.Lock() records := spool.ring.drain(activityFlushRecords) if len(records) == 0 { - spool.lastFlushAt = a.now() + spool.lastFlushAt = started a.mu.Unlock() return } @@ -384,8 +387,8 @@ func (a *activityLog) evictOldestLocked(spool *activitySpool) { Msg("agent-vault: dropped an unshipped activity chunk, the buffer is full") } -func (a *activityLog) flushSpool(ctx context.Context, spool *activitySpool, final bool) { - a.sealRing(spool) +func (a *activityLog) flushSpool(ctx context.Context, spool *activitySpool, final bool, started time.Time) { + a.sealRing(spool, started) if a.holding() { return } diff --git a/packages/agentvault/activity_test.go b/packages/agentvault/activity_test.go index bedea763f..3e5428bff 100644 --- a/packages/agentvault/activity_test.go +++ b/packages/agentvault/activity_test.go @@ -910,3 +910,62 @@ func TestShutdownDoesNotRaceTheRunLoop(t *testing.T) { } } } + +type slowPutShipper struct { + *fakeShipper + advance func(time.Duration) +} + +func (s *slowPutShipper) putObject(ctx context.Context, url string, ciphertext []byte) error { + s.advance(2 * time.Second) + return s.fakeShipper.putObject(ctx, url, ciphertext) +} + +func TestEverySessionShipsOnEveryTickWhileEarlierOnesTakeTimeToUpload(t *testing.T) { + shipper := &slowPutShipper{fakeShipper: &fakeShipper{}} + log, advance, _ := newTestLog(shipper) + shipper.advance = advance + + tickAt := log.now() + for i := 1; i <= 3; i++ { + log.record(testGrant("s1"), aRecord("api.github.com")) + log.record(testGrant("s2"), aRecord("api.github.com")) + tickAt = tickAt.Add(activityFlushInterval) + advance(tickAt.Sub(log.now())) + log.flushAll(context.Background(), false) + + if got := len(shipper.puts()); got != 2*i { + t.Fatalf("after tick %d there were %d uploads; every session should ship on every tick", i, got) + } + } +} + +func TestASessionShipsAtATickThatLandsJustShortOfAMinute(t *testing.T) { + shipper := &fakeShipper{} + log, advance, tick := newTestLog(shipper) + + log.record(testGrant("s1"), aRecord("api.github.com")) + tick() + log.record(testGrant("s1"), aRecord("api.github.com")) + advance(activityFlushInterval - 10*time.Millisecond) + log.flushAll(context.Background(), false) + + if got := len(shipper.puts()); got != 2 { + t.Fatalf("there were %d uploads; a tick a few milliseconds early skipped the session", got) + } +} + +func TestASessionDoesNotShipAgainHalfwayToTheNextTick(t *testing.T) { + shipper := &fakeShipper{} + log, advance, tick := newTestLog(shipper) + + log.record(testGrant("s1"), aRecord("api.github.com")) + tick() + log.record(testGrant("s1"), aRecord("api.github.com")) + advance(activityFlushInterval / 2) + log.flushAll(context.Background(), false) + + if got := len(shipper.puts()); got != 1 { + t.Fatalf("there were %d uploads; a session shipped twice within one interval", got) + } +} From dfeb2f024910f9a186eecda0260846bb45d31f65 Mon Sep 17 00:00:00 2001 From: saif <11242541+saifsmailbox98@users.noreply.github.com> Date: Fri, 25 Sep 2026 15:24:49 +0530 Subject: [PATCH 21/43] refactor(agent-vault): stop reporting activity uploads on the heartbeat --- packages/agentvault/activity.go | 24 --------- packages/agentvault/run.go | 9 +--- packages/agentvault/run_pollloop_test.go | 65 ------------------------ packages/api/agent_vault.go | 7 +-- 4 files changed, 3 insertions(+), 102 deletions(-) diff --git a/packages/agentvault/activity.go b/packages/agentvault/activity.go index ab7958bf1..fb1e8791a 100644 --- a/packages/agentvault/activity.go +++ b/packages/agentvault/activity.go @@ -86,9 +86,6 @@ type activityLog struct { clockSkewReported bool - uploads uint64 - uploadsReported uint64 - closed bool wake chan struct{} } @@ -411,31 +408,10 @@ func (a *activityLog) flushSpool(ctx context.Context, spool *activitySpool, fina spool.pending = spool.pending[1:] a.sealedBytes -= len(chunk.ciphertext) } - a.uploads++ a.mu.Unlock() } } -// What the next heartbeat reports. It is acknowledged only once Infisical has answered, so a heartbeat that -// fails reports the same uploads again. -func (a *activityLog) uploadReport() (snapshot uint64, uploaded bool) { - if a == nil { - return 0, false - } - a.mu.Lock() - defer a.mu.Unlock() - return a.uploads, a.uploads > a.uploadsReported -} - -func (a *activityLog) ackUploadReport(snapshot uint64) { - if a == nil { - return - } - a.mu.Lock() - defer a.mu.Unlock() - a.uploadsReported = max(a.uploadsReported, snapshot) -} - func (a *activityLog) shipChunk(ctx context.Context, spool *activitySpool, chunk *sealedChunk, final bool) bool { if chunk.uploadURL == "" || a.now().Add(10*time.Second).After(chunk.urlExpires) { // Past the shutdown budget, a new row could only be written for an upload that can no longer happen. diff --git a/packages/agentvault/run.go b/packages/agentvault/run.go index d7cfa4fab..37d5b6fbd 100644 --- a/packages/agentvault/run.go +++ b/packages/agentvault/run.go @@ -334,8 +334,7 @@ func (ps *proxyServer) tick(st *store) (tokenRejected bool) { httpClient, err := util.GetRestyClientWithCustomHeaders() if err == nil { httpClient.SetAuthToken(ps.opts.ProxyToken()).SetTimeout(controlPlaneTimeout) - uploads, uploaded := ps.activity.uploadReport() - res, hbErr := api.CallAgentVaultHeartbeat(httpClient, api.AgentVaultHeartbeatRequest{ActivityUploaded: uploaded}) + res, hbErr := api.CallAgentVaultHeartbeat(httpClient) if hbErr != nil { tokenRejected = isTokenRejected(hbErr) log.Warn().Err(hbErr).Msg("agent-vault: heartbeat failed") @@ -345,11 +344,7 @@ func (ps *proxyServer) tick(st *store) (tokenRejected bool) { AllowedHosts: res.Config.AllowedHosts, PollInterval: res.Config.PollInterval, } - usable := usableProxyConfig(next) - if usable { - ps.activity.ackUploadReport(uploads) - } - if !usable { + if !usableProxyConfig(next) { log.Warn(). Str("trafficPolicy", next.TrafficPolicy). Int("pollInterval", next.PollInterval). diff --git a/packages/agentvault/run_pollloop_test.go b/packages/agentvault/run_pollloop_test.go index b301a9c63..b593c19c1 100644 --- a/packages/agentvault/run_pollloop_test.go +++ b/packages/agentvault/run_pollloop_test.go @@ -6,7 +6,6 @@ import ( "net/http/httptest" "os" "path/filepath" - "sync" "testing" "time" @@ -162,67 +161,3 @@ func TestTickKeepsTheCurrentSettingsWhenTheHeartbeatCarriesNone(t *testing.T) { t.Fatalf("a heartbeat with no settings was persisted: %+v", back.Config) } } - -func TestHeartbeatsReportUploadsUntilInfisicalAnswers(t *testing.T) { - answered, err := json.Marshal(api.AgentVaultHeartbeatResponse{ - Config: api.AgentVaultProxyConfig{TrafficPolicy: TrafficPolicyAnyHost, PollInterval: 60}, - }) - if err != nil { - t.Fatal(err) - } - var mu sync.Mutex - var reported bool - reply := []byte(`{"message":"no"}`) - status := http.StatusBadRequest - srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - var req api.AgentVaultHeartbeatRequest - _ = json.NewDecoder(r.Body).Decode(&req) - mu.Lock() - reported = req.ActivityUploaded - code, body := status, reply - mu.Unlock() - w.Header().Set("Content-Type", "application/json") - w.WriteHeader(code) - _, _ = w.Write(body) - })) - t.Cleanup(srv.Close) - prev := config.INFISICAL_URL - config.INFISICAL_URL = srv.URL - t.Cleanup(func() { config.INFISICAL_URL = prev }) - - ps := &proxyServer{ - opts: Options{ProxyToken: func() string { return "tok" }}, - config: ProxyConfig{TrafficPolicy: TrafficPolicyAnyHost, PollInterval: 60}, - activity: newActivityLog("p1", &fakeShipper{}), - } - resolver, err := newInfisicalResolver(ps.opts.ProxyToken) - if err != nil { - t.Fatal(err) - } - ps.cache = newSessionCache(resolver, ps.pollInterval) - st := newStore(t.TempDir()) - ps.activity.uploads = 1 - - tickReports := func(code int, body []byte) bool { - mu.Lock() - status, reply = code, body - mu.Unlock() - ps.tick(st) - mu.Lock() - defer mu.Unlock() - return reported - } - - if !tickReports(http.StatusBadRequest, []byte(`{"message":"no"}`)) { - t.Fatal("an upload was not reported") - } - if !tickReports(http.StatusOK, []byte(`{"ok":true}`)) { - t.Fatal("a failed heartbeat acknowledged the upload") - } - if !tickReports(http.StatusOK, answered) { - t.Fatal("a reply that was not Infisical's acknowledged the upload") - } - if tickReports(http.StatusOK, answered) { - t.Fatal("an upload Infisical already acknowledged was reported again") - } -} diff --git a/packages/api/agent_vault.go b/packages/api/agent_vault.go index 193c33e51..c6eb860e8 100644 --- a/packages/api/agent_vault.go +++ b/packages/api/agent_vault.go @@ -48,21 +48,16 @@ func CallLoginAgentVaultProxy(httpClient *resty.Client, request LoginAgentVaultP return res, nil } -type AgentVaultHeartbeatRequest struct { - ActivityUploaded bool `json:"activityUploaded"` -} - type AgentVaultHeartbeatResponse struct { Config AgentVaultProxyConfig `json:"config"` } -func CallAgentVaultHeartbeat(httpClient *resty.Client, request AgentVaultHeartbeatRequest) (AgentVaultHeartbeatResponse, error) { +func CallAgentVaultHeartbeat(httpClient *resty.Client) (AgentVaultHeartbeatResponse, error) { var res AgentVaultHeartbeatResponse response, err := httpClient. R(). SetResult(&res). SetHeader("User-Agent", USER_AGENT). - SetBody(request). Post(fmt.Sprintf("%v/v1/agent-vault/proxy/heartbeat", config.INFISICAL_URL)) if err != nil { From 9e5695fd0f4e1f65e70d00840f58ee305039f4a5 Mon Sep 17 00:00:00 2001 From: saif <11242541+saifsmailbox98@users.noreply.github.com> Date: Fri, 25 Sep 2026 18:50:46 +0530 Subject: [PATCH 22/43] feat(agent-vault): send each activity chunk's SHA-256 so the browser can tell an edited object from a decryption failure --- packages/agentvault/activity_crypto.go | 7 +++++++ packages/agentvault/activity_crypto_test.go | 16 ++++++++++++++++ packages/agentvault/activity_spool.go | 19 ++++++++++--------- packages/api/agent_vault.go | 19 ++++++++++--------- 4 files changed, 43 insertions(+), 18 deletions(-) diff --git a/packages/agentvault/activity_crypto.go b/packages/agentvault/activity_crypto.go index bb43ada44..ad5b1a7c9 100644 --- a/packages/agentvault/activity_crypto.go +++ b/packages/agentvault/activity_crypto.go @@ -49,6 +49,13 @@ func encodeActivityIV(iv []byte) string { return base64.RawStdEncoding.EncodeToString(iv) } +// The browser checks the downloaded object against this before decrypting, so an edited object reads as +// changed rather than as a decryption failure. +func activityCiphertextSHA256(ciphertext []byte) string { + sum := sha256.Sum256(ciphertext) + return base64.RawStdEncoding.EncodeToString(sum[:]) +} + func newActivityChunkID(now time.Time) string { return ulid.MustNew(ulid.Timestamp(now), newULIDEntropy()).String() } diff --git a/packages/agentvault/activity_crypto_test.go b/packages/agentvault/activity_crypto_test.go index 77bf019bb..7c0deb2a0 100644 --- a/packages/agentvault/activity_crypto_test.go +++ b/packages/agentvault/activity_crypto_test.go @@ -4,6 +4,7 @@ import ( "bytes" "crypto/aes" "crypto/cipher" + "crypto/sha256" "encoding/base64" "encoding/hex" "encoding/json" @@ -82,6 +83,21 @@ func TestSealMatchesNodeVector(t *testing.T) { } } +func TestASealedChunkCarriesTheDigestOfExactlyWhatIsUploaded(t *testing.T) { + spool := newActivitySpool(newActivityGrant("sess-1", "proj-1", make([]byte, 32)), time.Now()) + chunk, err := spool.sealSlice("proxy-1", vectorRecords(), []byte("[]"), 0, time.Now()) + if err != nil { + t.Fatal(err) + } + sum := sha256.Sum256(chunk.ciphertext) + if want := base64.RawStdEncoding.EncodeToString(sum[:]); chunk.meta.CiphertextSha256 != want { + t.Fatalf("the chunk reports digest %q, its ciphertext hashes to %q", chunk.meta.CiphertextSha256, want) + } + if len(chunk.meta.CiphertextSha256) != 43 { + t.Fatalf("the digest is %d characters, the backend expects 43", len(chunk.meta.CiphertextSha256)) + } +} + func TestSealedChunkOpensWithTheTagAppended(t *testing.T) { key := mustHex(t, vectorKeyHex) aad := buildActivityAAD(vectorContext.projectID, vectorContext.sessionID, vectorContext.proxyID, vectorContext.chunkID) diff --git a/packages/agentvault/activity_spool.go b/packages/agentvault/activity_spool.go index 685b37231..91e9d8c85 100644 --- a/packages/agentvault/activity_spool.go +++ b/packages/agentvault/activity_spool.go @@ -185,15 +185,16 @@ func (s *activitySpool) sealSlice(proxyID string, records []activityRecord, plai first, last := records[0], records[len(records)-1] return &sealedChunk{ meta: api.CreateAgentVaultActivityChunkRequest{ - ChunkID: chunkID, - StartedAt: first.Ts, - EndedAt: last.Ts, - FirstSeq: first.Seq, - LastSeq: last.Seq, - RecordCount: len(records), - DroppedCount: dropped, - CiphertextBytes: len(ciphertext), - IV: encodeActivityIV(iv), + ChunkID: chunkID, + StartedAt: first.Ts, + EndedAt: last.Ts, + FirstSeq: first.Seq, + LastSeq: last.Seq, + RecordCount: len(records), + DroppedCount: dropped, + CiphertextBytes: len(ciphertext), + IV: encodeActivityIV(iv), + CiphertextSha256: activityCiphertextSHA256(ciphertext), }, ciphertext: ciphertext, }, nil diff --git a/packages/api/agent_vault.go b/packages/api/agent_vault.go index c6eb860e8..0821beca7 100644 --- a/packages/api/agent_vault.go +++ b/packages/api/agent_vault.go @@ -140,15 +140,16 @@ func CallResolveAgentVaultSession(httpClient *resty.Client, sessionToken string, } type CreateAgentVaultActivityChunkRequest struct { - ChunkID string `json:"chunkId"` - StartedAt string `json:"startedAt"` - EndedAt string `json:"endedAt"` - FirstSeq uint64 `json:"firstSeq"` - LastSeq uint64 `json:"lastSeq"` - RecordCount int `json:"recordCount"` - DroppedCount uint64 `json:"droppedCount"` - CiphertextBytes int `json:"ciphertextBytes"` - IV string `json:"iv"` + ChunkID string `json:"chunkId"` + StartedAt string `json:"startedAt"` + EndedAt string `json:"endedAt"` + FirstSeq uint64 `json:"firstSeq"` + LastSeq uint64 `json:"lastSeq"` + RecordCount int `json:"recordCount"` + DroppedCount uint64 `json:"droppedCount"` + CiphertextBytes int `json:"ciphertextBytes"` + IV string `json:"iv"` + CiphertextSha256 string `json:"ciphertextSha256"` } type CreateAgentVaultActivityChunkResponse struct { From 573139288ee68a78b68c1f01bae9173aae1238b6 Mon Sep 17 00:00:00 2001 From: saif <11242541+saifsmailbox98@users.noreply.github.com> Date: Sat, 26 Sep 2026 02:44:13 +0530 Subject: [PATCH 23/43] improvement(agent-vault): seal activity chunks under sessionId and chunkId only --- packages/agentvault/activity.go | 7 ++-- packages/agentvault/activity_crypto.go | 4 +-- packages/agentvault/activity_crypto_test.go | 37 ++++++++++++-------- packages/agentvault/activity_resolve_test.go | 17 +++------ packages/agentvault/activity_spool.go | 6 ++-- packages/agentvault/activity_test.go | 4 +-- packages/agentvault/activity_wiring_test.go | 2 +- packages/agentvault/resolve.go | 9 ++--- packages/api/agent_vault.go | 1 - 9 files changed, 39 insertions(+), 48 deletions(-) diff --git a/packages/agentvault/activity.go b/packages/agentvault/activity.go index fb1e8791a..a75a5932a 100644 --- a/packages/agentvault/activity.go +++ b/packages/agentvault/activity.go @@ -36,7 +36,6 @@ const ( type activityGrant struct { sessionID string - projectID string key []byte issued uint64 @@ -45,8 +44,8 @@ type activityGrant struct { // Counts every grant the proxy is handed, so a refusal can be told apart from a key issued after it. var activityGrantsIssued atomic.Uint64 -func newActivityGrant(sessionID, projectID string, key []byte) *activityGrant { - return &activityGrant{sessionID: sessionID, projectID: projectID, key: key, issued: activityGrantsIssued.Add(1)} +func newActivityGrant(sessionID string, key []byte) *activityGrant { + return &activityGrant{sessionID: sessionID, key: key, issued: activityGrantsIssued.Add(1)} } type forgottenSpool struct { @@ -321,7 +320,7 @@ func (a *activityLog) sealRing(spool *activitySpool, started time.Time) { groupDropped = dropped } - chunk, err := spool.sealSlice(a.proxyID, group.records, group.plaintext, groupDropped, now) + chunk, err := spool.sealSlice(group.records, group.plaintext, groupDropped, now) if err != nil { a.dropUnsealed(spool, len(group.records), groupDropped, err) continue diff --git a/packages/agentvault/activity_crypto.go b/packages/agentvault/activity_crypto.go index ad5b1a7c9..bb6451cb4 100644 --- a/packages/agentvault/activity_crypto.go +++ b/packages/agentvault/activity_crypto.go @@ -18,8 +18,8 @@ const activityAADVersion = "v1" const activityIVBytes = 12 // Must byte-match frontend activityDecrypt.ts and the vector pinned in agent-vault-activity-crypto.test.ts. -func buildActivityAAD(projectID, sessionID, proxyID, chunkID string) []byte { - sum := sha256.Sum256([]byte(fmt.Sprintf("%s|%s|%s|%s|%s", projectID, sessionID, proxyID, chunkID, activityAADVersion))) +func buildActivityAAD(sessionID, chunkID string) []byte { + sum := sha256.Sum256([]byte(fmt.Sprintf("%s|%s|%s", sessionID, chunkID, activityAADVersion))) return sum[:] } diff --git a/packages/agentvault/activity_crypto_test.go b/packages/agentvault/activity_crypto_test.go index 7c0deb2a0..a49ba8f16 100644 --- a/packages/agentvault/activity_crypto_test.go +++ b/packages/agentvault/activity_crypto_test.go @@ -17,15 +17,13 @@ import ( const ( vectorKeyHex = "000102030405060708090a0b0c0d0e0f101112131415161718191a1b1c1d1e1f" vectorIVHex = "aabbccddeeff001122334455" - vectorAADHex = "ba75c71ef714535e84246066ca0a34685c42a03a130dd92fe1d795ad40908a7c" + vectorAADHex = "0bc4c5b3d6ea7cd6bfc440da46d6ce9b73f17c5efa64e90e88373ee8ba09a837" vectorIVBase64 = "qrvM3e7/ABEiM0RV" - vectorCiphertext = "PLRwxBbgu+W68Br1N9gY1oUy8wjJxQClAtBh0NfJS1UcWOCPn3laS615sIqwFONhPIPNWRI3CA+a5tUJ7aoim0sQkE4d9gzou2mc/AWiCdToVBJPtdumA9jIzh3yAI81YPwcoDXEVnq2+7ooNNJShGdLX95itbrna/t4nFKRKSSgNzbH23eMtSMcSo72puk/2iwh4sVbTKzC2kwvbf1U6Mgd21zkIq2jDKKwhcT6mTfjPivW4FzmmkspQVMoWwANRX+QVyXzrMipZfoq5N/UcUI6rCvav2ddgiSoqXrTvwiXaUgv" + vectorCiphertext = "PLRwxBbgu+W68Br1N9gY1oUy8wjJxQClAtBh0NfJS1UcWOCPn3laS615sIqwFONhPIPNWRI3CA+a5tUJ7aoim0sQkE4d9gzou2mc/AWiCdToVBJPtdumA9jIzh3yAI81YPwcoDXEVnq2+7ooNNJShGdLX95itbrna/t4nFKRKSSgNzbH23eMtSMcSo72puk/2iwh4sVbTKzC2kwvbf1U6Mgd21zkIq2jDKKwhcT6mTfjPivW4FzmmkspQVMoWwANRX+QVyXzrMipZfoq5N/UcUI6rCu64JVIU0dTBbrrs+2AuZxL" ) -var vectorContext = struct{ projectID, sessionID, proxyID, chunkID string }{ - projectID: "proj-1", +var vectorContext = struct{ sessionID, chunkID string }{ sessionID: "sess-1", - proxyID: "proxy-1", chunkID: "01K5ABCDEFGHJKMNPQRSTVWXYZ", } @@ -56,7 +54,7 @@ func mustHex(t *testing.T, s string) []byte { } func TestActivityAADMatchesTheBackendVector(t *testing.T) { - got := buildActivityAAD(vectorContext.projectID, vectorContext.sessionID, vectorContext.proxyID, vectorContext.chunkID) + got := buildActivityAAD(vectorContext.sessionID, vectorContext.chunkID) if hex.EncodeToString(got) != vectorAADHex { t.Fatalf("AAD is %s, the backend and the browser build %s", hex.EncodeToString(got), vectorAADHex) } @@ -69,7 +67,7 @@ func TestSealMatchesNodeVector(t *testing.T) { } iv := mustHex(t, vectorIVHex) - aad := buildActivityAAD(vectorContext.projectID, vectorContext.sessionID, vectorContext.proxyID, vectorContext.chunkID) + aad := buildActivityAAD(vectorContext.sessionID, vectorContext.chunkID) ciphertext, gotIV, err := sealActivityWithRand(bytes.NewReader(iv), mustHex(t, vectorKeyHex), plaintext, aad) if err != nil { t.Fatal(err) @@ -84,8 +82,9 @@ func TestSealMatchesNodeVector(t *testing.T) { } func TestASealedChunkCarriesTheDigestOfExactlyWhatIsUploaded(t *testing.T) { - spool := newActivitySpool(newActivityGrant("sess-1", "proj-1", make([]byte, 32)), time.Now()) - chunk, err := spool.sealSlice("proxy-1", vectorRecords(), []byte("[]"), 0, time.Now()) + key := make([]byte, 32) + spool := newActivitySpool(newActivityGrant("sess-1", key), time.Now()) + chunk, err := spool.sealSlice(vectorRecords(), []byte("[]"), 0, time.Now()) if err != nil { t.Fatal(err) } @@ -96,11 +95,21 @@ func TestASealedChunkCarriesTheDigestOfExactlyWhatIsUploaded(t *testing.T) { if len(chunk.meta.CiphertextSha256) != 43 { t.Fatalf("the digest is %d characters, the backend expects 43", len(chunk.meta.CiphertextSha256)) } + + iv, err := base64.RawStdEncoding.DecodeString(chunk.meta.IV) + if err != nil { + t.Fatal(err) + } + block, _ := aes.NewCipher(key) + gcm, _ := cipher.NewGCM(block) + if _, err := gcm.Open(nil, iv, chunk.ciphertext, buildActivityAAD("sess-1", chunk.meta.ChunkID)); err != nil { + t.Fatalf("the chunk does not open under its session and chunk ID: %v", err) + } } func TestSealedChunkOpensWithTheTagAppended(t *testing.T) { key := mustHex(t, vectorKeyHex) - aad := buildActivityAAD(vectorContext.projectID, vectorContext.sessionID, vectorContext.proxyID, vectorContext.chunkID) + aad := buildActivityAAD(vectorContext.sessionID, vectorContext.chunkID) plaintext, _ := json.Marshal(vectorRecords()) ciphertext, iv, err := sealActivity(key, plaintext, aad) @@ -125,7 +134,7 @@ func TestSealedChunkOpensWithTheTagAppended(t *testing.T) { func TestAChunkCannotBeReplayedElsewhere(t *testing.T) { key := mustHex(t, vectorKeyHex) plaintext, _ := json.Marshal(vectorRecords()) - aad := buildActivityAAD(vectorContext.projectID, vectorContext.sessionID, vectorContext.proxyID, vectorContext.chunkID) + aad := buildActivityAAD(vectorContext.sessionID, vectorContext.chunkID) ciphertext, iv, err := sealActivity(key, plaintext, aad) if err != nil { t.Fatal(err) @@ -138,10 +147,8 @@ func TestAChunkCannotBeReplayedElsewhere(t *testing.T) { name string aad []byte }{ - {"another project", buildActivityAAD("other", vectorContext.sessionID, vectorContext.proxyID, vectorContext.chunkID)}, - {"another session", buildActivityAAD(vectorContext.projectID, "other", vectorContext.proxyID, vectorContext.chunkID)}, - {"another proxy", buildActivityAAD(vectorContext.projectID, vectorContext.sessionID, "other", vectorContext.chunkID)}, - {"another chunk", buildActivityAAD(vectorContext.projectID, vectorContext.sessionID, vectorContext.proxyID, "other")}, + {"another session", buildActivityAAD("other", vectorContext.chunkID)}, + {"another chunk", buildActivityAAD(vectorContext.sessionID, "other")}, } { if _, err := gcm.Open(nil, iv, ciphertext, wrong.aad); err == nil { t.Fatalf("a chunk opened under %s", wrong.name) diff --git a/packages/agentvault/activity_resolve_test.go b/packages/agentvault/activity_resolve_test.go index a4964b115..36140b8c4 100644 --- a/packages/agentvault/activity_resolve_test.go +++ b/packages/agentvault/activity_resolve_test.go @@ -8,7 +8,7 @@ import ( ) func enabledGrantWire(key string) api.AgentVaultActivityGrant { - return api.AgentVaultActivityGrant{Enabled: true, SessionKey: key, ProjectID: "proj-1"} + return api.AgentVaultActivityGrant{Enabled: true, SessionKey: key} } func aKey(b byte) []byte { @@ -26,8 +26,8 @@ func TestTheFirstResolveTakesTheKeyOffTheWire(t *testing.T) { if got == nil { t.Fatal("activity was enabled but no grant was built") } - if got.sessionID != "s1" || got.projectID != "proj-1" { - t.Fatalf("grant names session %q project %q", got.sessionID, got.projectID) + if got.sessionID != "s1" { + t.Fatalf("grant names session %q", got.sessionID) } if string(got.key) != string(want) { t.Fatal("the key on the grant is not the key Infisical sent") @@ -35,7 +35,7 @@ func TestTheFirstResolveTakesTheKeyOffTheWire(t *testing.T) { } func TestACachedKeySurvivesAResolveThatOmitsIt(t *testing.T) { - held := &activityGrant{sessionID: "s1", projectID: "proj-1", key: aKey(9)} + held := &activityGrant{sessionID: "s1", key: aKey(9)} got := toActivityGrant("s1", enabledGrantWire(""), held) @@ -54,7 +54,7 @@ func TestNoKeyAndNoCachedCopyMeansNoRecording(t *testing.T) { } func TestActivityBeingOffClearsAnyCachedGrant(t *testing.T) { - held := &activityGrant{sessionID: "s1", projectID: "proj-1", key: aKey(9)} + held := &activityGrant{sessionID: "s1", key: aKey(9)} if got := toActivityGrant("s1", api.AgentVaultActivityGrant{Enabled: false}, held); got != nil { t.Fatal("the proxy kept recording after logging was switched off") @@ -75,10 +75,3 @@ func TestAnUnusableKeyIsRefusedRatherThanUsed(t *testing.T) { } } } - -func TestAGrantWithoutAProjectIsRefused(t *testing.T) { - wire := api.AgentVaultActivityGrant{Enabled: true, SessionKey: base64.StdEncoding.EncodeToString(aKey(7))} - if got := toActivityGrant("s1", wire, nil); got != nil { - t.Fatal("a grant was built with no project named") - } -} diff --git a/packages/agentvault/activity_spool.go b/packages/agentvault/activity_spool.go index 91e9d8c85..d92101a0c 100644 --- a/packages/agentvault/activity_spool.go +++ b/packages/agentvault/activity_spool.go @@ -114,7 +114,6 @@ func (c *sealedChunk) lostCount() uint64 { type activitySpool struct { sessionID string - projectID string key []byte ring activityRing @@ -129,7 +128,6 @@ type activitySpool struct { func newActivitySpool(g *activityGrant, now time.Time) *activitySpool { return &activitySpool{ sessionID: g.sessionID, - projectID: g.projectID, key: g.key, ring: newActivityRing(activitySpoolCapacity), lastRecordAt: now, @@ -174,9 +172,9 @@ func packActivityRecords(records []activityRecord) ([]activityGroup, error) { return groups, nil } -func (s *activitySpool) sealSlice(proxyID string, records []activityRecord, plaintext []byte, dropped uint64, now time.Time) (*sealedChunk, error) { +func (s *activitySpool) sealSlice(records []activityRecord, plaintext []byte, dropped uint64, now time.Time) (*sealedChunk, error) { chunkID := newActivityChunkID(now) - aad := buildActivityAAD(s.projectID, s.sessionID, proxyID, chunkID) + aad := buildActivityAAD(s.sessionID, chunkID) ciphertext, iv, err := sealActivity(s.key, plaintext, aad) if err != nil { return nil, err diff --git a/packages/agentvault/activity_test.go b/packages/agentvault/activity_test.go index 3e5428bff..7b42f4134 100644 --- a/packages/agentvault/activity_test.go +++ b/packages/agentvault/activity_test.go @@ -118,7 +118,7 @@ func apiErr(status int, name string) error { } func testGrant(sessionID string) *activityGrant { - return newActivityGrant(sessionID, "proj-1", make([]byte, 32)) + return newActivityGrant(sessionID, make([]byte, 32)) } func newTestLog(shipper activityShipper) (log *activityLog, advance func(time.Duration), tick func()) { @@ -829,7 +829,7 @@ func TestRecordsLostToASealFailureAreStillCounted(t *testing.T) { shipper := &fakeShipper{} log, _, tick := newTestLog(shipper) - grant := &activityGrant{sessionID: "s1", projectID: "proj-1", key: make([]byte, 7)} + grant := &activityGrant{sessionID: "s1", key: make([]byte, 7)} for i := 0; i < 3; i++ { log.record(grant, aRecord("api.github.com")) } diff --git a/packages/agentvault/activity_wiring_test.go b/packages/agentvault/activity_wiring_test.go index 93077f403..00c115166 100644 --- a/packages/agentvault/activity_wiring_test.go +++ b/packages/agentvault/activity_wiring_test.go @@ -19,7 +19,7 @@ func (g grantingResolver) resolve(string, *activityGrant) (*resolveResult, error return &resolveResult{ SessionID: "s1", Services: g.services, - Activity: &activityGrant{sessionID: "s1", projectID: "proj-1", key: make([]byte, 32)}, + Activity: &activityGrant{sessionID: "s1", key: make([]byte, 32)}, }, nil } diff --git a/packages/agentvault/resolve.go b/packages/agentvault/resolve.go index 7d29559cb..50dbccc6b 100644 --- a/packages/agentvault/resolve.go +++ b/packages/agentvault/resolve.go @@ -86,14 +86,9 @@ func toActivityGrant(sessionID string, wire api.AgentVaultActivityGrant, held *a if !wire.Enabled { return nil } - if wire.ProjectID == "" { - log.Warn().Str("sessionId", sessionID).Msg("agent-vault: activity is enabled but no project was named, not recording") - return nil - } - if wire.SessionKey == "" { if held != nil { - return newActivityGrant(sessionID, wire.ProjectID, held.key) + return newActivityGrant(sessionID, held.key) } log.Warn().Str("sessionId", sessionID).Msg("agent-vault: activity is enabled but no key was sent, not recording") return nil @@ -105,7 +100,7 @@ func toActivityGrant(sessionID string, wire api.AgentVaultActivityGrant, held *a return nil } - return newActivityGrant(sessionID, wire.ProjectID, key) + return newActivityGrant(sessionID, key) } func toCredential(wire api.AgentVaultCredential) credential { diff --git a/packages/api/agent_vault.go b/packages/api/agent_vault.go index 0821beca7..8849d79ec 100644 --- a/packages/api/agent_vault.go +++ b/packages/api/agent_vault.go @@ -106,7 +106,6 @@ type AgentVaultService struct { type AgentVaultActivityGrant struct { Enabled bool `json:"enabled"` SessionKey string `json:"sessionKey"` - ProjectID string `json:"projectId"` } type ResolveAgentVaultSessionRequest struct { From 46083b1529a778c8a012c043989c5ea41f9ce308 Mon Sep 17 00:00:00 2001 From: saif <11242541+saifsmailbox98@users.noreply.github.com> Date: Sat, 26 Sep 2026 06:59:42 +0530 Subject: [PATCH 24/43] refactor(agent-vault): rename activity logs to session logs, on the wire and in the code --- packages/agentvault/cache.go | 44 ++--- packages/agentvault/cache_test.go | 4 +- packages/agentvault/policy.go | 2 +- packages/agentvault/proxy.go | 18 +- packages/agentvault/proxy_policy_test.go | 2 +- .../agentvault/proxy_requesttarget_test.go | 4 +- packages/agentvault/proxy_truncation_test.go | 2 +- .../agentvault/proxy_tunnel_errors_test.go | 4 +- packages/agentvault/resolve.go | 34 ++-- packages/agentvault/run.go | 20 +-- .../{activity.go => session_log.go} | 156 +++++++++--------- ...tivity_crypto.go => session_log_crypto.go} | 26 +-- ...pto_test.go => session_log_crypto_test.go} | 44 ++--- ...ve_test.go => session_log_resolve_test.go} | 24 +-- .../{activity_ship.go => session_log_ship.go} | 20 +-- ..._ship_test.go => session_log_ship_test.go} | 16 +- ...activity_spool.go => session_log_spool.go} | 72 ++++---- .../{activity_test.go => session_log_test.go} | 144 ++++++++-------- ...ing_test.go => session_log_wiring_test.go} | 18 +- packages/api/agent_vault.go | 26 +-- 20 files changed, 340 insertions(+), 340 deletions(-) rename packages/agentvault/{activity.go => session_log.go} (63%) rename packages/agentvault/{activity_crypto.go => session_log_crypto.go} (55%) rename packages/agentvault/{activity_crypto_test.go => session_log_crypto_test.go} (76%) rename packages/agentvault/{activity_resolve_test.go => session_log_resolve_test.go} (61%) rename packages/agentvault/{activity_ship.go => session_log_ship.go} (76%) rename packages/agentvault/{activity_ship_test.go => session_log_ship_test.go} (90%) rename packages/agentvault/{activity_spool.go => session_log_spool.go} (59%) rename packages/agentvault/{activity_test.go => session_log_test.go} (86%) rename packages/agentvault/{activity_wiring_test.go => session_log_wiring_test.go} (93%) diff --git a/packages/agentvault/cache.go b/packages/agentvault/cache.go index ce11199f0..f68d35048 100644 --- a/packages/agentvault/cache.go +++ b/packages/agentvault/cache.go @@ -65,12 +65,12 @@ type resolvedService struct { } type sessionEntry struct { - sessionID string - expiresAt *time.Time - services []*resolvedService - activity *activityGrant - lastSeen time.Time - fetchedAt time.Time + sessionID string + expiresAt *time.Time + services []*resolvedService + sessionLog *sessionLogGrant + lastSeen time.Time + fetchedAt time.Time } // The map key is the sha256 of the token, never the token itself, so a heap dump yields no live credential. @@ -80,7 +80,7 @@ func sessionKey(token string) string { } type sessionResolver interface { - resolve(sessionToken string, held *activityGrant) (*resolveResult, error) + resolve(sessionToken string, held *sessionLogGrant) (*resolveResult, error) } type sessionCache struct { @@ -159,11 +159,11 @@ func (c *sessionCache) get(sessionToken string) ([]*resolvedService, error) { } type cacheLookup struct { - services []*resolvedService - activity *activityGrant + services []*resolvedService + sessionLog *sessionLogGrant } -func (c *sessionCache) lookup(sessionToken string) ([]*resolvedService, *activityGrant, error) { +func (c *sessionCache) lookup(sessionToken string) ([]*resolvedService, *sessionLogGrant, error) { key := sessionKey(sessionToken) c.mu.Lock() @@ -181,7 +181,7 @@ func (c *sessionCache) lookup(sessionToken string) ([]*resolvedService, *activit delete(c.tokens, key) } else { entry.lastSeen = time.Now() - svcs, grant := entry.services, entry.activity + svcs, grant := entry.services, entry.sessionLog c.mu.Unlock() return svcs, grant, nil } @@ -212,21 +212,21 @@ func (c *sessionCache) lookup(sessionToken string) ([]*resolvedService, *activit defer c.mu.Unlock() c.evictIfFullLocked() c.entries[key] = &sessionEntry{ - sessionID: result.SessionID, - expiresAt: result.ExpiresAt, - services: result.Services, - activity: result.Activity, - lastSeen: time.Now(), - fetchedAt: time.Now(), + sessionID: result.SessionID, + expiresAt: result.ExpiresAt, + services: result.Services, + sessionLog: result.SessionLog, + lastSeen: time.Now(), + fetchedAt: time.Now(), } c.tokens[key] = sessionToken - return cacheLookup{services: result.Services, activity: result.Activity}, nil + return cacheLookup{services: result.Services, sessionLog: result.SessionLog}, nil }) if err != nil { return nil, nil, err } out := resolved.(cacheLookup) - return out.services, out.activity, nil + return out.services, out.sessionLog, nil } func (c *sessionCache) evictIfFullLocked() { @@ -303,9 +303,9 @@ func (c *sessionCache) refresh() { func (c *sessionCache) refreshOne(key, token string) { c.mu.Lock() - var held *activityGrant + var held *sessionLogGrant if entry, ok := c.entries[key]; ok { - held = entry.activity + held = entry.sessionLog } c.mu.Unlock() @@ -321,7 +321,7 @@ func (c *sessionCache) refreshOne(key, token string) { entry.sessionID = result.SessionID entry.expiresAt = result.ExpiresAt entry.services = result.Services - entry.activity = result.Activity + entry.sessionLog = result.SessionLog entry.fetchedAt = time.Now() } } diff --git a/packages/agentvault/cache_test.go b/packages/agentvault/cache_test.go index ce36e37fa..c9265c881 100644 --- a/packages/agentvault/cache_test.go +++ b/packages/agentvault/cache_test.go @@ -16,10 +16,10 @@ type stubResolver struct { result *resolveResult err error delay time.Duration - lastHeld *activityGrant + lastHeld *sessionLogGrant } -func (s *stubResolver) resolve(_ string, held *activityGrant) (*resolveResult, error) { +func (s *stubResolver) resolve(_ string, held *sessionLogGrant) (*resolveResult, error) { s.mu.Lock() s.calls++ s.lastHeld = held diff --git a/packages/agentvault/policy.go b/packages/agentvault/policy.go index f5ee7eacd..0a03105b8 100644 --- a/packages/agentvault/policy.go +++ b/packages/agentvault/policy.go @@ -104,7 +104,7 @@ func requestPath(req *http.Request) string { path := req.URL.EscapedPath() if path == "" { // forward refuses an opaque target before any policy reads the path, so this branch is what the log - // line and the activity record show for one. A genuinely empty path is the root. + // line and the session log record show for one. A genuinely empty path is the root. if req.URL.Opaque != "" { return req.URL.Opaque } diff --git a/packages/agentvault/proxy.go b/packages/agentvault/proxy.go index 62791717a..c33007213 100644 --- a/packages/agentvault/proxy.go +++ b/packages/agentvault/proxy.go @@ -75,11 +75,11 @@ type Options struct { } type proxyServer struct { - opts Options - ca *caManager - cache *sessionCache - activity *activityLog - transport http.RoundTripper + opts Options + ca *caManager + cache *sessionCache + sessionLogs *sessionLogRecorder + transport http.RoundTripper configMu sync.RWMutex config ProxyConfig @@ -420,12 +420,12 @@ func (ps *proxyServer) forwardHTTP(w http.ResponseWriter, r *http.Request, schem event.Msg("agent-vault: request") // reqPath was taken before forward, so a credential substituted into the path never reaches the record. - if outcome.activity != nil { + if outcome.sessionLog != nil { var service, bundle *string if matched != nil { service, bundle = &matched.name, &matched.accessBundleName } - ps.activity.record(outcome.activity, activityRecord{ + ps.sessionLogs.record(outcome.sessionLog, sessionLogRecord{ Method: reqMethod, Host: hostname, Port: port, @@ -476,7 +476,7 @@ func (ps *proxyServer) blocksOffBundle(matched *resolvedService, hostname, port type forwardOutcome struct { brokered bool substituted []string - activity *activityGrant + sessionLog *sessionLogGrant } func (ps *proxyServer) forward(req *http.Request, scheme, hostname, port, sessionToken string) (*http.Response, *resolvedService, forwardOutcome, error) { @@ -486,7 +486,7 @@ func (ps *proxyServer) forward(req *http.Request, scheme, hostname, port, sessio if err != nil { return nil, nil, outcome, fmt.Errorf("%w: %w", errSessionResolve, err) } - outcome.activity = grant + outcome.sessionLog = grant // TRACE and TRACK make the upstream reflect the injected credential back in the response body. Upper // -cased like allowsMethod already was, or a lowercase "trace" walks past. Refused here rather than in diff --git a/packages/agentvault/proxy_policy_test.go b/packages/agentvault/proxy_policy_test.go index 24b47b67d..3b89f020b 100644 --- a/packages/agentvault/proxy_policy_test.go +++ b/packages/agentvault/proxy_policy_test.go @@ -27,7 +27,7 @@ type echoed struct { type fixedResolver struct{ services []*resolvedService } -func (r fixedResolver) resolve(string, *activityGrant) (*resolveResult, error) { +func (r fixedResolver) resolve(string, *sessionLogGrant) (*resolveResult, error) { return &resolveResult{SessionID: "s1", Services: r.services}, nil } diff --git a/packages/agentvault/proxy_requesttarget_test.go b/packages/agentvault/proxy_requesttarget_test.go index 4fff2ca76..bc8dc076f 100644 --- a/packages/agentvault/proxy_requesttarget_test.go +++ b/packages/agentvault/proxy_requesttarget_test.go @@ -134,7 +134,7 @@ func TestAnOpaqueRequestTargetIsRecordedAsBlocked(t *testing.T) { ps := &proxyServer{transport: newUpstreamTransport(), ca: newCaManager(key, cert)} ps.setConfig(ProxyConfig{TrafficPolicy: TrafficPolicyAnyHost}) ps.cache = newSessionCache(grantingResolver{}, ps.pollInterval) - ps.activity = newActivityLog("proxy-1", &fakeShipper{}) + ps.sessionLogs = newSessionLogRecorder("proxy-1", &fakeShipper{}) front := httptest.NewServer(http.HandlerFunc(ps.dispatch)) t.Cleanup(front.Close) fu, _ := url.Parse(front.URL) @@ -144,7 +144,7 @@ func TestAnOpaqueRequestTargetIsRecordedAsBlocked(t *testing.T) { t.Fatalf("status = %d, want 400", resp.StatusCode) } - got := drainOneRecord(t, ps.activity) + got := drainOneRecord(t, ps.sessionLogs) if got.Decision != decisionBlocked || got.Status != http.StatusBadRequest || got.Path != "admin/secrets" { t.Fatalf("record is %+v, expected a blocked 400 for admin/secrets", got) } diff --git a/packages/agentvault/proxy_truncation_test.go b/packages/agentvault/proxy_truncation_test.go index 96b887b5d..22c87decc 100644 --- a/packages/agentvault/proxy_truncation_test.go +++ b/packages/agentvault/proxy_truncation_test.go @@ -17,7 +17,7 @@ import ( type sessionOnlyResolver struct{} -func (sessionOnlyResolver) resolve(string, *activityGrant) (*resolveResult, error) { +func (sessionOnlyResolver) resolve(string, *sessionLogGrant) (*resolveResult, error) { return &resolveResult{SessionID: "s1"}, nil } diff --git a/packages/agentvault/proxy_tunnel_errors_test.go b/packages/agentvault/proxy_tunnel_errors_test.go index a8642600a..6a0742a9c 100644 --- a/packages/agentvault/proxy_tunnel_errors_test.go +++ b/packages/agentvault/proxy_tunnel_errors_test.go @@ -19,7 +19,7 @@ import ( type expiringResolver struct{ ttl time.Duration } -func (r expiringResolver) resolve(string, *activityGrant) (*resolveResult, error) { +func (r expiringResolver) resolve(string, *sessionLogGrant) (*resolveResult, error) { exp := time.Now().Add(r.ttl) return &resolveResult{SessionID: "s1", ExpiresAt: &exp}, nil } @@ -94,7 +94,7 @@ func TestAnUpstreamFailureInsideTheTunnelKeepsTheDetailOutOfTheBody(t *testing.T type rejectedProxyResolver struct{} -func (rejectedProxyResolver) resolve(string, *activityGrant) (*resolveResult, error) { +func (rejectedProxyResolver) resolve(string, *sessionLogGrant) (*resolveResult, error) { return nil, &api.APIError{StatusCode: 401, Name: proxyTokenRejectedName, ErrorMessage: "Agent Vault proxy token has been revoked"} } diff --git a/packages/agentvault/resolve.go b/packages/agentvault/resolve.go index 50dbccc6b..3545d7899 100644 --- a/packages/agentvault/resolve.go +++ b/packages/agentvault/resolve.go @@ -19,13 +19,13 @@ const ( ) type resolveResult struct { - SessionID string - ExpiresAt *time.Time - Services []*resolvedService - Activity *activityGrant + SessionID string + ExpiresAt *time.Time + Services []*resolvedService + SessionLog *sessionLogGrant } -const activityKeyBytes = 32 +const sessionLogKeyBytes = 32 // A seam so the cache can be tested without a server, not because a second implementation is expected. type infisicalResolver struct { @@ -44,9 +44,9 @@ func newInfisicalResolver(proxyToken func() string) (*infisicalResolver, error) return &infisicalResolver{client: client}, nil } -func (r *infisicalResolver) resolve(sessionToken string, held *activityGrant) (*resolveResult, error) { +func (r *infisicalResolver) resolve(sessionToken string, held *sessionLogGrant) (*resolveResult, error) { res, err := api.CallResolveAgentVaultSession(r.client, sessionToken, api.ResolveAgentVaultSessionRequest{ - HasActivityKey: held != nil, + HasSessionLogKey: held != nil, }) if err != nil { return nil, err @@ -75,32 +75,32 @@ func (r *infisicalResolver) resolve(sessionToken string, held *activityGrant) (* } return &resolveResult{ - SessionID: res.SessionID, - ExpiresAt: expiresAt, - Services: services, - Activity: toActivityGrant(res.SessionID, res.Activity, held), + SessionID: res.SessionID, + ExpiresAt: expiresAt, + Services: services, + SessionLog: toSessionLogGrant(res.SessionID, res.SessionLogs, held), }, nil } -func toActivityGrant(sessionID string, wire api.AgentVaultActivityGrant, held *activityGrant) *activityGrant { +func toSessionLogGrant(sessionID string, wire api.AgentVaultSessionLogGrant, held *sessionLogGrant) *sessionLogGrant { if !wire.Enabled { return nil } if wire.SessionKey == "" { if held != nil { - return newActivityGrant(sessionID, held.key) + return newSessionLogGrant(sessionID, held.key) } - log.Warn().Str("sessionId", sessionID).Msg("agent-vault: activity is enabled but no key was sent, not recording") + log.Warn().Str("sessionId", sessionID).Msg("agent-vault: session logs are on but no key was sent, not recording") return nil } key, err := base64.StdEncoding.DecodeString(wire.SessionKey) - if err != nil || len(key) != activityKeyBytes { - log.Warn().Str("sessionId", sessionID).Msg("agent-vault: the activity key Infisical sent is unusable, not recording") + if err != nil || len(key) != sessionLogKeyBytes { + log.Warn().Str("sessionId", sessionID).Msg("agent-vault: the session log key Infisical sent is unusable, not recording") return nil } - return newActivityGrant(sessionID, key) + return newSessionLogGrant(sessionID, key) } func toCredential(wire api.AgentVaultCredential) credential { diff --git a/packages/agentvault/run.go b/packages/agentvault/run.go index 37d5b6fbd..c57463b83 100644 --- a/packages/agentvault/run.go +++ b/packages/agentvault/run.go @@ -203,11 +203,11 @@ func Start(opts Options, enrollmentToken string) error { } ps.cache = newSessionCache(resolver, ps.pollInterval) - shipper, err := newActivityShipper(opts.ProxyToken) + shipper, err := newSessionLogShipper(opts.ProxyToken) if err != nil { return err } - ps.activity = newActivityLog(state.ProxyID, shipper) + ps.sessionLogs = newSessionLogRecorder(state.ProxyID, shipper) // Port 0 is not "unset": it is the ordinary ask for any free port, so it is never substituted. listener, err := net.Listen("tcp", fmt.Sprintf(":%d", opts.Port)) @@ -231,7 +231,7 @@ func Start(opts Options, enrollmentToken string) error { stop := make(chan struct{}) fatal := make(chan error, 1) go ps.pollLoop(st, stop, fatal) - go ps.activity.run(stop) + go ps.sessionLogs.run(stop) serveErr := make(chan error, 1) go func() { serveErr <- front.Serve(limited) }() @@ -248,17 +248,17 @@ func Start(opts Options, enrollmentToken string) error { signals := make(chan os.Signal, 1) signal.Notify(signals, os.Interrupt, syscall.SIGTERM) - // Its own budget: in-flight requests can spend all of the shutdown's, and this is the last minute of activity. - closeActivity := func() { - ctx, cancel := context.WithTimeout(context.Background(), activityCloseTimeout) + // Its own budget: in-flight requests can spend all of the shutdown's, and this is the last minute of session logs. + closeSessionLogs := func() { + ctx, cancel := context.WithTimeout(context.Background(), sessionLogCloseTimeout) defer cancel() - ps.activity.close(ctx) + ps.sessionLogs.close(ctx) } select { case err := <-serveErr: close(stop) - closeActivity() + closeSessionLogs() ps.cache.close() if errors.Is(err, http.ErrServerClosed) { return nil @@ -270,7 +270,7 @@ func Start(opts Options, enrollmentToken string) error { ctx, cancel := context.WithTimeout(context.Background(), 10*time.Second) defer cancel() _ = front.Shutdown(ctx) - closeActivity() + closeSessionLogs() ps.cache.close() return nil case err := <-fatal: @@ -278,7 +278,7 @@ func Start(opts Options, enrollmentToken string) error { ctx, cancel := context.WithTimeout(context.Background(), 10*time.Second) defer cancel() _ = front.Shutdown(ctx) - closeActivity() + closeSessionLogs() ps.cache.close() return err } diff --git a/packages/agentvault/activity.go b/packages/agentvault/session_log.go similarity index 63% rename from packages/agentvault/activity.go rename to packages/agentvault/session_log.go index a75a5932a..23891fd64 100644 --- a/packages/agentvault/activity.go +++ b/packages/agentvault/session_log.go @@ -11,30 +11,30 @@ import ( ) const ( - activityFlushInterval = 60 * time.Second - activityFlushSlack = time.Second - activityFlushRecords = 1000 - activityMaxChunkPlaintext = 4 << 20 + sessionLogFlushInterval = 60 * time.Second + sessionLogFlushSlack = time.Second + sessionLogFlushRecords = 1000 + sessionLogMaxChunkPlaintext = 4 << 20 - activitySpoolCapacity = 5000 - activityTotalCapacity = 200_000 + sessionLogSpoolCapacity = 5000 + sessionLogTotalCapacity = 200_000 - activityPendingChunks = 10 - activityTotalSealedBytes = 64 << 20 + sessionLogPendingChunks = 10 + sessionLogTotalSealedBytes = 64 << 20 - activityIdleClose = 15 * time.Minute + sessionLogIdleClose = 15 * time.Minute - activityPauseBackoff = 15 * time.Minute - activityPutTimeout = 10 * time.Second - activityFinalTimeout = 3 * time.Second - activityCloseTimeout = 5 * time.Second + sessionLogPauseBackoff = 15 * time.Minute + sessionLogPutTimeout = 10 * time.Second + sessionLogFinalTimeout = 3 * time.Second + sessionLogCloseTimeout = 5 * time.Second - activityCeilingReachedName = "AgentVaultActivityCeilingReached" - activityDisabledName = "AgentVaultActivityDisabled" - activityClockSkewName = "AgentVaultActivityClockSkew" + sessionLogCeilingReachedName = "AgentVaultSessionLogCeilingReached" + sessionLogDisabledName = "AgentVaultSessionLogDisabled" + sessionLogClockSkewName = "AgentVaultSessionLogClockSkew" ) -type activityGrant struct { +type sessionLogGrant struct { sessionID string key []byte @@ -42,10 +42,10 @@ type activityGrant struct { } // Counts every grant the proxy is handed, so a refusal can be told apart from a key issued after it. -var activityGrantsIssued atomic.Uint64 +var sessionLogGrantsIssued atomic.Uint64 -func newActivityGrant(sessionID string, key []byte) *activityGrant { - return &activityGrant{sessionID: sessionID, key: key, issued: activityGrantsIssued.Add(1)} +func newSessionLogGrant(sessionID string, key []byte) *sessionLogGrant { + return &sessionLogGrant{sessionID: sessionID, key: key, issued: sessionLogGrantsIssued.Add(1)} } type forgottenSpool struct { @@ -53,21 +53,21 @@ type forgottenSpool struct { dropped uint64 } -type activityShipper interface { - createChunk(ctx context.Context, final bool, sessionID string, req api.CreateAgentVaultActivityChunkRequest) (api.CreateAgentVaultActivityChunkResponse, error) +type sessionLogShipper interface { + createChunk(ctx context.Context, final bool, sessionID string, req api.CreateAgentVaultSessionLogChunkRequest) (api.CreateAgentVaultSessionLogChunkResponse, error) putObject(ctx context.Context, url string, ciphertext []byte) error } -type activityLog struct { +type sessionLogRecorder struct { proxyID string - shipper activityShipper + shipper sessionLogShipper now func() time.Time // close's flushAll can overlap the run loop's; unguarded, one chunk ships twice. flushMu sync.Mutex mu sync.Mutex - spools map[string]*activitySpool + spools map[string]*sessionLogSpool forgotten map[string]forgottenSpool @@ -89,18 +89,18 @@ type activityLog struct { wake chan struct{} } -func newActivityLog(proxyID string, shipper activityShipper) *activityLog { - return &activityLog{ +func newSessionLogRecorder(proxyID string, shipper sessionLogShipper) *sessionLogRecorder { + return &sessionLogRecorder{ proxyID: proxyID, shipper: shipper, now: time.Now, - spools: make(map[string]*activitySpool), + spools: make(map[string]*sessionLogSpool), forgotten: make(map[string]forgottenSpool), wake: make(chan struct{}, 1), } } -func (a *activityLog) record(g *activityGrant, rec activityRecord) { +func (a *sessionLogRecorder) record(g *sessionLogGrant, rec sessionLogRecord) { if a == nil || g == nil { return } @@ -113,7 +113,7 @@ func (a *activityLog) record(g *activityGrant, rec activityRecord) { spool, ok := a.spools[g.sessionID] if !ok { - spool = newActivitySpool(g, a.now()) + spool = newSessionLogSpool(g, a.now()) if prior, ok := a.forgotten[g.sessionID]; ok { spool.nextSeq, spool.ring.dropped = prior.nextSeq, prior.dropped delete(a.forgotten, g.sessionID) @@ -133,7 +133,7 @@ func (a *activityLog) record(g *activityGrant, rec activityRecord) { return } a.switchedOff = false - log.Info().Msg("agent-vault: activity logging is back on, recording again") + log.Info().Msg("agent-vault: session logs are back on, recording again") } if !a.pauseUntil.IsZero() && a.now().Before(a.pauseUntil) { @@ -141,7 +141,7 @@ func (a *activityLog) record(g *activityGrant, rec activityRecord) { return } - if a.total >= activityTotalCapacity { + if a.total >= sessionLogTotalCapacity { spool.ring.dropped++ return } @@ -150,7 +150,7 @@ func (a *activityLog) record(g *activityGrant, rec activityRecord) { a.total++ } - if spool.ring.len() >= activityFlushRecords { + if spool.ring.len() >= sessionLogFlushRecords { select { case a.wake <- struct{}{}: default: @@ -158,11 +158,11 @@ func (a *activityLog) record(g *activityGrant, rec activityRecord) { } } -func (a *activityLog) run(stop <-chan struct{}) { +func (a *sessionLogRecorder) run(stop <-chan struct{}) { if a == nil { return } - ticker := time.NewTicker(activityFlushInterval) + ticker := time.NewTicker(sessionLogFlushInterval) defer ticker.Stop() for { @@ -177,7 +177,7 @@ func (a *activityLog) run(stop <-chan struct{}) { } } -func (a *activityLog) close(ctx context.Context) { +func (a *sessionLogRecorder) close(ctx context.Context) { if a == nil { return } @@ -197,25 +197,25 @@ func (a *activityLog) close(ctx context.Context) { } } if lost > 0 { - log.Warn().Int("records", lost).Msg("agent-vault: activity records were not shipped before shutdown") + log.Warn().Int("records", lost).Msg("agent-vault: session log records were not shipped before shutdown") } } -func (a *activityLog) dueSpools(final bool, now time.Time) []*activitySpool { +func (a *sessionLogRecorder) dueSpools(final bool, now time.Time) []*sessionLogSpool { a.mu.Lock() defer a.mu.Unlock() - due := make([]*activitySpool, 0, len(a.spools)) + due := make([]*sessionLogSpool, 0, len(a.spools)) for id, spool := range a.spools { if spool.ring.len() == 0 && len(spool.pending) == 0 { - if !final && now.Sub(spool.lastRecordAt) > activityIdleClose { + if !final && now.Sub(spool.lastRecordAt) > sessionLogIdleClose { a.forgetSpoolLocked(id, spool) } continue } // The slack absorbs ticker jitter: without it a spool stamped at one tick is a hair short of due at the next. - if final || spool.ring.len() >= activityFlushRecords || len(spool.pending) > 0 || - (spool.ring.len() > 0 && now.Sub(spool.lastFlushAt) >= activityFlushInterval-activityFlushSlack) { + if final || spool.ring.len() >= sessionLogFlushRecords || len(spool.pending) > 0 || + (spool.ring.len() > 0 && now.Sub(spool.lastFlushAt) >= sessionLogFlushInterval-sessionLogFlushSlack) { due = append(due, spool) } } @@ -223,7 +223,7 @@ func (a *activityLog) dueSpools(final bool, now time.Time) []*activitySpool { } // Keeps the drop count too, so a spool forgotten while logging was off still reports what it lost. -func (a *activityLog) forgetSpoolLocked(sessionID string, spool *activitySpool) { +func (a *sessionLogRecorder) forgetSpoolLocked(sessionID string, spool *sessionLogSpool) { if len(a.forgotten) >= maxSessionCacheEntries { a.forgotten = make(map[string]forgottenSpool) } @@ -231,24 +231,24 @@ func (a *activityLog) forgetSpoolLocked(sessionID string, spool *activitySpool) delete(a.spools, sessionID) } -func (a *activityLog) holding() bool { +func (a *sessionLogRecorder) holding() bool { a.mu.Lock() defer a.mu.Unlock() return a.switchedOff || (!a.pauseUntil.IsZero() && a.now().Before(a.pauseUntil)) } -func (a *activityLog) pause() { +func (a *sessionLogRecorder) pause() { a.mu.Lock() defer a.mu.Unlock() - a.pauseUntil = a.now().Add(activityPauseBackoff) + a.pauseUntil = a.now().Add(sessionLogPauseBackoff) } // Drops everything held and counts it. A key issued after the refused request went out means logging may already // be back on, so then the refusal is stale and nothing is dropped. -func (a *activityLog) switchOff(grantsIssuedAtSend uint64) { +func (a *sessionLogRecorder) switchOff(grantsIssuedAtSend uint64) { a.mu.Lock() defer a.mu.Unlock() - if activityGrantsIssued.Load() > grantsIssuedAtSend { + if sessionLogGrantsIssued.Load() > grantsIssuedAtSend { return } @@ -269,13 +269,13 @@ func (a *activityLog) switchOff(grantsIssuedAtSend uint64) { if !a.switchedOff { log.Warn().Int("records", lost). - Msg("agent-vault: activity logging is switched off for this project, dropping what was held until it is back on") + Msg("agent-vault: session logs are off for this project, dropping what was held until they are back on") } a.switchedOff = true a.switchedOffThrough = grantsIssuedAtSend } -func (a *activityLog) flushAll(ctx context.Context, final bool) { +func (a *sessionLogRecorder) flushAll(ctx context.Context, final bool) { if a == nil { return } @@ -295,10 +295,10 @@ func (a *activityLog) flushAll(ctx context.Context, final bool) { } } -func (a *activityLog) sealRing(spool *activitySpool, started time.Time) { +func (a *sessionLogRecorder) sealRing(spool *sessionLogSpool, started time.Time) { for { a.mu.Lock() - records := spool.ring.drain(activityFlushRecords) + records := spool.ring.drain(sessionLogFlushRecords) if len(records) == 0 { spool.lastFlushAt = started a.mu.Unlock() @@ -309,7 +309,7 @@ func (a *activityLog) sealRing(spool *activitySpool, started time.Time) { now := a.now() a.mu.Unlock() - groups, err := packActivityRecords(records) + groups, err := packSessionLogRecords(records) if err != nil { a.dropUnsealed(spool, len(records), dropped, err) continue @@ -337,19 +337,19 @@ func (a *activityLog) sealRing(spool *activitySpool, started time.Time) { } } -func (a *activityLog) dropUnsealed(spool *activitySpool, records int, dropped uint64, err error) { +func (a *sessionLogRecorder) dropUnsealed(spool *sessionLogSpool, records int, dropped uint64, err error) { a.mu.Lock() spool.ring.dropped += dropped + uint64(records) a.mu.Unlock() log.Error().Err(err).Str("sessionId", spool.sessionID).Int("records", records). - Msg("agent-vault: could not seal an activity chunk, dropping those records") + Msg("agent-vault: could not seal a session log chunk, dropping those records") } -func (a *activityLog) enforcePendingCapsLocked(spool *activitySpool) { - for len(spool.pending) > activityPendingChunks { +func (a *sessionLogRecorder) enforcePendingCapsLocked(spool *sessionLogSpool) { + for len(spool.pending) > sessionLogPendingChunks { a.evictOldestLocked(spool) } - for a.sealedBytes > activityTotalSealedBytes { + for a.sealedBytes > sessionLogTotalSealedBytes { victim := a.oldestPendingLocked() if victim == nil { return @@ -358,8 +358,8 @@ func (a *activityLog) enforcePendingCapsLocked(spool *activitySpool) { } } -func (a *activityLog) oldestPendingLocked() *activitySpool { - var oldest *activitySpool +func (a *sessionLogRecorder) oldestPendingLocked() *sessionLogSpool { + var oldest *sessionLogSpool for _, spool := range a.spools { if len(spool.pending) == 0 { continue @@ -371,7 +371,7 @@ func (a *activityLog) oldestPendingLocked() *activitySpool { return oldest } -func (a *activityLog) evictOldestLocked(spool *activitySpool) { +func (a *sessionLogRecorder) evictOldestLocked(spool *sessionLogSpool) { oldest := spool.pending[0] spool.pending = spool.pending[1:] a.sealedBytes -= len(oldest.ciphertext) @@ -380,10 +380,10 @@ func (a *activityLog) evictOldestLocked(spool *activitySpool) { Str("sessionId", spool.sessionID). Str("chunkId", oldest.meta.ChunkID). Int("records", oldest.meta.RecordCount). - Msg("agent-vault: dropped an unshipped activity chunk, the buffer is full") + Msg("agent-vault: dropped an unshipped session log chunk, the buffer is full") } -func (a *activityLog) flushSpool(ctx context.Context, spool *activitySpool, final bool, started time.Time) { +func (a *sessionLogRecorder) flushSpool(ctx context.Context, spool *sessionLogSpool, final bool, started time.Time) { a.sealRing(spool, started) if a.holding() { return @@ -411,13 +411,13 @@ func (a *activityLog) flushSpool(ctx context.Context, spool *activitySpool, fina } } -func (a *activityLog) shipChunk(ctx context.Context, spool *activitySpool, chunk *sealedChunk, final bool) bool { +func (a *sessionLogRecorder) shipChunk(ctx context.Context, spool *sessionLogSpool, chunk *sealedChunk, final bool) bool { if chunk.uploadURL == "" || a.now().Add(10*time.Second).After(chunk.urlExpires) { // Past the shutdown budget, a new row could only be written for an upload that can no longer happen. if ctx.Err() != nil { return false } - grantsIssuedAtSend := activityGrantsIssued.Load() + grantsIssuedAtSend := sessionLogGrantsIssued.Load() res, err := a.shipper.createChunk(ctx, final, spool.sessionID, chunk.meta) if err != nil { return a.handleCreateFailure(spool, chunk, err, grantsIssuedAtSend) @@ -433,7 +433,7 @@ func (a *activityLog) shipChunk(ctx context.Context, spool *activitySpool, chunk putCtx := ctx if !final { var cancel context.CancelFunc - putCtx, cancel = context.WithTimeout(ctx, activityPutTimeout) + putCtx, cancel = context.WithTimeout(ctx, sessionLogPutTimeout) defer cancel() } @@ -443,17 +443,17 @@ func (a *activityLog) shipChunk(ctx context.Context, spool *activitySpool, chunk a.s3Down = true a.mu.Unlock() log.Warn().Err(err).Str("sessionId", spool.sessionID).Str("chunkId", chunk.meta.ChunkID). - Msg("agent-vault: could not upload an activity chunk, will retry") + Msg("agent-vault: could not upload a session log chunk, will retry") return false } return true } -func (a *activityLog) handleCreateFailure(spool *activitySpool, chunk *sealedChunk, err error, grantsIssuedAtSend uint64) bool { +func (a *sessionLogRecorder) handleCreateFailure(spool *sessionLogSpool, chunk *sealedChunk, err error, grantsIssuedAtSend uint64) bool { switch { case isProxyTokenRejected(err): - log.Warn().Err(err).Msg("agent-vault: Infisical rejected this proxy's token, holding activity") + log.Warn().Err(err).Msg("agent-vault: Infisical rejected this proxy's token, holding session logs") return false case isSessionGone(err): @@ -468,33 +468,33 @@ func (a *activityLog) handleCreateFailure(spool *activitySpool, chunk *sealedChu delete(a.forgotten, spool.sessionID) a.mu.Unlock() log.Warn().Err(err).Str("sessionId", spool.sessionID).Int("records", lost). - Msg("agent-vault: Infisical no longer accepts activity for this session, dropping what was held") + Msg("agent-vault: Infisical no longer accepts session logs for this session, dropping what was held") return false - case isActivityErrorNamed(err, activityCeilingReachedName): + case isSessionLogErrorNamed(err, sessionLogCeilingReachedName): a.pause() - log.Error().Err(err).Msg("agent-vault: activity logging has reached its limit for this organization, retrying in 15m") + log.Error().Err(err).Msg("agent-vault: session logs have reached their limit for this organization, retrying in 15m") return false - case isActivityErrorNamed(err, activityDisabledName): + case isSessionLogErrorNamed(err, sessionLogDisabledName): a.switchOff(grantsIssuedAtSend) return false - case isActivityErrorNamed(err, activityClockSkewName): + case isSessionLogErrorNamed(err, sessionLogClockSkewName): a.dropRefused(spool, chunk) a.mu.Lock() reported := a.clockSkewReported a.clockSkewReported = true a.mu.Unlock() if !reported { - log.Error().Err(err).Msg("agent-vault: activity is being refused because this machine's clock is wrong; fix the clock to resume recording") + log.Error().Err(err).Msg("agent-vault: session logs are being refused because this machine's clock is wrong; fix the clock to resume recording") } return false case isPoisonChunk(err): a.dropRefused(spool, chunk) log.Error().Err(err).Str("chunkId", chunk.meta.ChunkID).Int("records", chunk.meta.RecordCount). - Msg("agent-vault: Infisical rejected an activity chunk as malformed, dropping it") + Msg("agent-vault: Infisical rejected a session log chunk as malformed, dropping it") return false default: @@ -502,12 +502,12 @@ func (a *activityLog) handleCreateFailure(spool *activitySpool, chunk *sealedChu a.infisicalDown = true a.mu.Unlock() log.Warn().Err(err).Str("sessionId", spool.sessionID). - Msg("agent-vault: could not record activity, will retry") + Msg("agent-vault: could not record session logs, will retry") return false } } -func (a *activityLog) dropRefused(spool *activitySpool, chunk *sealedChunk) { +func (a *sessionLogRecorder) dropRefused(spool *sessionLogSpool, chunk *sealedChunk) { a.mu.Lock() defer a.mu.Unlock() if len(spool.pending) > 0 && spool.pending[0] == chunk { diff --git a/packages/agentvault/activity_crypto.go b/packages/agentvault/session_log_crypto.go similarity index 55% rename from packages/agentvault/activity_crypto.go rename to packages/agentvault/session_log_crypto.go index bb6451cb4..b2910e43d 100644 --- a/packages/agentvault/activity_crypto.go +++ b/packages/agentvault/session_log_crypto.go @@ -13,31 +13,31 @@ import ( "github.com/oklog/ulid" ) -const activityAADVersion = "v1" +const sessionLogAADVersion = "v1" -const activityIVBytes = 12 +const sessionLogIVBytes = 12 -// Must byte-match frontend activityDecrypt.ts and the vector pinned in agent-vault-activity-crypto.test.ts. -func buildActivityAAD(sessionID, chunkID string) []byte { - sum := sha256.Sum256([]byte(fmt.Sprintf("%s|%s|%s", sessionID, chunkID, activityAADVersion))) +// Must byte-match frontend sessionLogDecrypt.ts and the vector pinned in agent-vault-session-log-crypto.test.ts. +func buildSessionLogAAD(sessionID, chunkID string) []byte { + sum := sha256.Sum256([]byte(fmt.Sprintf("%s|%s|%s", sessionID, chunkID, sessionLogAADVersion))) return sum[:] } -func sealActivity(key, plaintext, aad []byte) (ciphertext []byte, iv []byte, err error) { - return sealActivityWithRand(rand.Reader, key, plaintext, aad) +func sealSessionLog(key, plaintext, aad []byte) (ciphertext []byte, iv []byte, err error) { + return sealSessionLogWithRand(rand.Reader, key, plaintext, aad) } -func sealActivityWithRand(random io.Reader, key, plaintext, aad []byte) (ciphertext []byte, iv []byte, err error) { +func sealSessionLogWithRand(random io.Reader, key, plaintext, aad []byte) (ciphertext []byte, iv []byte, err error) { block, err := aes.NewCipher(key) if err != nil { - return nil, nil, fmt.Errorf("agent-vault: activity key is not a valid AES key: %w", err) + return nil, nil, fmt.Errorf("agent-vault: session log key is not a valid AES key: %w", err) } gcm, err := cipher.NewGCM(block) if err != nil { return nil, nil, fmt.Errorf("agent-vault: could not build GCM: %w", err) } - iv = make([]byte, activityIVBytes) + iv = make([]byte, sessionLogIVBytes) if _, err = io.ReadFull(random, iv); err != nil { return nil, nil, fmt.Errorf("agent-vault: could not read a nonce: %w", err) } @@ -45,18 +45,18 @@ func sealActivityWithRand(random io.Reader, key, plaintext, aad []byte) (ciphert return gcm.Seal(nil, iv, plaintext, aad), iv, nil } -func encodeActivityIV(iv []byte) string { +func encodeSessionLogIV(iv []byte) string { return base64.RawStdEncoding.EncodeToString(iv) } // The browser checks the downloaded object against this before decrypting, so an edited object reads as // changed rather than as a decryption failure. -func activityCiphertextSHA256(ciphertext []byte) string { +func sessionLogCiphertextSHA256(ciphertext []byte) string { sum := sha256.Sum256(ciphertext) return base64.RawStdEncoding.EncodeToString(sum[:]) } -func newActivityChunkID(now time.Time) string { +func newSessionLogChunkID(now time.Time) string { return ulid.MustNew(ulid.Timestamp(now), newULIDEntropy()).String() } diff --git a/packages/agentvault/activity_crypto_test.go b/packages/agentvault/session_log_crypto_test.go similarity index 76% rename from packages/agentvault/activity_crypto_test.go rename to packages/agentvault/session_log_crypto_test.go index a49ba8f16..3dda3f1de 100644 --- a/packages/agentvault/activity_crypto_test.go +++ b/packages/agentvault/session_log_crypto_test.go @@ -27,9 +27,9 @@ var vectorContext = struct{ sessionID, chunkID string }{ chunkID: "01K5ABCDEFGHJKMNPQRSTVWXYZ", } -func vectorRecords() []activityRecord { +func vectorRecords() []sessionLogRecord { service, bundle := "github", "code-review" - return []activityRecord{{ + return []sessionLogRecord{{ Ts: "2026-09-16T10:31:04.221Z", Seq: 1, ProxyID: "proxy-1", @@ -53,8 +53,8 @@ func mustHex(t *testing.T, s string) []byte { return b } -func TestActivityAADMatchesTheBackendVector(t *testing.T) { - got := buildActivityAAD(vectorContext.sessionID, vectorContext.chunkID) +func TestSessionLogAADMatchesTheBackendVector(t *testing.T) { + got := buildSessionLogAAD(vectorContext.sessionID, vectorContext.chunkID) if hex.EncodeToString(got) != vectorAADHex { t.Fatalf("AAD is %s, the backend and the browser build %s", hex.EncodeToString(got), vectorAADHex) } @@ -67,14 +67,14 @@ func TestSealMatchesNodeVector(t *testing.T) { } iv := mustHex(t, vectorIVHex) - aad := buildActivityAAD(vectorContext.sessionID, vectorContext.chunkID) - ciphertext, gotIV, err := sealActivityWithRand(bytes.NewReader(iv), mustHex(t, vectorKeyHex), plaintext, aad) + aad := buildSessionLogAAD(vectorContext.sessionID, vectorContext.chunkID) + ciphertext, gotIV, err := sealSessionLogWithRand(bytes.NewReader(iv), mustHex(t, vectorKeyHex), plaintext, aad) if err != nil { t.Fatal(err) } - if encodeActivityIV(gotIV) != vectorIVBase64 { - t.Fatalf("IV encodes as %q, the backend expects %q", encodeActivityIV(gotIV), vectorIVBase64) + if encodeSessionLogIV(gotIV) != vectorIVBase64 { + t.Fatalf("IV encodes as %q, the backend expects %q", encodeSessionLogIV(gotIV), vectorIVBase64) } if base64.StdEncoding.EncodeToString(ciphertext) != vectorCiphertext { t.Fatal("the sealed bytes differ from the vector Infisical and the browser are checked against") @@ -83,7 +83,7 @@ func TestSealMatchesNodeVector(t *testing.T) { func TestASealedChunkCarriesTheDigestOfExactlyWhatIsUploaded(t *testing.T) { key := make([]byte, 32) - spool := newActivitySpool(newActivityGrant("sess-1", key), time.Now()) + spool := newSessionLogSpool(newSessionLogGrant("sess-1", key), time.Now()) chunk, err := spool.sealSlice(vectorRecords(), []byte("[]"), 0, time.Now()) if err != nil { t.Fatal(err) @@ -102,17 +102,17 @@ func TestASealedChunkCarriesTheDigestOfExactlyWhatIsUploaded(t *testing.T) { } block, _ := aes.NewCipher(key) gcm, _ := cipher.NewGCM(block) - if _, err := gcm.Open(nil, iv, chunk.ciphertext, buildActivityAAD("sess-1", chunk.meta.ChunkID)); err != nil { + if _, err := gcm.Open(nil, iv, chunk.ciphertext, buildSessionLogAAD("sess-1", chunk.meta.ChunkID)); err != nil { t.Fatalf("the chunk does not open under its session and chunk ID: %v", err) } } func TestSealedChunkOpensWithTheTagAppended(t *testing.T) { key := mustHex(t, vectorKeyHex) - aad := buildActivityAAD(vectorContext.sessionID, vectorContext.chunkID) + aad := buildSessionLogAAD(vectorContext.sessionID, vectorContext.chunkID) plaintext, _ := json.Marshal(vectorRecords()) - ciphertext, iv, err := sealActivity(key, plaintext, aad) + ciphertext, iv, err := sealSessionLog(key, plaintext, aad) if err != nil { t.Fatal(err) } @@ -134,8 +134,8 @@ func TestSealedChunkOpensWithTheTagAppended(t *testing.T) { func TestAChunkCannotBeReplayedElsewhere(t *testing.T) { key := mustHex(t, vectorKeyHex) plaintext, _ := json.Marshal(vectorRecords()) - aad := buildActivityAAD(vectorContext.sessionID, vectorContext.chunkID) - ciphertext, iv, err := sealActivity(key, plaintext, aad) + aad := buildSessionLogAAD(vectorContext.sessionID, vectorContext.chunkID) + ciphertext, iv, err := sealSessionLog(key, plaintext, aad) if err != nil { t.Fatal(err) } @@ -147,8 +147,8 @@ func TestAChunkCannotBeReplayedElsewhere(t *testing.T) { name string aad []byte }{ - {"another session", buildActivityAAD("other", vectorContext.chunkID)}, - {"another chunk", buildActivityAAD(vectorContext.sessionID, "other")}, + {"another session", buildSessionLogAAD("other", vectorContext.chunkID)}, + {"another chunk", buildSessionLogAAD(vectorContext.sessionID, "other")}, } { if _, err := gcm.Open(nil, iv, ciphertext, wrong.aad); err == nil { t.Fatalf("a chunk opened under %s", wrong.name) @@ -160,12 +160,12 @@ func TestIVsDoNotRepeat(t *testing.T) { key := mustHex(t, vectorKeyHex) seen := make(map[string]bool, 256) for i := 0; i < 256; i++ { - _, iv, err := sealActivity(key, []byte("[]"), nil) + _, iv, err := sealSessionLog(key, []byte("[]"), nil) if err != nil { t.Fatal(err) } - if len(iv) != activityIVBytes { - t.Fatalf("IV is %d bytes, the contract is %d", len(iv), activityIVBytes) + if len(iv) != sessionLogIVBytes { + t.Fatalf("IV is %d bytes, the contract is %d", len(iv), sessionLogIVBytes) } if seen[string(iv)] { t.Fatal("an IV repeated, which would void GCM's guarantees for this key") @@ -175,8 +175,8 @@ func TestIVsDoNotRepeat(t *testing.T) { } func TestChunkIDsAreULIDsThatSortByTime(t *testing.T) { - earlier := newActivityChunkID(time.Date(2026, 9, 16, 10, 0, 0, 0, time.UTC)) - later := newActivityChunkID(time.Date(2026, 9, 16, 11, 0, 0, 0, time.UTC)) + earlier := newSessionLogChunkID(time.Date(2026, 9, 16, 10, 0, 0, 0, time.UTC)) + later := newSessionLogChunkID(time.Date(2026, 9, 16, 11, 0, 0, 0, time.UTC)) if len(earlier) != 26 { t.Fatalf("a chunk id is %d characters, the server's column is 26", len(earlier)) @@ -193,7 +193,7 @@ func TestChunkIDsAreUniqueWithinAMillisecond(t *testing.T) { now := time.Now() seen := make(map[string]bool, 1000) for i := 0; i < 1000; i++ { - id := newActivityChunkID(now) + id := newSessionLogChunkID(now) if seen[id] { t.Fatal("a chunk id repeated, which would collide with the server's unique index") } diff --git a/packages/agentvault/activity_resolve_test.go b/packages/agentvault/session_log_resolve_test.go similarity index 61% rename from packages/agentvault/activity_resolve_test.go rename to packages/agentvault/session_log_resolve_test.go index 36140b8c4..d0ff4e03a 100644 --- a/packages/agentvault/activity_resolve_test.go +++ b/packages/agentvault/session_log_resolve_test.go @@ -7,12 +7,12 @@ import ( "github.com/Infisical/infisical-merge/packages/api" ) -func enabledGrantWire(key string) api.AgentVaultActivityGrant { - return api.AgentVaultActivityGrant{Enabled: true, SessionKey: key} +func enabledGrantWire(key string) api.AgentVaultSessionLogGrant { + return api.AgentVaultSessionLogGrant{Enabled: true, SessionKey: key} } func aKey(b byte) []byte { - key := make([]byte, activityKeyBytes) + key := make([]byte, sessionLogKeyBytes) for i := range key { key[i] = b } @@ -21,10 +21,10 @@ func aKey(b byte) []byte { func TestTheFirstResolveTakesTheKeyOffTheWire(t *testing.T) { want := aKey(7) - got := toActivityGrant("s1", enabledGrantWire(base64.StdEncoding.EncodeToString(want)), nil) + got := toSessionLogGrant("s1", enabledGrantWire(base64.StdEncoding.EncodeToString(want)), nil) if got == nil { - t.Fatal("activity was enabled but no grant was built") + t.Fatal("session logs were on but no grant was built") } if got.sessionID != "s1" { t.Fatalf("grant names session %q", got.sessionID) @@ -35,9 +35,9 @@ func TestTheFirstResolveTakesTheKeyOffTheWire(t *testing.T) { } func TestACachedKeySurvivesAResolveThatOmitsIt(t *testing.T) { - held := &activityGrant{sessionID: "s1", key: aKey(9)} + held := &sessionLogGrant{sessionID: "s1", key: aKey(9)} - got := toActivityGrant("s1", enabledGrantWire(""), held) + got := toSessionLogGrant("s1", enabledGrantWire(""), held) if got == nil { t.Fatal("the grant was cleared when the response carried no key; logging would stop after one poll") @@ -48,15 +48,15 @@ func TestACachedKeySurvivesAResolveThatOmitsIt(t *testing.T) { } func TestNoKeyAndNoCachedCopyMeansNoRecording(t *testing.T) { - if got := toActivityGrant("s1", enabledGrantWire(""), nil); got != nil { + if got := toSessionLogGrant("s1", enabledGrantWire(""), nil); got != nil { t.Fatal("a grant was built with no key at all") } } -func TestActivityBeingOffClearsAnyCachedGrant(t *testing.T) { - held := &activityGrant{sessionID: "s1", key: aKey(9)} +func TestSessionLogBeingOffClearsAnyCachedGrant(t *testing.T) { + held := &sessionLogGrant{sessionID: "s1", key: aKey(9)} - if got := toActivityGrant("s1", api.AgentVaultActivityGrant{Enabled: false}, held); got != nil { + if got := toSessionLogGrant("s1", api.AgentVaultSessionLogGrant{Enabled: false}, held); got != nil { t.Fatal("the proxy kept recording after logging was switched off") } } @@ -70,7 +70,7 @@ func TestAnUnusableKeyIsRefusedRatherThanUsed(t *testing.T) { {"too short", base64.StdEncoding.EncodeToString(make([]byte, 16))}, {"too long", base64.StdEncoding.EncodeToString(make([]byte, 64))}, } { - if got := toActivityGrant("s1", enabledGrantWire(wire.key), nil); got != nil { + if got := toSessionLogGrant("s1", enabledGrantWire(wire.key), nil); got != nil { t.Fatalf("a key that is %s was accepted", wire.name) } } diff --git a/packages/agentvault/activity_ship.go b/packages/agentvault/session_log_ship.go similarity index 76% rename from packages/agentvault/activity_ship.go rename to packages/agentvault/session_log_ship.go index 8e3917e94..37627613e 100644 --- a/packages/agentvault/activity_ship.go +++ b/packages/agentvault/session_log_ship.go @@ -14,7 +14,7 @@ import ( "github.com/go-resty/resty/v2" ) -func isActivityErrorNamed(err error, name string) bool { +func isSessionLogErrorNamed(err error, name string) bool { var apiErr *api.APIError return errors.As(err, &apiErr) && apiErr.Name == name } @@ -30,13 +30,13 @@ func isPoisonChunk(err error) bool { return apiErr.StatusCode >= 400 && apiErr.StatusCode < 500 } -type activityShipperClient struct { +type sessionLogShipperClient struct { steady *resty.Client final *resty.Client put *http.Client } -func newActivityShipper(proxyToken func() string) (*activityShipperClient, error) { +func newSessionLogShipper(proxyToken func() string) (*sessionLogShipperClient, error) { steady, err := util.GetRestyClientWithPolicy(util.RetryPolicy{}) if err != nil { return nil, err @@ -49,21 +49,21 @@ func newActivityShipper(proxyToken func() string) (*activityShipperClient, error if err != nil { return nil, err } - final.SetAuthToken(proxyToken()).SetTimeout(activityFinalTimeout) + final.SetAuthToken(proxyToken()).SetTimeout(sessionLogFinalTimeout) - return &activityShipperClient{ + return &sessionLogShipperClient{ steady: steady, final: final, // Not resty: this goes to the customer's bucket and must never carry the Infisical auth token. put: &http.Client{ - Timeout: activityPutTimeout, + Timeout: sessionLogPutTimeout, Transport: http.DefaultTransport.(*http.Transport).Clone(), CheckRedirect: func(*http.Request, []*http.Request) error { return http.ErrUseLastResponse }, }, }, nil } -var errInsecureUploadURL = errors.New("agent-vault: refusing to upload activity to a link that is not https") +var errInsecureUploadURL = errors.New("agent-vault: refusing to upload session logs to a link that is not https") func scrubURLError(err error) error { var urlErr *url.Error @@ -73,15 +73,15 @@ func scrubURLError(err error) error { return err } -func (c *activityShipperClient) createChunk(ctx context.Context, final bool, sessionID string, req api.CreateAgentVaultActivityChunkRequest) (api.CreateAgentVaultActivityChunkResponse, error) { +func (c *sessionLogShipperClient) createChunk(ctx context.Context, final bool, sessionID string, req api.CreateAgentVaultSessionLogChunkRequest) (api.CreateAgentVaultSessionLogChunkResponse, error) { client := c.steady if final { client = c.final } - return api.CallCreateAgentVaultActivityChunk(ctx, client, sessionID, req) + return api.CallCreateAgentVaultSessionLogChunk(ctx, client, sessionID, req) } -func (c *activityShipperClient) putObject(ctx context.Context, uploadURL string, ciphertext []byte) error { +func (c *sessionLogShipperClient) putObject(ctx context.Context, uploadURL string, ciphertext []byte) error { target, err := url.Parse(uploadURL) if err != nil { return scrubURLError(err) diff --git a/packages/agentvault/activity_ship_test.go b/packages/agentvault/session_log_ship_test.go similarity index 90% rename from packages/agentvault/activity_ship_test.go rename to packages/agentvault/session_log_ship_test.go index 9c22740ee..cda76c735 100644 --- a/packages/agentvault/activity_ship_test.go +++ b/packages/agentvault/session_log_ship_test.go @@ -54,14 +54,14 @@ func TestTheChunkPostCarriesTheProxyTokenAndTheBucketPutDoesNot(t *testing.T) { config.INFISICAL_URL = infisical.URL + "/api" defer func() { config.INFISICAL_URL = old }() - shipper, err := newActivityShipper(func() string { return "proxy-token" }) + shipper, err := newSessionLogShipper(func() string { return "proxy-token" }) if err != nil { t.Fatal(err) } shipper.put.Transport = bucket.Client().Transport ciphertext := []byte("sealed-bytes") - res, err := shipper.createChunk(context.Background(), false, "sess-1", api.CreateAgentVaultActivityChunkRequest{ + res, err := shipper.createChunk(context.Background(), false, "sess-1", api.CreateAgentVaultSessionLogChunkRequest{ ChunkID: "01K5ABCDEFGHJKMNPQRSTVWXYZ", RecordCount: 1, CiphertextBytes: len(ciphertext), @@ -79,7 +79,7 @@ func TestTheChunkPostCarriesTheProxyTokenAndTheBucketPutDoesNot(t *testing.T) { if postAuth != "Bearer proxy-token" { t.Fatalf("Infisical saw Authorization %q", postAuth) } - if postedPath != "/api/v1/agent-vault/proxy/sessions/sess-1/activity/chunks" { + if postedPath != "/api/v1/agent-vault/proxy/sessions/sess-1/logs/chunks" { t.Fatalf("posted to %q", postedPath) } if putAuth != "" { @@ -106,7 +106,7 @@ func TestABucketRefusalIsAnErrorThatNamesNoURL(t *testing.T) { })) defer bucket.Close() - shipper, err := newActivityShipper(func() string { return "proxy-token" }) + shipper, err := newSessionLogShipper(func() string { return "proxy-token" }) if err != nil { t.Fatal(err) } @@ -129,7 +129,7 @@ func TestAnUploadLinkThatIsNotHttpsIsRefused(t *testing.T) { })) defer bucket.Close() - shipper, err := newActivityShipper(func() string { return "proxy-token" }) + shipper, err := newSessionLogShipper(func() string { return "proxy-token" }) if err != nil { t.Fatal(err) } @@ -154,7 +154,7 @@ func TestARedirectFromTheBucketIsNotFollowed(t *testing.T) { })) defer bucket.Close() - shipper, err := newActivityShipper(func() string { return "proxy-token" }) + shipper, err := newSessionLogShipper(func() string { return "proxy-token" }) if err != nil { t.Fatal(err) } @@ -175,7 +175,7 @@ func TestAnUnreachableBucketIsAnErrorThatNamesNoSignature(t *testing.T) { target := bucket.URL + "/object?X-Amz-Signature=secret" bucket.Close() - shipper, err := newActivityShipper(func() string { return "proxy-token" }) + shipper, err := newSessionLogShipper(func() string { return "proxy-token" }) if err != nil { t.Fatal(err) } @@ -195,7 +195,7 @@ func TestAChunkAlreadyStoredCountsAsUploaded(t *testing.T) { })) defer bucket.Close() - shipper, err := newActivityShipper(func() string { return "proxy-token" }) + shipper, err := newSessionLogShipper(func() string { return "proxy-token" }) if err != nil { t.Fatal(err) } diff --git a/packages/agentvault/activity_spool.go b/packages/agentvault/session_log_spool.go similarity index 59% rename from packages/agentvault/activity_spool.go rename to packages/agentvault/session_log_spool.go index d92101a0c..1427ea0bb 100644 --- a/packages/agentvault/activity_spool.go +++ b/packages/agentvault/session_log_spool.go @@ -8,7 +8,7 @@ import ( ) // Never add headers, bodies or the query string: they can carry the injected credential. -type activityRecord struct { +type sessionLogRecord struct { Ts string `json:"ts"` Seq uint64 `json:"seq"` ProxyID string `json:"proxyId"` @@ -22,23 +22,23 @@ type activityRecord struct { AccessBundle *string `json:"accessBundle"` } -const activityRingInitialSize = 64 +const sessionLogRingInitialSize = 64 -type activityRing struct { - buf []activityRecord +type sessionLogRing struct { + buf []sessionLogRecord capacity int head int n int dropped uint64 } -func newActivityRing(capacity int) activityRing { - return activityRing{capacity: capacity} +func newSessionLogRing(capacity int) sessionLogRing { + return sessionLogRing{capacity: capacity} } -func (r *activityRing) len() int { return r.n } +func (r *sessionLogRing) len() int { return r.n } -func (r *activityRing) push(rec activityRecord) (evicted bool) { +func (r *sessionLogRing) push(rec sessionLogRecord) (evicted bool) { if r.capacity == 0 { r.dropped++ return true @@ -57,10 +57,10 @@ func (r *activityRing) push(rec activityRecord) (evicted bool) { return false } -func (r *activityRing) grow() { - size := max(activityRingInitialSize, 2*len(r.buf)) +func (r *sessionLogRing) grow() { + size := max(sessionLogRingInitialSize, 2*len(r.buf)) size = min(size, r.capacity) - next := make([]activityRecord, size) + next := make([]sessionLogRecord, size) for i := 0; i < r.n; i++ { next[i] = r.buf[(r.head+i)%len(r.buf)] } @@ -68,14 +68,14 @@ func (r *activityRing) grow() { r.head = 0 } -func (r *activityRing) drain(max int) []activityRecord { +func (r *sessionLogRing) drain(max int) []sessionLogRecord { if r.n == 0 || max <= 0 { return nil } if max > r.n { max = r.n } - out := make([]activityRecord, max) + out := make([]sessionLogRecord, max) for i := 0; i < max; i++ { out[i] = r.buf[(r.head+i)%len(r.buf)] } @@ -88,14 +88,14 @@ func (r *activityRing) drain(max int) []activityRecord { return out } -func (r *activityRing) takeDropped() uint64 { +func (r *sessionLogRing) takeDropped() uint64 { dropped := r.dropped r.dropped = 0 return dropped } type sealedChunk struct { - meta api.CreateAgentVaultActivityChunkRequest + meta api.CreateAgentVaultSessionLogChunkRequest ciphertext []byte uploadURL string urlExpires time.Time @@ -112,11 +112,11 @@ func (c *sealedChunk) lostCount() uint64 { return c.meta.DroppedCount + uint64(c.meta.RecordCount) } -type activitySpool struct { +type sessionLogSpool struct { sessionID string key []byte - ring activityRing + ring sessionLogRing nextSeq uint64 pending []*sealedChunk @@ -125,31 +125,31 @@ type activitySpool struct { lastFlushAt time.Time } -func newActivitySpool(g *activityGrant, now time.Time) *activitySpool { - return &activitySpool{ +func newSessionLogSpool(g *sessionLogGrant, now time.Time) *sessionLogSpool { + return &sessionLogSpool{ sessionID: g.sessionID, key: g.key, - ring: newActivityRing(activitySpoolCapacity), + ring: newSessionLogRing(sessionLogSpoolCapacity), lastRecordAt: now, lastFlushAt: now, } } -type activityGroup struct { - records []activityRecord +type sessionLogGroup struct { + records []sessionLogRecord plaintext []byte } -func packActivityRecords(records []activityRecord) ([]activityGroup, error) { +func packSessionLogRecords(records []sessionLogRecord) ([]sessionLogGroup, error) { whole, err := json.Marshal(records) if err != nil { return nil, err } - if len(whole) <= activityMaxChunkPlaintext { - return []activityGroup{{records: records, plaintext: whole}}, nil + if len(whole) <= sessionLogMaxChunkPlaintext { + return []sessionLogGroup{{records: records, plaintext: whole}}, nil } - var groups []activityGroup + var groups []sessionLogGroup var buf []byte start := 0 for i, rec := range records { @@ -157,8 +157,8 @@ func packActivityRecords(records []activityRecord) ([]activityGroup, error) { if err != nil { return nil, err } - if len(buf) > 0 && len(buf)+1+len(part)+1 > activityMaxChunkPlaintext { - groups = append(groups, activityGroup{records: records[start:i], plaintext: append(buf, ']')}) + if len(buf) > 0 && len(buf)+1+len(part)+1 > sessionLogMaxChunkPlaintext { + groups = append(groups, sessionLogGroup{records: records[start:i], plaintext: append(buf, ']')}) buf, start = nil, i } if len(buf) == 0 { @@ -168,21 +168,21 @@ func packActivityRecords(records []activityRecord) ([]activityGroup, error) { } buf = append(buf, part...) } - groups = append(groups, activityGroup{records: records[start:], plaintext: append(buf, ']')}) + groups = append(groups, sessionLogGroup{records: records[start:], plaintext: append(buf, ']')}) return groups, nil } -func (s *activitySpool) sealSlice(records []activityRecord, plaintext []byte, dropped uint64, now time.Time) (*sealedChunk, error) { - chunkID := newActivityChunkID(now) - aad := buildActivityAAD(s.sessionID, chunkID) - ciphertext, iv, err := sealActivity(s.key, plaintext, aad) +func (s *sessionLogSpool) sealSlice(records []sessionLogRecord, plaintext []byte, dropped uint64, now time.Time) (*sealedChunk, error) { + chunkID := newSessionLogChunkID(now) + aad := buildSessionLogAAD(s.sessionID, chunkID) + ciphertext, iv, err := sealSessionLog(s.key, plaintext, aad) if err != nil { return nil, err } first, last := records[0], records[len(records)-1] return &sealedChunk{ - meta: api.CreateAgentVaultActivityChunkRequest{ + meta: api.CreateAgentVaultSessionLogChunkRequest{ ChunkID: chunkID, StartedAt: first.Ts, EndedAt: last.Ts, @@ -191,8 +191,8 @@ func (s *activitySpool) sealSlice(records []activityRecord, plaintext []byte, dr RecordCount: len(records), DroppedCount: dropped, CiphertextBytes: len(ciphertext), - IV: encodeActivityIV(iv), - CiphertextSha256: activityCiphertextSHA256(ciphertext), + IV: encodeSessionLogIV(iv), + CiphertextSha256: sessionLogCiphertextSHA256(ciphertext), }, ciphertext: ciphertext, }, nil diff --git a/packages/agentvault/activity_test.go b/packages/agentvault/session_log_test.go similarity index 86% rename from packages/agentvault/activity_test.go rename to packages/agentvault/session_log_test.go index 7b42f4134..e0ab61de7 100644 --- a/packages/agentvault/activity_test.go +++ b/packages/agentvault/session_log_test.go @@ -43,7 +43,7 @@ type fakeShipper struct { nextURL int } -func (f *fakeShipper) createChunk(_ context.Context, final bool, sessionID string, req api.CreateAgentVaultActivityChunkRequest) (api.CreateAgentVaultActivityChunkResponse, error) { +func (f *fakeShipper) createChunk(_ context.Context, final bool, sessionID string, req api.CreateAgentVaultSessionLogChunkRequest) (api.CreateAgentVaultSessionLogChunkResponse, error) { f.mu.Lock() defer f.mu.Unlock() @@ -55,14 +55,14 @@ func (f *fakeShipper) createChunk(_ context.Context, final bool, sessionID strin f.postResults = f.postResults[1:] } if result.err != nil { - return api.CreateAgentVaultActivityChunkResponse{}, result.err + return api.CreateAgentVaultSessionLogChunkResponse{}, result.err } url := result.url if url == "" { f.nextURL++ url = fmt.Sprintf("https://bucket.example/put/%d", f.nextURL) } - return api.CreateAgentVaultActivityChunkResponse{ChunkID: req.ChunkID, UploadURL: url, ExpiresInSeconds: 300}, nil + return api.CreateAgentVaultSessionLogChunkResponse{ChunkID: req.ChunkID, UploadURL: url, ExpiresInSeconds: 300}, nil } func (f *fakeShipper) putObject(_ context.Context, url string, ciphertext []byte) error { @@ -114,15 +114,15 @@ func (f *fakeShipper) puts() []shipperCall { } func apiErr(status int, name string) error { - return &api.APIError{StatusCode: status, Name: name, Operation: "CallCreateAgentVaultActivityChunk"} + return &api.APIError{StatusCode: status, Name: name, Operation: "CallCreateAgentVaultSessionLogChunk"} } -func testGrant(sessionID string) *activityGrant { - return newActivityGrant(sessionID, make([]byte, 32)) +func testGrant(sessionID string) *sessionLogGrant { + return newSessionLogGrant(sessionID, make([]byte, 32)) } -func newTestLog(shipper activityShipper) (log *activityLog, advance func(time.Duration), tick func()) { - log = newActivityLog("proxy-1", shipper) +func newTestLog(shipper sessionLogShipper) (log *sessionLogRecorder, advance func(time.Duration), tick func()) { + log = newSessionLogRecorder("proxy-1", shipper) now := time.Date(2026, 9, 16, 10, 0, 0, 0, time.UTC) var mu sync.Mutex log.now = func() time.Time { @@ -136,18 +136,18 @@ func newTestLog(shipper activityShipper) (log *activityLog, advance func(time.Du mu.Unlock() } tick = func() { - advance(activityFlushInterval) + advance(sessionLogFlushInterval) log.flushAll(context.Background(), false) } return log, advance, tick } -func aRecord(host string) activityRecord { - return activityRecord{Method: "GET", Host: host, Port: "443", Path: "/zen", Status: 200, Decision: decisionPassthrough} +func aRecord(host string) sessionLogRecord { + return sessionLogRecord{Method: "GET", Host: host, Port: "443", Path: "/zen", Status: 200, Decision: decisionPassthrough} } func TestRecordingIsANoOpWithoutALogOrAGrant(t *testing.T) { - var nilLog *activityLog + var nilLog *sessionLogRecorder nilLog.record(testGrant("s1"), aRecord("api.github.com")) log, _, _ := newTestLog(&fakeShipper{}) @@ -158,9 +158,9 @@ func TestRecordingIsANoOpWithoutALogOrAGrant(t *testing.T) { } func TestTheRingDropsTheOldestAndCountsIt(t *testing.T) { - ring := newActivityRing(3) + ring := newSessionLogRing(3) for i := 0; i < 5; i++ { - ring.push(activityRecord{Seq: uint64(i)}) + ring.push(sessionLogRecord{Seq: uint64(i)}) } if ring.len() != 3 { @@ -177,11 +177,11 @@ func TestTheRingDropsTheOldestAndCountsIt(t *testing.T) { } func TestTheRingAllocatesOnlyWhatItHolds(t *testing.T) { - ring := newActivityRing(activitySpoolCapacity) - ring.push(activityRecord{Seq: 0}) + ring := newSessionLogRing(sessionLogSpoolCapacity) + ring.push(sessionLogRecord{Seq: 0}) - if got := cap(ring.buf); got > activityRingInitialSize { - t.Fatalf("one record reserved room for %d, expected at most %d", got, activityRingInitialSize) + if got := cap(ring.buf); got > sessionLogRingInitialSize { + t.Fatalf("one record reserved room for %d, expected at most %d", got, sessionLogRingInitialSize) } ring.drain(10) @@ -191,22 +191,22 @@ func TestTheRingAllocatesOnlyWhatItHolds(t *testing.T) { } func TestTheRingKeepsItsOrderWhileItGrowsPastAWrap(t *testing.T) { - ring := newActivityRing(activitySpoolCapacity) + ring := newSessionLogRing(sessionLogSpoolCapacity) var next uint64 push := func(n int) { for i := 0; i < n; i++ { - ring.push(activityRecord{Seq: next}) + ring.push(sessionLogRecord{Seq: next}) next++ } } - push(activityRingInitialSize) + push(sessionLogRingInitialSize) ring.drain(10) - push(activityRingInitialSize * 3) + push(sessionLogRingInitialSize * 3) - drained := ring.drain(activitySpoolCapacity) - if len(drained) != activityRingInitialSize*4-10 { - t.Fatalf("drained %d records, expected %d", len(drained), activityRingInitialSize*4-10) + drained := ring.drain(sessionLogSpoolCapacity) + if len(drained) != sessionLogRingInitialSize*4-10 { + t.Fatalf("drained %d records, expected %d", len(drained), sessionLogRingInitialSize*4-10) } for i, rec := range drained { if rec.Seq != uint64(10+i) { @@ -219,16 +219,16 @@ func TestTheRingKeepsItsOrderWhileItGrowsPastAWrap(t *testing.T) { } func TestTheRingDrainsInSlicesTheServerAccepts(t *testing.T) { - ring := newActivityRing(activitySpoolCapacity) + ring := newSessionLogRing(sessionLogSpoolCapacity) for i := 0; i < 2500; i++ { - ring.push(activityRecord{Seq: uint64(i)}) + ring.push(sessionLogRecord{Seq: uint64(i)}) } var slices int for ring.len() > 0 { - got := ring.drain(activityFlushRecords) - if len(got) > activityFlushRecords { - t.Fatalf("a slice held %d records, the server's limit is %d", len(got), activityFlushRecords) + got := ring.drain(sessionLogFlushRecords) + if len(got) > sessionLogFlushRecords { + t.Fatalf("a slice held %d records, the server's limit is %d", len(got), sessionLogFlushRecords) } slices++ } @@ -242,7 +242,7 @@ func TestTheDropCountIsReportedOnceAndRidesTheFirstChunk(t *testing.T) { log, _, _ := newTestLog(shipper) grant := testGrant("s1") - for i := 0; i < activitySpoolCapacity+50; i++ { + for i := 0; i < sessionLogSpoolCapacity+50; i++ { log.record(grant, aRecord("api.github.com")) } log.flushAll(context.Background(), true) @@ -260,27 +260,27 @@ func TestASequenceNumberIsConsumedEvenWhenARecordIsDropped(t *testing.T) { log, _, _ := newTestLog(&fakeShipper{}) grant := testGrant("s1") - for i := 0; i < activitySpoolCapacity+10; i++ { + for i := 0; i < sessionLogSpoolCapacity+10; i++ { log.record(grant, aRecord("api.github.com")) } - if got := log.spools["s1"].nextSeq; got != uint64(activitySpoolCapacity+10) { - t.Fatalf("nextSeq is %d after %d records; drops must still consume a number", got, activitySpoolCapacity+10) + if got := log.spools["s1"].nextSeq; got != uint64(sessionLogSpoolCapacity+10) { + t.Fatalf("nextSeq is %d after %d records; drops must still consume a number", got, sessionLogSpoolCapacity+10) } } func TestTheProxyWideFuseDropsTheNewest(t *testing.T) { log, _, _ := newTestLog(&fakeShipper{}) - for i := 0; i < activityTotalCapacity/activitySpoolCapacity+2; i++ { + for i := 0; i < sessionLogTotalCapacity/sessionLogSpoolCapacity+2; i++ { grant := testGrant(fmt.Sprintf("s%d", i)) - for j := 0; j < activitySpoolCapacity; j++ { + for j := 0; j < sessionLogSpoolCapacity; j++ { log.record(grant, aRecord("api.github.com")) } } - if log.total > activityTotalCapacity { - t.Fatalf("the proxy holds %d records, past the %d fuse", log.total, activityTotalCapacity) + if log.total > sessionLogTotalCapacity { + t.Fatalf("the proxy holds %d records, past the %d fuse", log.total, sessionLogTotalCapacity) } } @@ -356,7 +356,7 @@ func TestARejectedProxyTokenKeepsEverything(t *testing.T) { } func TestTheCeilingPausesTheWholeProxyAndLiftsAfterTheBackoff(t *testing.T) { - shipper := &fakeShipper{postResults: []scriptedResult{{err: apiErr(400, activityCeilingReachedName)}}} + shipper := &fakeShipper{postResults: []scriptedResult{{err: apiErr(400, sessionLogCeilingReachedName)}}} log, advance, tick := newTestLog(shipper) log.record(testGrant("s1"), aRecord("api.github.com")) @@ -372,7 +372,7 @@ func TestTheCeilingPausesTheWholeProxyAndLiftsAfterTheBackoff(t *testing.T) { t.Fatalf("a second session posted while paused; the pause is proxy-wide") } - advance(activityPauseBackoff + time.Second) + advance(sessionLogPauseBackoff + time.Second) log.flushAll(context.Background(), false) if len(shipper.posts()) < 2 { t.Fatal("nothing was retried after the pause lifted") @@ -380,7 +380,7 @@ func TestTheCeilingPausesTheWholeProxyAndLiftsAfterTheBackoff(t *testing.T) { } func TestBeingSwitchedOffDropsWhatWasHeldAndCountsIt(t *testing.T) { - shipper := &fakeShipper{postResults: []scriptedResult{{err: apiErr(400, activityDisabledName)}}} + shipper := &fakeShipper{postResults: []scriptedResult{{err: apiErr(400, sessionLogDisabledName)}}} log, _, tick := newTestLog(shipper) grant := testGrant("s1") @@ -411,7 +411,7 @@ func TestBeingSwitchedOffDropsWhatWasHeldAndCountsIt(t *testing.T) { } func TestAKeyIssuedAfterTheSwitchOffResumesRecordingAtOnce(t *testing.T) { - shipper := &fakeShipper{postResults: []scriptedResult{{err: apiErr(400, activityDisabledName)}}} + shipper := &fakeShipper{postResults: []scriptedResult{{err: apiErr(400, sessionLogDisabledName)}}} log, _, tick := newTestLog(shipper) log.record(testGrant("s1"), aRecord("api.github.com")) @@ -433,12 +433,12 @@ func TestAKeyIssuedAfterTheSwitchOffResumesRecordingAtOnce(t *testing.T) { } func TestARefusalRacingANewKeyDropsNothing(t *testing.T) { - shipper := &fakeShipper{postResults: []scriptedResult{{err: apiErr(400, activityDisabledName)}}} + shipper := &fakeShipper{postResults: []scriptedResult{{err: apiErr(400, sessionLogDisabledName)}}} log, _, _ := newTestLog(shipper) grant := testGrant("s1") log.record(grant, aRecord("api.github.com")) - log.switchOff(activityGrantsIssued.Load() - 1) + log.switchOff(sessionLogGrantsIssued.Load() - 1) if log.switchedOff || log.spools["s1"].ring.len() != 1 { t.Fatal("a refusal sent before a newer key was issued still dropped what was held") @@ -446,13 +446,13 @@ func TestARefusalRacingANewKeyDropsNothing(t *testing.T) { } func TestDropsAreStillReportedAfterAnIdleSpoolIsForgotten(t *testing.T) { - shipper := &fakeShipper{postResults: []scriptedResult{{err: apiErr(400, activityDisabledName)}}} + shipper := &fakeShipper{postResults: []scriptedResult{{err: apiErr(400, sessionLogDisabledName)}}} log, advance, tick := newTestLog(shipper) log.record(testGrant("s1"), aRecord("api.github.com")) tick() - advance(activityIdleClose + time.Minute) + advance(sessionLogIdleClose + time.Minute) log.flushAll(context.Background(), false) if _, ok := log.spools["s1"]; ok { t.Fatal("the idle spool was not forgotten, so this test proves nothing") @@ -468,7 +468,7 @@ func TestDropsAreStillReportedAfterAnIdleSpoolIsForgotten(t *testing.T) { } func TestRecordsArePausedAsCountedGapsNotSilentLosses(t *testing.T) { - shipper := &fakeShipper{postResults: []scriptedResult{{err: apiErr(400, activityCeilingReachedName)}}} + shipper := &fakeShipper{postResults: []scriptedResult{{err: apiErr(400, sessionLogCeilingReachedName)}}} log, _, tick := newTestLog(shipper) grant := testGrant("s1") @@ -518,7 +518,7 @@ func TestARefusedChunkIsCountedOnTheNextOne(t *testing.T) { } func TestAClockSkewRefusalIsDroppedCountedAndLoggedOnce(t *testing.T) { - skew := scriptedResult{err: apiErr(http.StatusBadRequest, activityClockSkewName)} + skew := scriptedResult{err: apiErr(http.StatusBadRequest, sessionLogClockSkewName)} shipper := &fakeShipper{postResults: []scriptedResult{skew, skew}} log, _, tick := newTestLog(shipper) @@ -550,7 +550,7 @@ func TestAFlushTooBigForOneChunkIsSplitBySize(t *testing.T) { log, _, tick := newTestLog(shipper) grant := testGrant("s1") - for i := 0; i < activityFlushRecords; i++ { + for i := 0; i < sessionLogFlushRecords; i++ { rec := aRecord("api.github.com") rec.Path = truncatePath("/" + strings.Repeat("&", maxLoggedPathLen)) log.record(grant, rec) @@ -565,8 +565,8 @@ func TestAFlushTooBigForOneChunkIsSplitBySize(t *testing.T) { var next uint64 var total int for i, chunk := range pending { - if chunk.meta.CiphertextBytes-gcmTag > activityMaxChunkPlaintext { - t.Fatalf("chunk %d holds %d bytes of plaintext, over %d", i, chunk.meta.CiphertextBytes-gcmTag, activityMaxChunkPlaintext) + if chunk.meta.CiphertextBytes-gcmTag > sessionLogMaxChunkPlaintext { + t.Fatalf("chunk %d holds %d bytes of plaintext, over %d", i, chunk.meta.CiphertextBytes-gcmTag, sessionLogMaxChunkPlaintext) } if chunk.meta.FirstSeq != next { t.Fatalf("chunk %d starts at seq %d, expected %d; a record was lost or reordered", i, chunk.meta.FirstSeq, next) @@ -574,8 +574,8 @@ func TestAFlushTooBigForOneChunkIsSplitBySize(t *testing.T) { next = chunk.meta.LastSeq + 1 total += chunk.meta.RecordCount } - if total != activityFlushRecords { - t.Fatalf("the chunks hold %d records, expected %d", total, activityFlushRecords) + if total != sessionLogFlushRecords { + t.Fatalf("the chunks hold %d records, expected %d", total, sessionLogFlushRecords) } } @@ -601,14 +601,14 @@ func TestThePendingCapEvictsTheOldestAndCountsIt(t *testing.T) { log, _, tick := newTestLog(shipper) grant := testGrant("s1") - for i := 0; i < activityPendingChunks+3; i++ { + for i := 0; i < sessionLogPendingChunks+3; i++ { log.record(grant, aRecord("api.github.com")) tick() } spool := log.spools["s1"] - if len(spool.pending) > activityPendingChunks { - t.Fatalf("pending holds %d chunks, the cap is %d", len(spool.pending), activityPendingChunks) + if len(spool.pending) > sessionLogPendingChunks { + t.Fatalf("pending holds %d chunks, the cap is %d", len(spool.pending), sessionLogPendingChunks) } if spool.ring.dropped == 0 { t.Fatal("evicted chunks were not counted as dropped records") @@ -619,14 +619,14 @@ func TestTheByteCapEvictsTheOldestChunkOnTheProxy(t *testing.T) { log, _, _ := newTestLog(&fakeShipper{}) blob := make([]byte, 12<<20) - add := func(sessionID string, order uint64, posted bool, carried uint64) *activitySpool { + add := func(sessionID string, order uint64, posted bool, carried uint64) *sessionLogSpool { spool, ok := log.spools[sessionID] if !ok { - spool = newActivitySpool(testGrant(sessionID), log.now()) + spool = newSessionLogSpool(testGrant(sessionID), log.now()) log.spools[sessionID] = spool } spool.pending = append(spool.pending, &sealedChunk{ - meta: api.CreateAgentVaultActivityChunkRequest{ChunkID: fmt.Sprintf("c%d", order), RecordCount: 100, DroppedCount: carried}, + meta: api.CreateAgentVaultSessionLogChunkRequest{ChunkID: fmt.Sprintf("c%d", order), RecordCount: 100, DroppedCount: carried}, ciphertext: blob, sealOrder: order, posted: posted, @@ -638,15 +638,15 @@ func TestTheByteCapEvictsTheOldestChunkOnTheProxy(t *testing.T) { log.mu.Lock() add("oldest", 0, false, 7) add("posted", 1, true, 3) - var newest *activitySpool + var newest *sessionLogSpool for i := 2; i < 7; i++ { newest = add(fmt.Sprintf("s%d", i), uint64(i), false, 0) } log.enforcePendingCapsLocked(newest) log.mu.Unlock() - if log.sealedBytes > activityTotalSealedBytes { - t.Fatalf("the proxy holds %d sealed bytes, past the %d cap", log.sealedBytes, activityTotalSealedBytes) + if log.sealedBytes > sessionLogTotalSealedBytes { + t.Fatalf("the proxy holds %d sealed bytes, past the %d cap", log.sealedBytes, sessionLogTotalSealedBytes) } if got := len(log.spools["oldest"].pending) + len(log.spools["posted"].pending); got != 0 { t.Fatalf("the two oldest chunks were not evicted, %d remain", got) @@ -690,7 +690,7 @@ func TestReachingTheSliceSizeWakesTheLoopOnce(t *testing.T) { log, _, _ := newTestLog(&fakeShipper{}) grant := testGrant("s1") - for i := 0; i < activityFlushRecords*2; i++ { + for i := 0; i < sessionLogFlushRecords*2; i++ { log.record(grant, aRecord("api.github.com")) } @@ -706,7 +706,7 @@ func TestAnIdleSpoolIsForgotten(t *testing.T) { log.record(testGrant("s1"), aRecord("api.github.com")) tick() - advance(activityIdleClose + time.Minute) + advance(sessionLogIdleClose + time.Minute) log.flushAll(context.Background(), false) if _, ok := log.spools["s1"]; ok { @@ -715,8 +715,8 @@ func TestAnIdleSpoolIsForgotten(t *testing.T) { } func TestIdleCloseOutlastsTheSessionCacheTTL(t *testing.T) { - if activityIdleClose <= sessionInactiveTTL { - t.Fatalf("idle close (%s) must outlast the session cache TTL (%s)", activityIdleClose, sessionInactiveTTL) + if sessionLogIdleClose <= sessionInactiveTTL { + t.Fatalf("idle close (%s) must outlast the session cache TTL (%s)", sessionLogIdleClose, sessionInactiveTTL) } } @@ -746,7 +746,7 @@ func TestABlockedHostIsStillRecorded(t *testing.T) { shipper := &fakeShipper{} log, _, _ := newTestLog(shipper) - log.record(testGrant("s1"), activityRecord{ + log.record(testGrant("s1"), sessionLogRecord{ Method: "POST", Host: "evil.example", Port: "443", Path: "/collect", Status: 403, Decision: decisionBlocked, }) log.flushAll(context.Background(), true) @@ -800,7 +800,7 @@ func TestSequenceNumbersSurviveASpoolBeingForgotten(t *testing.T) { log.record(grant, aRecord("api.github.com")) tick() - advance(activityIdleClose + time.Minute) + advance(sessionLogIdleClose + time.Minute) log.flushAll(context.Background(), false) if _, ok := log.spools["s1"]; ok { t.Fatal("the idle spool was not forgotten, so this test proves nothing") @@ -829,7 +829,7 @@ func TestRecordsLostToASealFailureAreStillCounted(t *testing.T) { shipper := &fakeShipper{} log, _, tick := newTestLog(shipper) - grant := &activityGrant{sessionID: "s1", key: make([]byte, 7)} + grant := &sessionLogGrant{sessionID: "s1", key: make([]byte, 7)} for i := 0; i < 3; i++ { log.record(grant, aRecord("api.github.com")) } @@ -930,7 +930,7 @@ func TestEverySessionShipsOnEveryTickWhileEarlierOnesTakeTimeToUpload(t *testing for i := 1; i <= 3; i++ { log.record(testGrant("s1"), aRecord("api.github.com")) log.record(testGrant("s2"), aRecord("api.github.com")) - tickAt = tickAt.Add(activityFlushInterval) + tickAt = tickAt.Add(sessionLogFlushInterval) advance(tickAt.Sub(log.now())) log.flushAll(context.Background(), false) @@ -947,7 +947,7 @@ func TestASessionShipsAtATickThatLandsJustShortOfAMinute(t *testing.T) { log.record(testGrant("s1"), aRecord("api.github.com")) tick() log.record(testGrant("s1"), aRecord("api.github.com")) - advance(activityFlushInterval - 10*time.Millisecond) + advance(sessionLogFlushInterval - 10*time.Millisecond) log.flushAll(context.Background(), false) if got := len(shipper.puts()); got != 2 { @@ -962,7 +962,7 @@ func TestASessionDoesNotShipAgainHalfwayToTheNextTick(t *testing.T) { log.record(testGrant("s1"), aRecord("api.github.com")) tick() log.record(testGrant("s1"), aRecord("api.github.com")) - advance(activityFlushInterval / 2) + advance(sessionLogFlushInterval / 2) log.flushAll(context.Background(), false) if got := len(shipper.puts()); got != 1 { diff --git a/packages/agentvault/activity_wiring_test.go b/packages/agentvault/session_log_wiring_test.go similarity index 93% rename from packages/agentvault/activity_wiring_test.go rename to packages/agentvault/session_log_wiring_test.go index 00c115166..2173c80ce 100644 --- a/packages/agentvault/activity_wiring_test.go +++ b/packages/agentvault/session_log_wiring_test.go @@ -15,26 +15,26 @@ type grantingResolver struct { services []*resolvedService } -func (g grantingResolver) resolve(string, *activityGrant) (*resolveResult, error) { +func (g grantingResolver) resolve(string, *sessionLogGrant) (*resolveResult, error) { return &resolveResult{ - SessionID: "s1", - Services: g.services, - Activity: &activityGrant{sessionID: "s1", key: make([]byte, 32)}, + SessionID: "s1", + Services: g.services, + SessionLog: &sessionLogGrant{sessionID: "s1", key: make([]byte, 32)}, }, nil } -func newRecordingProxy(t *testing.T, policy string, services []*resolvedService) (*httptest.Server, *activityLog, *fakeShipper) { +func newRecordingProxy(t *testing.T, policy string, services []*resolvedService) (*httptest.Server, *sessionLogRecorder, *fakeShipper) { t.Helper() shipper := &fakeShipper{} ps := &proxyServer{transport: newUpstreamTransport()} ps.setConfig(ProxyConfig{TrafficPolicy: policy}) ps.cache = newSessionCache(grantingResolver{services: services}, ps.pollInterval) - ps.activity = newActivityLog("proxy-1", shipper) + ps.sessionLogs = newSessionLogRecorder("proxy-1", shipper) front := httptest.NewServer(http.HandlerFunc(ps.dispatch)) t.Cleanup(front.Close) - return front, ps.activity, shipper + return front, ps.sessionLogs, shipper } func proxiedGet(t *testing.T, front *httptest.Server, target string) *http.Response { @@ -55,11 +55,11 @@ func proxiedGet(t *testing.T, front *httptest.Server, target string) *http.Respo return res } -func drainOneRecord(t *testing.T, log *activityLog) activityRecord { +func drainOneRecord(t *testing.T, log *sessionLogRecorder) sessionLogRecord { t.Helper() spool, ok := log.spools["s1"] if !ok { - t.Fatal("the request produced no activity spool") + t.Fatal("the request produced no session log spool") } records := spool.ring.drain(10) if len(records) != 1 { diff --git a/packages/api/agent_vault.go b/packages/api/agent_vault.go index 8849d79ec..0da8832a1 100644 --- a/packages/api/agent_vault.go +++ b/packages/api/agent_vault.go @@ -103,20 +103,20 @@ type AgentVaultService struct { Substitutions []AgentVaultSubstitution `json:"substitutions"` } -type AgentVaultActivityGrant struct { +type AgentVaultSessionLogGrant struct { Enabled bool `json:"enabled"` SessionKey string `json:"sessionKey"` } type ResolveAgentVaultSessionRequest struct { - HasActivityKey bool `json:"hasActivityKey"` + HasSessionLogKey bool `json:"hasSessionLogKey"` } type ResolveAgentVaultSessionResponse struct { - SessionID string `json:"sessionId"` - ExpiresAt string `json:"expiresAt"` - Services []AgentVaultService `json:"services"` - Activity AgentVaultActivityGrant `json:"activity"` + SessionID string `json:"sessionId"` + ExpiresAt string `json:"expiresAt"` + Services []AgentVaultService `json:"services"` + SessionLogs AgentVaultSessionLogGrant `json:"sessionLogs"` } func CallResolveAgentVaultSession(httpClient *resty.Client, sessionToken string, request ResolveAgentVaultSessionRequest) (ResolveAgentVaultSessionResponse, error) { @@ -138,7 +138,7 @@ func CallResolveAgentVaultSession(httpClient *resty.Client, sessionToken string, return res, nil } -type CreateAgentVaultActivityChunkRequest struct { +type CreateAgentVaultSessionLogChunkRequest struct { ChunkID string `json:"chunkId"` StartedAt string `json:"startedAt"` EndedAt string `json:"endedAt"` @@ -151,27 +151,27 @@ type CreateAgentVaultActivityChunkRequest struct { CiphertextSha256 string `json:"ciphertextSha256"` } -type CreateAgentVaultActivityChunkResponse struct { +type CreateAgentVaultSessionLogChunkResponse struct { ChunkID string `json:"chunkId"` UploadURL string `json:"uploadUrl"` ExpiresInSeconds int `json:"expiresInSeconds"` } -func CallCreateAgentVaultActivityChunk(ctx context.Context, httpClient *resty.Client, sessionID string, request CreateAgentVaultActivityChunkRequest) (CreateAgentVaultActivityChunkResponse, error) { - var res CreateAgentVaultActivityChunkResponse +func CallCreateAgentVaultSessionLogChunk(ctx context.Context, httpClient *resty.Client, sessionID string, request CreateAgentVaultSessionLogChunkRequest) (CreateAgentVaultSessionLogChunkResponse, error) { + var res CreateAgentVaultSessionLogChunkResponse response, err := httpClient. R(). SetContext(ctx). SetResult(&res). SetHeader("User-Agent", USER_AGENT). SetBody(request). - Post(fmt.Sprintf("%v/v1/agent-vault/proxy/sessions/%s/activity/chunks", config.INFISICAL_URL, sessionID)) + Post(fmt.Sprintf("%v/v1/agent-vault/proxy/sessions/%s/logs/chunks", config.INFISICAL_URL, sessionID)) if err != nil { - return CreateAgentVaultActivityChunkResponse{}, NewGenericRequestError("CallCreateAgentVaultActivityChunk", err) + return CreateAgentVaultSessionLogChunkResponse{}, NewGenericRequestError("CallCreateAgentVaultSessionLogChunk", err) } if response.IsError() { - return CreateAgentVaultActivityChunkResponse{}, NewAPIErrorWithResponse("CallCreateAgentVaultActivityChunk", response, nil) + return CreateAgentVaultSessionLogChunkResponse{}, NewAPIErrorWithResponse("CallCreateAgentVaultSessionLogChunk", response, nil) } return res, nil } From 96f740db7015838b6715cba8946cc6f856886738 Mon Sep 17 00:00:00 2001 From: saif <11242541+saifsmailbox98@users.noreply.github.com> Date: Sat, 26 Sep 2026 10:37:22 +0530 Subject: [PATCH 25/43] refactor(agent-vault): mint session log chunk ids as UUIDv7 instead of ULID uuid.NewV7 is monotonic within the process, so chunks split from one flush keep their seal order, and a minting failure now drops the batch with an error instead of panicking. oklog/ulid goes back to an indirect dependency. The pinned crypto vector is regenerated for the new fixture id. --- go.mod | 2 +- packages/agentvault/session_log.go | 3 +- packages/agentvault/session_log_crypto.go | 13 +++--- .../agentvault/session_log_crypto_test.go | 45 +++++++++---------- packages/agentvault/session_log_ship_test.go | 4 +- packages/agentvault/session_log_spool.go | 7 ++- 6 files changed, 38 insertions(+), 36 deletions(-) diff --git a/go.mod b/go.mod index 25fdb3a36..25291073f 100644 --- a/go.mod +++ b/go.mod @@ -43,7 +43,6 @@ require ( github.com/muesli/reflow v0.3.0 github.com/muesli/roff v0.1.0 github.com/oiweiwei/go-msrpc v1.5.1 - github.com/oklog/ulid v1.3.1 github.com/pelletier/go-toml/v2 v2.4.3 github.com/pkg/browser v0.0.0-20240102092130-5ac0b6a4141c github.com/pkg/errors v0.9.1 @@ -191,6 +190,7 @@ require ( github.com/oiweiwei/go-oem v1.0.0 // indirect github.com/oiweiwei/go-smb2.fork v1.0.1 // indirect github.com/oiweiwei/gokrb5.fork/v9 v9.0.6 // indirect + github.com/oklog/ulid v1.3.1 // indirect github.com/onsi/ginkgo/v2 v2.22.2 // indirect github.com/onsi/gomega v1.36.2 // indirect github.com/oracle/oci-go-sdk/v65 v65.95.2 // indirect diff --git a/packages/agentvault/session_log.go b/packages/agentvault/session_log.go index 23891fd64..e27b434c7 100644 --- a/packages/agentvault/session_log.go +++ b/packages/agentvault/session_log.go @@ -306,7 +306,6 @@ func (a *sessionLogRecorder) sealRing(spool *sessionLogSpool, started time.Time) } a.total -= len(records) dropped := spool.ring.takeDropped() - now := a.now() a.mu.Unlock() groups, err := packSessionLogRecords(records) @@ -320,7 +319,7 @@ func (a *sessionLogRecorder) sealRing(spool *sessionLogSpool, started time.Time) groupDropped = dropped } - chunk, err := spool.sealSlice(group.records, group.plaintext, groupDropped, now) + chunk, err := spool.sealSlice(group.records, group.plaintext, groupDropped) if err != nil { a.dropUnsealed(spool, len(group.records), groupDropped, err) continue diff --git a/packages/agentvault/session_log_crypto.go b/packages/agentvault/session_log_crypto.go index b2910e43d..717779ad4 100644 --- a/packages/agentvault/session_log_crypto.go +++ b/packages/agentvault/session_log_crypto.go @@ -8,9 +8,8 @@ import ( "encoding/base64" "fmt" "io" - "time" - "github.com/oklog/ulid" + "github.com/google/uuid" ) const sessionLogAADVersion = "v1" @@ -56,8 +55,10 @@ func sessionLogCiphertextSHA256(ciphertext []byte) string { return base64.RawStdEncoding.EncodeToString(sum[:]) } -func newSessionLogChunkID(now time.Time) string { - return ulid.MustNew(ulid.Timestamp(now), newULIDEntropy()).String() +func newSessionLogChunkID() (string, error) { + id, err := uuid.NewV7() + if err != nil { + return "", fmt.Errorf("agent-vault: could not mint a session log chunk id: %w", err) + } + return id.String(), nil } - -func newULIDEntropy() io.Reader { return rand.Reader } diff --git a/packages/agentvault/session_log_crypto_test.go b/packages/agentvault/session_log_crypto_test.go index 3dda3f1de..e7325d0a9 100644 --- a/packages/agentvault/session_log_crypto_test.go +++ b/packages/agentvault/session_log_crypto_test.go @@ -8,23 +8,24 @@ import ( "encoding/base64" "encoding/hex" "encoding/json" + "strings" "testing" "time" - "github.com/oklog/ulid" + "github.com/google/uuid" ) const ( vectorKeyHex = "000102030405060708090a0b0c0d0e0f101112131415161718191a1b1c1d1e1f" vectorIVHex = "aabbccddeeff001122334455" - vectorAADHex = "0bc4c5b3d6ea7cd6bfc440da46d6ce9b73f17c5efa64e90e88373ee8ba09a837" + vectorAADHex = "38d8f3b86fbbd1061f9d2a8bd91e6c49c51b6231936a7ecfc21c56e129f37a1c" vectorIVBase64 = "qrvM3e7/ABEiM0RV" - vectorCiphertext = "PLRwxBbgu+W68Br1N9gY1oUy8wjJxQClAtBh0NfJS1UcWOCPn3laS615sIqwFONhPIPNWRI3CA+a5tUJ7aoim0sQkE4d9gzou2mc/AWiCdToVBJPtdumA9jIzh3yAI81YPwcoDXEVnq2+7ooNNJShGdLX95itbrna/t4nFKRKSSgNzbH23eMtSMcSo72puk/2iwh4sVbTKzC2kwvbf1U6Mgd21zkIq2jDKKwhcT6mTfjPivW4FzmmkspQVMoWwANRX+QVyXzrMipZfoq5N/UcUI6rCu64JVIU0dTBbrrs+2AuZxL" + vectorCiphertext = "PLRwxBbgu+W68Br1N9gY1oUy8wjJxQClAtBh0NfJS1UcWOCPn3laS615sIqwFONhPIPNWRI3CA+a5tUJ7aoim0sQkE4d9gzou2mc/AWiCdToVBJPtdumA9jIzh3yAI81YPwcoDXEVnq2+7ooNNJShGdLX95itbrna/t4nFKRKSSgNzbH23eMtSMcSo72puk/2iwh4sVbTKzC2kwvbf1U6Mgd21zkIq2jDKKwhcT6mTfjPivW4FzmmkspQVMoWwANRX+QVyXzrMipZfoq5N/UcUI6rCvRUkqg+3ST5GVMelW0mjOO" ) var vectorContext = struct{ sessionID, chunkID string }{ sessionID: "sess-1", - chunkID: "01K5ABCDEFGHJKMNPQRSTVWXYZ", + chunkID: "01a0a9c5-231d-7abc-8def-0123456789ab", } func vectorRecords() []sessionLogRecord { @@ -84,7 +85,7 @@ func TestSealMatchesNodeVector(t *testing.T) { func TestASealedChunkCarriesTheDigestOfExactlyWhatIsUploaded(t *testing.T) { key := make([]byte, 32) spool := newSessionLogSpool(newSessionLogGrant("sess-1", key), time.Now()) - chunk, err := spool.sealSlice(vectorRecords(), []byte("[]"), 0, time.Now()) + chunk, err := spool.sealSlice(vectorRecords(), []byte("[]"), 0) if err != nil { t.Fatal(err) } @@ -174,29 +175,27 @@ func TestIVsDoNotRepeat(t *testing.T) { } } -func TestChunkIDsAreULIDsThatSortByTime(t *testing.T) { - earlier := newSessionLogChunkID(time.Date(2026, 9, 16, 10, 0, 0, 0, time.UTC)) - later := newSessionLogChunkID(time.Date(2026, 9, 16, 11, 0, 0, 0, time.UTC)) - - if len(earlier) != 26 { - t.Fatalf("a chunk id is %d characters, the server's column is 26", len(earlier)) +func TestChunkIDsAreLowercaseUUIDv7sThatSortInMintOrder(t *testing.T) { + earlier, err := newSessionLogChunkID() + if err != nil { + t.Fatal(err) + } + later, err := newSessionLogChunkID() + if err != nil { + t.Fatal(err) } + if !(earlier < later) { t.Fatalf("%q did not sort before %q", earlier, later) } - if _, err := ulid.Parse(earlier); err != nil { + if earlier != strings.ToLower(earlier) { + t.Fatalf("%q is not lowercase, which is how Infisical returns it and the browser rebuilds the AAD", earlier) + } + parsed, err := uuid.Parse(earlier) + if err != nil { t.Fatalf("a minted chunk id did not parse: %v", err) } -} - -func TestChunkIDsAreUniqueWithinAMillisecond(t *testing.T) { - now := time.Now() - seen := make(map[string]bool, 1000) - for i := 0; i < 1000; i++ { - id := newSessionLogChunkID(now) - if seen[id] { - t.Fatal("a chunk id repeated, which would collide with the server's unique index") - } - seen[id] = true + if parsed.Version() != 7 { + t.Fatalf("a chunk id is UUID version %d, the server only accepts 7", parsed.Version()) } } diff --git a/packages/agentvault/session_log_ship_test.go b/packages/agentvault/session_log_ship_test.go index cda76c735..e6cd17575 100644 --- a/packages/agentvault/session_log_ship_test.go +++ b/packages/agentvault/session_log_ship_test.go @@ -46,7 +46,7 @@ func TestTheChunkPostCarriesTheProxyTokenAndTheBucketPutDoesNot(t *testing.T) { postedPath = r.URL.Path mu.Unlock() w.Header().Set("Content-Type", "application/json") - _, _ = w.Write([]byte(`{"chunkId":"01K5ABCDEFGHJKMNPQRSTVWXYZ","uploadUrl":"` + bucket.URL + `/object","expiresInSeconds":300}`)) + _, _ = w.Write([]byte(`{"chunkId":"01a0a9c5-231d-7abc-8def-0123456789ab","uploadUrl":"` + bucket.URL + `/object","expiresInSeconds":300}`)) })) defer infisical.Close() @@ -62,7 +62,7 @@ func TestTheChunkPostCarriesTheProxyTokenAndTheBucketPutDoesNot(t *testing.T) { ciphertext := []byte("sealed-bytes") res, err := shipper.createChunk(context.Background(), false, "sess-1", api.CreateAgentVaultSessionLogChunkRequest{ - ChunkID: "01K5ABCDEFGHJKMNPQRSTVWXYZ", + ChunkID: "01a0a9c5-231d-7abc-8def-0123456789ab", RecordCount: 1, CiphertextBytes: len(ciphertext), }) diff --git a/packages/agentvault/session_log_spool.go b/packages/agentvault/session_log_spool.go index 1427ea0bb..682200938 100644 --- a/packages/agentvault/session_log_spool.go +++ b/packages/agentvault/session_log_spool.go @@ -172,8 +172,11 @@ func packSessionLogRecords(records []sessionLogRecord) ([]sessionLogGroup, error return groups, nil } -func (s *sessionLogSpool) sealSlice(records []sessionLogRecord, plaintext []byte, dropped uint64, now time.Time) (*sealedChunk, error) { - chunkID := newSessionLogChunkID(now) +func (s *sessionLogSpool) sealSlice(records []sessionLogRecord, plaintext []byte, dropped uint64) (*sealedChunk, error) { + chunkID, err := newSessionLogChunkID() + if err != nil { + return nil, err + } aad := buildSessionLogAAD(s.sessionID, chunkID) ciphertext, iv, err := sealSessionLog(s.key, plaintext, aad) if err != nil { From 653bb8ef0f556ef943795d9c0b7fb8d4c0b29934 Mon Sep 17 00:00:00 2001 From: saif <11242541+saifsmailbox98@users.noreply.github.com> Date: Sun, 27 Sep 2026 15:12:01 +0530 Subject: [PATCH 26/43] test(agent-vault): cover held-byte release and asking for the session log key Every way a chunk leaves the proxy returns its bytes to the pending cap, and resolve asks for the key until the proxy holds it. --- packages/agentvault/resolve_client_test.go | 62 ++++++++++++++++++++++ packages/agentvault/session_log_test.go | 22 ++++++++ 2 files changed, 84 insertions(+) diff --git a/packages/agentvault/resolve_client_test.go b/packages/agentvault/resolve_client_test.go index 340521e9f..87dfa5a46 100644 --- a/packages/agentvault/resolve_client_test.go +++ b/packages/agentvault/resolve_client_test.go @@ -1,8 +1,11 @@ package agentvault import ( + "encoding/base64" + "encoding/json" "net/http" "net/http/httptest" + "sync" "sync/atomic" "testing" @@ -33,3 +36,62 @@ func TestAResolveDoesNotRetryA429WhileTheAgentWaits(t *testing.T) { t.Fatalf("one resolve against a 429 made %d requests; retries belong to the poll loop, not the request path", got) } } + +func TestAResolveAsksForTheKeyUntilItHoldsOne(t *testing.T) { + key := aKey(7) + var mu sync.Mutex + var asked []bool + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + var body struct { + HasSessionLogKey bool `json:"hasSessionLogKey"` + } + if err := json.NewDecoder(r.Body).Decode(&body); err != nil { + t.Errorf("the resolve body did not decode: %v", err) + } + mu.Lock() + asked = append(asked, body.HasSessionLogKey) + mu.Unlock() + + sessionKey := "" + if !body.HasSessionLogKey { + sessionKey = base64.StdEncoding.EncodeToString(key) + } + w.Header().Set("Content-Type", "application/json") + _ = json.NewEncoder(w).Encode(map[string]any{ + "sessionId": "s1", + "services": []any{}, + "sessionLogs": map[string]any{"enabled": true, "sessionKey": sessionKey}, + }) + })) + defer srv.Close() + old := config.INFISICAL_URL + config.INFISICAL_URL = srv.URL + "/api" + defer func() { config.INFISICAL_URL = old }() + + resolver, err := newInfisicalResolver(func() string { return "ptok" }) + if err != nil { + t.Fatal(err) + } + + first, err := resolver.resolve("tok", nil) + if err != nil { + t.Fatal(err) + } + if first.SessionLog == nil || string(first.SessionLog.key) != string(key) { + t.Fatal("the first resolve did not come back with the key Infisical sent") + } + + again, err := resolver.resolve("tok", first.SessionLog) + if err != nil { + t.Fatal(err) + } + if again.SessionLog == nil || string(again.SessionLog.key) != string(key) { + t.Fatal("the key was lost once the proxy said it already held it") + } + + mu.Lock() + defer mu.Unlock() + if len(asked) != 2 || asked[0] || !asked[1] { + t.Fatalf("hasSessionLogKey went out as %v; it must be false until the proxy holds the key, then true", asked) + } +} diff --git a/packages/agentvault/session_log_test.go b/packages/agentvault/session_log_test.go index e0ab61de7..bc464667e 100644 --- a/packages/agentvault/session_log_test.go +++ b/packages/agentvault/session_log_test.go @@ -969,3 +969,25 @@ func TestASessionDoesNotShipAgainHalfwayToTheNextTick(t *testing.T) { t.Fatalf("there were %d uploads; a session shipped twice within one interval", got) } } + +func TestEveryWayAChunkLeavesReleasesWhatItHeld(t *testing.T) { + for _, outcome := range []struct { + name string + post scriptedResult + }{ + {"uploaded", scriptedResult{}}, + {"refused as bad", scriptedResult{err: apiErr(http.StatusUnprocessableEntity, "")}}, + {"session gone", scriptedResult{err: apiErr(http.StatusNotFound, "")}}, + } { + shipper := &fakeShipper{postResults: []scriptedResult{outcome.post}} + log, _, tick := newTestLog(shipper) + + log.record(testGrant("s1"), aRecord("api.github.com")) + tick() + + if log.total != 0 || log.sealedBytes != 0 { + t.Fatalf("%s: the proxy still counts %d records and %d sealed bytes; the pending cap would fill and stop recording", + outcome.name, log.total, log.sealedBytes) + } + } +} From fa698dee5d4542a422105082d237a3e5f3110f3e Mon Sep 17 00:00:00 2001 From: saif <11242541+saifsmailbox98@users.noreply.github.com> Date: Sun, 27 Sep 2026 15:28:52 +0530 Subject: [PATCH 27/43] test(agent-vault): pin retry vs drop, the record limit per chunk, shutdown shipping once, and what a sealed chunk holds The old shutdown race test only caught its bug by luck; the new one blocks the run loop's upload so a second post fails deterministically. The wiring test now decrypts the uploaded chunk and checks its records. --- packages/agentvault/session_log_test.go | 159 ++++++++++++++---- .../agentvault/session_log_wiring_test.go | 27 +++ 2 files changed, 154 insertions(+), 32 deletions(-) diff --git a/packages/agentvault/session_log_test.go b/packages/agentvault/session_log_test.go index bc464667e..322b06fb9 100644 --- a/packages/agentvault/session_log_test.go +++ b/packages/agentvault/session_log_test.go @@ -19,6 +19,8 @@ type shipperCall struct { chunkID string url string bytes int + records int + iv string dropped uint64 body []byte final bool @@ -47,7 +49,7 @@ func (f *fakeShipper) createChunk(_ context.Context, final bool, sessionID strin f.mu.Lock() defer f.mu.Unlock() - f.calls = append(f.calls, shipperCall{kind: "post", sessionID: sessionID, chunkID: req.ChunkID, bytes: req.CiphertextBytes, dropped: req.DroppedCount, final: final}) + f.calls = append(f.calls, shipperCall{kind: "post", sessionID: sessionID, chunkID: req.ChunkID, bytes: req.CiphertextBytes, records: req.RecordCount, iv: req.IV, dropped: req.DroppedCount, final: final}) result := f.postDefault if len(f.postResults) > 0 { @@ -880,37 +882,6 @@ func TestShutdownPastItsBudgetStartsNoNewChunk(t *testing.T) { } } -func TestShutdownDoesNotRaceTheRunLoop(t *testing.T) { - shipper := &fakeShipper{} - log, _, _ := newTestLog(shipper) - log.now = time.Now - - stop := make(chan struct{}) - go log.run(stop) - - grant := testGrant("s1") - done := make(chan struct{}) - go func() { - defer close(done) - for i := 0; i < 2000; i++ { - log.record(grant, aRecord("api.github.com")) - } - }() - - <-done - close(stop) - log.close(context.Background()) - - if len(shipper.puts()) == 0 { - t.Fatal("shutdown shipped nothing") - } - for _, call := range shipper.puts() { - if call.bytes == 0 { - t.Fatal("an empty object was uploaded") - } - } -} - type slowPutShipper struct { *fakeShipper advance func(time.Duration) @@ -991,3 +962,127 @@ func TestEveryWayAChunkLeavesReleasesWhatItHeld(t *testing.T) { } } } + +func TestServerErrorsAreRetriedAndBadChunksAreDropped(t *testing.T) { + for _, tc := range []struct { + status int + kept bool + }{ + {http.StatusInternalServerError, true}, + {http.StatusBadGateway, true}, + {http.StatusServiceUnavailable, true}, + {http.StatusTooManyRequests, true}, + {http.StatusUnprocessableEntity, false}, + {http.StatusConflict, false}, + } { + shipper := &fakeShipper{postResults: []scriptedResult{{err: apiErr(tc.status, "")}}} + log, _, tick := newTestLog(shipper) + + log.record(testGrant("s1"), aRecord("api.github.com")) + tick() + + spool := log.spools["s1"] + if tc.kept { + if len(spool.pending) != 1 || !log.infisicalDown { + t.Fatalf("%d: the chunk was not kept for a retry (pending %d, infisicalDown %v)", tc.status, len(spool.pending), log.infisicalDown) + } + continue + } + if len(spool.pending) != 0 || spool.ring.dropped != 1 { + t.Fatalf("%d: a refused chunk was not dropped and counted (pending %d, dropped %d)", tc.status, len(spool.pending), spool.ring.dropped) + } + } +} + +func TestTheRecorderShipsNoChunkOverTheServersRecordLimit(t *testing.T) { + shipper := &fakeShipper{} + log, _, _ := newTestLog(shipper) + + for i := 0; i < 2500; i++ { + log.record(testGrant("s1"), aRecord("api.github.com")) + } + log.flushAll(context.Background(), true) + + posts := shipper.posts() + if len(posts) != 3 { + t.Fatalf("2500 records shipped as %d chunks, expected 3", len(posts)) + } + total := 0 + for _, post := range posts { + if post.records > sessionLogFlushRecords { + t.Fatalf("a chunk carried %d records; the server refuses more than %d", post.records, sessionLogFlushRecords) + } + total += post.records + } + if total != 2500 { + t.Fatalf("the chunks carried %d records, expected 2500", total) + } +} + +type blockingShipper struct { + fakeShipper + entered chan struct{} + release chan struct{} + again chan struct{} + once sync.Once + posted int +} + +func (b *blockingShipper) createChunk(ctx context.Context, final bool, sessionID string, req api.CreateAgentVaultSessionLogChunkRequest) (api.CreateAgentVaultSessionLogChunkResponse, error) { + b.mu.Lock() + b.posted++ + first := b.posted == 1 + b.mu.Unlock() + if first { + close(b.entered) + <-b.release + } else { + b.once.Do(func() { close(b.again) }) + } + return b.fakeShipper.createChunk(ctx, final, sessionID, req) +} + +func TestShutdownWaitsForAFlushInProgressSoNoChunkShipsTwice(t *testing.T) { + shipper := &blockingShipper{entered: make(chan struct{}), release: make(chan struct{}), again: make(chan struct{})} + log, advance, _ := newTestLog(shipper) + + log.record(testGrant("s1"), aRecord("api.github.com")) + advance(sessionLogFlushInterval) + + tickDone := make(chan struct{}) + go func() { + defer close(tickDone) + log.flushAll(context.Background(), false) + }() + <-shipper.entered + + closeDone := make(chan struct{}) + go func() { + defer close(closeDone) + log.close(context.Background()) + }() + + select { + case <-shipper.again: + close(shipper.release) + t.Fatal("shutdown posted a chunk while the run loop was still shipping it") + case <-time.After(200 * time.Millisecond): + } + + close(shipper.release) + <-tickDone + <-closeDone + + seen := map[string]int{} + for _, post := range shipper.posts() { + seen[post.chunkID]++ + } + for chunkID, n := range seen { + if n != 1 { + t.Fatalf("chunk %s was posted %d times", chunkID, n) + } + } + if len(seen) != 1 || len(shipper.puts()) != 1 { + t.Fatalf("one record shipped as %d chunks and %d uploads, expected 1 of each", len(seen), len(shipper.puts())) + } +} diff --git a/packages/agentvault/session_log_wiring_test.go b/packages/agentvault/session_log_wiring_test.go index 2173c80ce..d59b92c7f 100644 --- a/packages/agentvault/session_log_wiring_test.go +++ b/packages/agentvault/session_log_wiring_test.go @@ -3,6 +3,10 @@ package agentvault import ( "bytes" "context" + "crypto/aes" + "crypto/cipher" + "encoding/base64" + "encoding/json" "fmt" "net/http" "net/http/httptest" @@ -230,4 +234,27 @@ func TestAWholeRequestRoundTripsFromProxyToSealedChunk(t *testing.T) { if bytes.Contains(puts[0].body, []byte("/v1/thing")) || bytes.Contains(puts[0].body, []byte("\"method\"")) { t.Fatal("the uploaded chunk contains readable record fields; it was not sealed") } + + iv, err := base64.RawStdEncoding.DecodeString(posts[0].iv) + if err != nil { + t.Fatal(err) + } + block, _ := aes.NewCipher(make([]byte, 32)) + gcm, _ := cipher.NewGCM(block) + plaintext, err := gcm.Open(nil, iv, puts[0].body, buildSessionLogAAD("s1", posts[0].chunkID)) + if err != nil { + t.Fatalf("the uploaded chunk does not open with the session key and its AAD: %v", err) + } + var records []sessionLogRecord + if err := json.Unmarshal(plaintext, &records); err != nil { + t.Fatalf("the opened chunk is not a JSON record list: %v", err) + } + if len(records) != 3 || posts[0].records != 3 { + t.Fatalf("the chunk holds %d records and declares %d, expected 3", len(records), posts[0].records) + } + for i, record := range records { + if want := fmt.Sprintf("/v1/thing/%d", i); record.Path != want || record.Decision != decisionPassthrough { + t.Fatalf("record %d is %s %s, expected %s %s", i, record.Decision, record.Path, decisionPassthrough, want) + } + } } From 5bb2f12facd99f95a03020956e5404312c749083 Mon Sep 17 00:00:00 2001 From: saif <11242541+saifsmailbox98@users.noreply.github.com> Date: Sun, 27 Sep 2026 15:31:58 +0530 Subject: [PATCH 28/43] test(agent-vault): name the pinned session log vector after the browser test that shares it The backend's copy of the vector tested no backend code and was removed, so these point at the browser instead. --- packages/agentvault/session_log_crypto.go | 2 +- packages/agentvault/session_log_crypto_test.go | 8 ++++---- 2 files changed, 5 insertions(+), 5 deletions(-) diff --git a/packages/agentvault/session_log_crypto.go b/packages/agentvault/session_log_crypto.go index 717779ad4..5f7822788 100644 --- a/packages/agentvault/session_log_crypto.go +++ b/packages/agentvault/session_log_crypto.go @@ -16,7 +16,7 @@ const sessionLogAADVersion = "v1" const sessionLogIVBytes = 12 -// Must byte-match frontend sessionLogDecrypt.ts and the vector pinned in agent-vault-session-log-crypto.test.ts. +// Must byte-match frontend sessionLogDecrypt.ts, whose test pins the same vector as session_log_crypto_test.go. func buildSessionLogAAD(sessionID, chunkID string) []byte { sum := sha256.Sum256([]byte(fmt.Sprintf("%s|%s|%s", sessionID, chunkID, sessionLogAADVersion))) return sum[:] diff --git a/packages/agentvault/session_log_crypto_test.go b/packages/agentvault/session_log_crypto_test.go index e7325d0a9..9c3fcf42f 100644 --- a/packages/agentvault/session_log_crypto_test.go +++ b/packages/agentvault/session_log_crypto_test.go @@ -54,14 +54,14 @@ func mustHex(t *testing.T, s string) []byte { return b } -func TestSessionLogAADMatchesTheBackendVector(t *testing.T) { +func TestSessionLogAADMatchesTheBrowserVector(t *testing.T) { got := buildSessionLogAAD(vectorContext.sessionID, vectorContext.chunkID) if hex.EncodeToString(got) != vectorAADHex { - t.Fatalf("AAD is %s, the backend and the browser build %s", hex.EncodeToString(got), vectorAADHex) + t.Fatalf("AAD is %s, the browser builds %s", hex.EncodeToString(got), vectorAADHex) } } -func TestSealMatchesNodeVector(t *testing.T) { +func TestSealMatchesTheBrowserVector(t *testing.T) { plaintext, err := json.Marshal(vectorRecords()) if err != nil { t.Fatal(err) @@ -78,7 +78,7 @@ func TestSealMatchesNodeVector(t *testing.T) { t.Fatalf("IV encodes as %q, the backend expects %q", encodeSessionLogIV(gotIV), vectorIVBase64) } if base64.StdEncoding.EncodeToString(ciphertext) != vectorCiphertext { - t.Fatal("the sealed bytes differ from the vector Infisical and the browser are checked against") + t.Fatal("the sealed bytes differ from the vector the browser opens") } } From ff600216154aba887d81b5038874942455f18240 Mon Sep 17 00:00:00 2001 From: saif <11242541+saifsmailbox98@users.noreply.github.com> Date: Sun, 27 Sep 2026 16:33:16 +0530 Subject: [PATCH 29/43] refactor(agent-vault): drop the unread chunkId from the chunk reply and the zero-capacity ring branch --- packages/agentvault/session_log_spool.go | 4 ---- packages/agentvault/session_log_test.go | 2 +- packages/api/agent_vault.go | 1 - 3 files changed, 1 insertion(+), 6 deletions(-) diff --git a/packages/agentvault/session_log_spool.go b/packages/agentvault/session_log_spool.go index 682200938..67c0ef8ad 100644 --- a/packages/agentvault/session_log_spool.go +++ b/packages/agentvault/session_log_spool.go @@ -39,10 +39,6 @@ func newSessionLogRing(capacity int) sessionLogRing { func (r *sessionLogRing) len() int { return r.n } func (r *sessionLogRing) push(rec sessionLogRecord) (evicted bool) { - if r.capacity == 0 { - r.dropped++ - return true - } if r.n == len(r.buf) && len(r.buf) < r.capacity { r.grow() } diff --git a/packages/agentvault/session_log_test.go b/packages/agentvault/session_log_test.go index 322b06fb9..3474b4027 100644 --- a/packages/agentvault/session_log_test.go +++ b/packages/agentvault/session_log_test.go @@ -64,7 +64,7 @@ func (f *fakeShipper) createChunk(_ context.Context, final bool, sessionID strin f.nextURL++ url = fmt.Sprintf("https://bucket.example/put/%d", f.nextURL) } - return api.CreateAgentVaultSessionLogChunkResponse{ChunkID: req.ChunkID, UploadURL: url, ExpiresInSeconds: 300}, nil + return api.CreateAgentVaultSessionLogChunkResponse{UploadURL: url, ExpiresInSeconds: 300}, nil } func (f *fakeShipper) putObject(_ context.Context, url string, ciphertext []byte) error { diff --git a/packages/api/agent_vault.go b/packages/api/agent_vault.go index 0da8832a1..8dd7d35ff 100644 --- a/packages/api/agent_vault.go +++ b/packages/api/agent_vault.go @@ -152,7 +152,6 @@ type CreateAgentVaultSessionLogChunkRequest struct { } type CreateAgentVaultSessionLogChunkResponse struct { - ChunkID string `json:"chunkId"` UploadURL string `json:"uploadUrl"` ExpiresInSeconds int `json:"expiresInSeconds"` } From cfa9f85f8adde0f765d89e0a200ed989638e6eb1 Mon Sep 17 00:00:00 2001 From: Saif Ur Rahman Date: Mon, 28 Sep 2026 06:06:35 +0530 Subject: [PATCH 30/43] feat(agent-vault): send each chunk's SHA-256 on upload so S3 can check it --- packages/agentvault/session_log_crypto.go | 5 +++++ packages/agentvault/session_log_ship.go | 2 ++ packages/agentvault/session_log_ship_test.go | 5 +++++ 3 files changed, 12 insertions(+) diff --git a/packages/agentvault/session_log_crypto.go b/packages/agentvault/session_log_crypto.go index 5f7822788..1da5905ad 100644 --- a/packages/agentvault/session_log_crypto.go +++ b/packages/agentvault/session_log_crypto.go @@ -55,6 +55,11 @@ func sessionLogCiphertextSHA256(ciphertext []byte) string { return base64.RawStdEncoding.EncodeToString(sum[:]) } +func sessionLogPaddedSHA256(ciphertext []byte) string { + sum := sha256.Sum256(ciphertext) + return base64.StdEncoding.EncodeToString(sum[:]) +} + func newSessionLogChunkID() (string, error) { id, err := uuid.NewV7() if err != nil { diff --git a/packages/agentvault/session_log_ship.go b/packages/agentvault/session_log_ship.go index 37627613e..14efcaae5 100644 --- a/packages/agentvault/session_log_ship.go +++ b/packages/agentvault/session_log_ship.go @@ -98,6 +98,8 @@ func (c *sessionLogShipperClient) putObject(ctx context.Context, uploadURL strin req.Header.Set("Content-Type", "application/octet-stream") req.Header.Set("Content-Length", strconv.Itoa(len(ciphertext))) req.Header.Set("If-None-Match", "*") + // Signed into the link, so S3 refuses any body whose digest isn't the one Infisical recorded. + req.Header.Set("X-Amz-Checksum-Sha256", sessionLogPaddedSHA256(ciphertext)) res, err := c.put.Do(req) if err != nil { diff --git a/packages/agentvault/session_log_ship_test.go b/packages/agentvault/session_log_ship_test.go index e6cd17575..f19d8a7d4 100644 --- a/packages/agentvault/session_log_ship_test.go +++ b/packages/agentvault/session_log_ship_test.go @@ -22,6 +22,7 @@ func TestTheChunkPostCarriesTheProxyTokenAndTheBucketPutDoesNot(t *testing.T) { putLength string putType string putIfNone string + putSHA256 string putBody []byte postedPath string ) @@ -32,6 +33,7 @@ func TestTheChunkPostCarriesTheProxyTokenAndTheBucketPutDoesNot(t *testing.T) { putLength = r.Header.Get("Content-Length") putType = r.Header.Get("Content-Type") putIfNone = r.Header.Get("If-None-Match") + putSHA256 = r.Header.Get("X-Amz-Checksum-Sha256") buf := make([]byte, r.ContentLength) _, _ = r.Body.Read(buf) putBody = buf @@ -94,6 +96,9 @@ func TestTheChunkPostCarriesTheProxyTokenAndTheBucketPutDoesNot(t *testing.T) { if putIfNone != "*" { t.Fatalf("the upload sent If-None-Match %q; it must be create-only", putIfNone) } + if putSHA256 != sessionLogCiphertextSHA256(ciphertext)+"=" { + t.Fatalf("the upload sent X-Amz-Checksum-Sha256 %q; it must be the padded digest Infisical signed", putSHA256) + } if string(putBody) != string(ciphertext) { t.Fatalf("the bucket received %q", string(putBody)) } From df3bfa7af47334a9565715ceadc9e6e592ee36b3 Mon Sep 17 00:00:00 2001 From: Saif Ur Rahman Date: Mon, 28 Sep 2026 08:43:33 +0530 Subject: [PATCH 31/43] fix(agent-vault): free shipped session log chunks, span each chunk from its earliest to latest record, and retry a 408 --- packages/agentvault/session_log.go | 14 +++++++------ packages/agentvault/session_log_ship.go | 2 +- packages/agentvault/session_log_spool.go | 26 ++++++++++++++++++++++-- packages/agentvault/session_log_test.go | 25 +++++++++++++++++++++++ 4 files changed, 58 insertions(+), 9 deletions(-) diff --git a/packages/agentvault/session_log.go b/packages/agentvault/session_log.go index e27b434c7..6997fdd13 100644 --- a/packages/agentvault/session_log.go +++ b/packages/agentvault/session_log.go @@ -124,8 +124,11 @@ func (a *sessionLogRecorder) record(g *sessionLogGrant, rec sessionLogRecord) { rec.Seq = spool.nextSeq spool.nextSeq++ rec.ProxyID = a.proxyID - rec.Ts = a.now().UTC().Format(time.RFC3339Nano) - spool.lastRecordAt = a.now() + now := a.now() + // Without the monotonic reading, Before and After compare wall time, which is what Ts shows. + rec.at = now.Round(0) + rec.Ts = rec.at.UTC().Format(time.RFC3339Nano) + spool.lastRecordAt = now if a.switchedOff { if g.issued <= a.switchedOffThrough { @@ -371,8 +374,7 @@ func (a *sessionLogRecorder) oldestPendingLocked() *sessionLogSpool { } func (a *sessionLogRecorder) evictOldestLocked(spool *sessionLogSpool) { - oldest := spool.pending[0] - spool.pending = spool.pending[1:] + oldest := spool.popPending() a.sealedBytes -= len(oldest.ciphertext) spool.ring.dropped += oldest.lostCount() log.Warn(). @@ -403,7 +405,7 @@ func (a *sessionLogRecorder) flushSpool(ctx context.Context, spool *sessionLogSp a.mu.Lock() if len(spool.pending) > 0 && spool.pending[0] == chunk { - spool.pending = spool.pending[1:] + spool.popPending() a.sealedBytes -= len(chunk.ciphertext) } a.mu.Unlock() @@ -510,7 +512,7 @@ func (a *sessionLogRecorder) dropRefused(spool *sessionLogSpool, chunk *sealedCh a.mu.Lock() defer a.mu.Unlock() if len(spool.pending) > 0 && spool.pending[0] == chunk { - spool.pending = spool.pending[1:] + spool.popPending() a.sealedBytes -= len(chunk.ciphertext) // Counted as dropped, or an agent that gets its own chunk refused could erase what it did. spool.ring.dropped += chunk.lostCount() diff --git a/packages/agentvault/session_log_ship.go b/packages/agentvault/session_log_ship.go index 14efcaae5..dedd54fec 100644 --- a/packages/agentvault/session_log_ship.go +++ b/packages/agentvault/session_log_ship.go @@ -24,7 +24,7 @@ func isPoisonChunk(err error) bool { if !errors.As(err, &apiErr) { return false } - if apiErr.StatusCode == http.StatusTooManyRequests { + if apiErr.StatusCode == http.StatusRequestTimeout || apiErr.StatusCode == http.StatusTooManyRequests { return false } return apiErr.StatusCode >= 400 && apiErr.StatusCode < 500 diff --git a/packages/agentvault/session_log_spool.go b/packages/agentvault/session_log_spool.go index 67c0ef8ad..73ec19fe2 100644 --- a/packages/agentvault/session_log_spool.go +++ b/packages/agentvault/session_log_spool.go @@ -20,6 +20,8 @@ type sessionLogRecord struct { Decision string `json:"decision"` Service *string `json:"service"` AccessBundle *string `json:"accessBundle"` + + at time.Time } const sessionLogRingInitialSize = 64 @@ -121,6 +123,16 @@ type sessionLogSpool struct { lastFlushAt time.Time } +func (s *sessionLogSpool) popPending() *sealedChunk { + chunk := s.pending[0] + s.pending[0] = nil + s.pending = s.pending[1:] + if len(s.pending) == 0 { + s.pending = nil + } + return chunk +} + func newSessionLogSpool(g *sessionLogGrant, now time.Time) *sessionLogSpool { return &sessionLogSpool{ sessionID: g.sessionID, @@ -179,12 +191,22 @@ func (s *sessionLogSpool) sealSlice(records []sessionLogRecord, plaintext []byte return nil, err } + // The wall clock can step, so the first and last records are not always the earliest and latest. + earliest, latest := records[0], records[0] + for _, rec := range records[1:] { + if rec.at.Before(earliest.at) { + earliest = rec + } + if rec.at.After(latest.at) { + latest = rec + } + } first, last := records[0], records[len(records)-1] return &sealedChunk{ meta: api.CreateAgentVaultSessionLogChunkRequest{ ChunkID: chunkID, - StartedAt: first.Ts, - EndedAt: last.Ts, + StartedAt: earliest.Ts, + EndedAt: latest.Ts, FirstSeq: first.Seq, LastSeq: last.Seq, RecordCount: len(records), diff --git a/packages/agentvault/session_log_test.go b/packages/agentvault/session_log_test.go index 3474b4027..c90658676 100644 --- a/packages/agentvault/session_log_test.go +++ b/packages/agentvault/session_log_test.go @@ -972,6 +972,7 @@ func TestServerErrorsAreRetriedAndBadChunksAreDropped(t *testing.T) { {http.StatusBadGateway, true}, {http.StatusServiceUnavailable, true}, {http.StatusTooManyRequests, true}, + {http.StatusRequestTimeout, true}, {http.StatusUnprocessableEntity, false}, {http.StatusConflict, false}, } { @@ -1086,3 +1087,27 @@ func TestShutdownWaitsForAFlushInProgressSoNoChunkShipsTwice(t *testing.T) { t.Fatalf("one record shipped as %d chunks and %d uploads, expected 1 of each", len(seen), len(shipper.puts())) } } + +func TestAChunkSpansItsEarliestAndLatestRecordWhenTheClockSteps(t *testing.T) { + log, advance, _ := newTestLog(&fakeShipper{}) + grant := testGrant("s1") + + advance(100 * time.Millisecond) + log.record(grant, aRecord("api.github.com")) + advance(-100 * time.Millisecond) + log.record(grant, aRecord("api.github.com")) + advance(50 * time.Millisecond) + log.record(grant, aRecord("api.github.com")) + + spool := log.spools["s1"] + chunk, err := spool.sealSlice(spool.ring.drain(3), []byte("[]"), 0) + if err != nil { + t.Fatal(err) + } + if chunk.meta.StartedAt != "2026-09-16T10:00:00Z" || chunk.meta.EndedAt != "2026-09-16T10:00:00.1Z" { + t.Fatalf("the chunk spans %s to %s, expected the earliest and latest record", chunk.meta.StartedAt, chunk.meta.EndedAt) + } + if chunk.meta.FirstSeq != 0 || chunk.meta.LastSeq != 2 { + t.Fatalf("the chunk spans seq %d to %d, expected the first and last record", chunk.meta.FirstSeq, chunk.meta.LastSeq) + } +} From 047d920d7c16f195a1b9c53cb6b6f3af4d925d5f Mon Sep 17 00:00:00 2001 From: Saif Ur Rahman Date: Mon, 28 Sep 2026 08:43:39 +0530 Subject: [PATCH 32/43] improvement(agent-vault): name S3's error code when the bucket refuses a session log upload --- packages/agentvault/session_log_ship.go | 22 +++++++++++++ packages/agentvault/session_log_ship_test.go | 33 ++++++++++++++++++++ 2 files changed, 55 insertions(+) diff --git a/packages/agentvault/session_log_ship.go b/packages/agentvault/session_log_ship.go index dedd54fec..ca64b6b08 100644 --- a/packages/agentvault/session_log_ship.go +++ b/packages/agentvault/session_log_ship.go @@ -3,10 +3,13 @@ package agentvault import ( "bytes" "context" + "encoding/xml" "errors" "fmt" + "io" "net/http" "net/url" + "regexp" "strconv" "github.com/Infisical/infisical-merge/packages/api" @@ -112,7 +115,26 @@ func (c *sessionLogShipperClient) putObject(ctx context.Context, uploadURL strin return nil } if res.StatusCode < 200 || res.StatusCode >= 300 { + if code := s3ErrorCode(res.Body); code != "" { + return fmt.Errorf("agent-vault: the bucket refused the upload with status %d (%s)", res.StatusCode, code) + } return fmt.Errorf("agent-vault: the bucket refused the upload with status %d", res.StatusCode) } return nil } + +var s3ErrorCodePattern = regexp.MustCompile(`^[A-Za-z0-9]{1,64}$`) + +// Only the code: S3's Message can quote the signed request. +func s3ErrorCode(body io.Reader) string { + var parsed struct { + Code string `xml:"Code"` + } + if err := xml.NewDecoder(io.LimitReader(body, 4<<10)).Decode(&parsed); err != nil { + return "" + } + if !s3ErrorCodePattern.MatchString(parsed.Code) { + return "" + } + return parsed.Code +} diff --git a/packages/agentvault/session_log_ship_test.go b/packages/agentvault/session_log_ship_test.go index f19d8a7d4..671033a64 100644 --- a/packages/agentvault/session_log_ship_test.go +++ b/packages/agentvault/session_log_ship_test.go @@ -124,6 +124,39 @@ func TestABucketRefusalIsAnErrorThatNamesNoURL(t *testing.T) { if got := err.Error(); strings.Contains(got, "X-Amz-Signature") || strings.Contains(got, bucket.URL) { t.Fatalf("the error names the signed url: %q", got) } + if got := err.Error(); !strings.HasSuffix(got, "status 403 (AccessDenied)") { + t.Fatalf("the error does not name S3's error code: %q", got) + } +} + +func TestABucketRefusalWithoutAnS3ErrorBodyStillNamesTheStatus(t *testing.T) { + for name, body := range map[string]string{ + "not xml": "upstream connect error", + "code too long": "" + strings.Repeat("A", 65) + "", + "odd code": "Access Denied?X-Amz-Signature=secret", + } { + t.Run(name, func(t *testing.T) { + bucket := httptest.NewTLSServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { + w.WriteHeader(http.StatusBadGateway) + _, _ = w.Write([]byte(body)) + })) + defer bucket.Close() + + shipper, err := newSessionLogShipper(func() string { return "proxy-token" }) + if err != nil { + t.Fatal(err) + } + shipper.put.Transport = bucket.Client().Transport + + err = shipper.putObject(context.Background(), bucket.URL+"/object", []byte("bytes")) + if err == nil { + t.Fatal("a 502 from the bucket was treated as a successful upload") + } + if got := err.Error(); !strings.HasSuffix(got, "status 502") { + t.Fatalf("the error should end at the status, got %q", got) + } + }) + } } func TestAnUploadLinkThatIsNotHttpsIsRefused(t *testing.T) { From 8b1c1f590795ab1cfbafb813ed00d5e00a1274ea Mon Sep 17 00:00:00 2001 From: Saif Ur Rahman Date: Mon, 28 Sep 2026 08:43:41 +0530 Subject: [PATCH 33/43] fix(agent-vault): treat only Infisical's NotFound as a gone session, and reset the upload breakers only on the minute tick --- packages/agentvault/cache.go | 12 ++-- packages/agentvault/cache_test.go | 23 ++++---- packages/agentvault/session_log.go | 42 +++++++++++--- packages/agentvault/session_log_test.go | 77 ++++++++++++++++++++++++- 4 files changed, 130 insertions(+), 24 deletions(-) diff --git a/packages/agentvault/cache.go b/packages/agentvault/cache.go index f68d35048..f32b086f4 100644 --- a/packages/agentvault/cache.go +++ b/packages/agentvault/cache.go @@ -138,17 +138,21 @@ func isProxyTokenRejected(err error) bool { return errors.As(err, &apiErr) && apiErr.Name == proxyTokenRejectedName } +// The name on Infisical's own 404s. A route miss or a middlebox answers 404 under another name or none. +const infisicalNotFoundName = "NotFound" + // Resolve answers 200, 401 or 404 by contract, and 401 with a name when the proxy's own token is the -// problem. Only those two statuses are a verdict on the session; anything else from a 4xx is a proxy-side -// fault or a middlebox, and the heartbeat classifier reads a 4xx the same way, so the two agree. A -// rejected proxy token is not a verdict on the session and is reported separately. +// problem. Only a 401 or a 404 named NotFound is a verdict on the session; any other 4xx, an unnamed 404 +// included, is a proxy-side fault or a middlebox, and the heartbeat classifier reads a 4xx the same way, so +// the two agree. A rejected proxy token is not a verdict on the session and is reported separately. func isSessionGone(err error) bool { var apiErr *api.APIError if errors.As(err, &apiErr) { if isProxyTokenRejected(err) { return false } - return apiErr.StatusCode == http.StatusUnauthorized || apiErr.StatusCode == http.StatusNotFound + return apiErr.StatusCode == http.StatusUnauthorized || + (apiErr.StatusCode == http.StatusNotFound && apiErr.Name == infisicalNotFoundName) } return errors.Is(err, errSessionGone) } diff --git a/packages/agentvault/cache_test.go b/packages/agentvault/cache_test.go index c9265c881..a01a2c822 100644 --- a/packages/agentvault/cache_test.go +++ b/packages/agentvault/cache_test.go @@ -97,7 +97,7 @@ func TestRefreshDropsAGoneSessionImmediately(t *testing.T) { t.Fatalf("get: %v", err) } - resolver.err = &api.APIError{StatusCode: status} + resolver.err = &api.APIError{StatusCode: status, Name: map[int]string{404: infisicalNotFoundName}[status]} cache.refresh() if len(cache.entries) != 0 { @@ -198,27 +198,30 @@ func TestStaleEntryIsNotServedFromTheCache(t *testing.T) { } } -// Only the statuses resolve answers by contract end a session. A 400, 403 or 422 cannot come from resolve -// itself, so it is a proxy-side fault or a middlebox and rides the grace window like an outage. +// Only the refusals resolve answers by contract end a session. A 400, 403 or 422, or a 404 that is not +// Infisical's NotFound, cannot come from resolve itself, so it is a proxy-side fault or a middlebox and rides +// the grace window like an outage. func TestRefreshTreatsOnlyTheContractRefusalsAsTerminal(t *testing.T) { for _, tc := range []struct { status int + name string kept bool }{ - {401, false}, {404, false}, - {400, true}, {403, true}, {405, true}, {407, true}, {422, true}, - {408, true}, {429, true}, {500, true}, {502, true}, + {401, "", false}, {404, infisicalNotFoundName, false}, + {404, "", true}, {404, "Not Found", true}, + {400, "", true}, {403, "", true}, {405, "", true}, {407, "", true}, {422, "", true}, + {408, "", true}, {429, "", true}, {500, "", true}, {502, "", true}, } { resolver := &stubResolver{result: &resolveResult{SessionID: "s1", Services: []*resolvedService{serviceWithSecret("v")}}} cache := newTestCache(resolver) if _, err := cache.get("tok"); err != nil { t.Fatal(err) } - resolver.err = &api.APIError{StatusCode: tc.status} + resolver.err = &api.APIError{StatusCode: tc.status, Name: tc.name} cache.refresh() _, kept := cache.get("tok") if (kept == nil) != tc.kept { - t.Fatalf("status %d: credential still served = %v, want %v", tc.status, kept == nil, tc.kept) + t.Fatalf("status %d %q: credential still served = %v, want %v", tc.status, tc.name, kept == nil, tc.kept) } } } @@ -307,7 +310,7 @@ func TestOnlyOneRefreshRunsAtATime(t *testing.T) { } func TestADefinitiveRefusalIsNotReResolvedEveryRequest(t *testing.T) { - resolver := &stubResolver{err: &api.APIError{StatusCode: 404, Name: "NotFound"}} + resolver := &stubResolver{err: &api.APIError{StatusCode: 404, Name: infisicalNotFoundName}} cache := newTestCache(resolver) for i := 0; i < 20; i++ { @@ -345,7 +348,7 @@ func TestAnOutageIsNotRememberedAsARefusal(t *testing.T) { } func TestARefusalExpires(t *testing.T) { - resolver := &stubResolver{err: &api.APIError{StatusCode: 404}} + resolver := &stubResolver{err: &api.APIError{StatusCode: 404, Name: infisicalNotFoundName}} cache := newTestCache(resolver) _, _ = cache.get("agv_dead") cache.mu.Lock() diff --git a/packages/agentvault/session_log.go b/packages/agentvault/session_log.go index 6997fdd13..5f7ecba91 100644 --- a/packages/agentvault/session_log.go +++ b/packages/agentvault/session_log.go @@ -175,7 +175,7 @@ func (a *sessionLogRecorder) run(stop <-chan struct{}) { case <-ticker.C: a.flushAll(context.Background(), false) case <-a.wake: - a.flushAll(context.Background(), false) + a.flush(context.Background(), flushWake) } } } @@ -204,12 +204,19 @@ func (a *sessionLogRecorder) close(ctx context.Context) { } } -func (a *sessionLogRecorder) dueSpools(final bool, now time.Time) []*sessionLogSpool { +func (a *sessionLogRecorder) dueSpools(pass flushPass, now time.Time) []*sessionLogSpool { a.mu.Lock() defer a.mu.Unlock() + final := pass == flushFinal due := make([]*sessionLogSpool, 0, len(a.spools)) for id, spool := range a.spools { + if pass == flushWake { + if spool.ring.len() >= sessionLogFlushRecords { + due = append(due, spool) + } + continue + } if spool.ring.len() == 0 && len(spool.pending) == 0 { if !final && now.Sub(spool.lastRecordAt) > sessionLogIdleClose { a.forgetSpoolLocked(id, spool) @@ -278,7 +285,23 @@ func (a *sessionLogRecorder) switchOff(grantsIssuedAtSend uint64) { a.switchedOffThrough = grantsIssuedAtSend } +type flushPass int + +const ( + flushTick flushPass = iota + flushWake + flushFinal +) + func (a *sessionLogRecorder) flushAll(ctx context.Context, final bool) { + pass := flushTick + if final { + pass = flushFinal + } + a.flush(ctx, pass) +} + +func (a *sessionLogRecorder) flush(ctx context.Context, pass flushPass) { if a == nil { return } @@ -286,14 +309,19 @@ func (a *sessionLogRecorder) flushAll(ctx context.Context, final bool) { a.flushMu.Lock() defer a.flushMu.Unlock() - a.mu.Lock() - a.s3Down = false - a.infisicalDown = false - a.mu.Unlock() + // A busy session wakes the loop every thousand records, so resetting here would retry a down bucket or + // Infisical that often instead of once a minute. + if pass != flushWake { + a.mu.Lock() + a.s3Down = false + a.infisicalDown = false + a.mu.Unlock() + } + final := pass == flushFinal // One start time for every spool, so shipping the earlier ones doesn't make the later ones late for the next tick. started := a.now() - for _, spool := range a.dueSpools(final, started) { + for _, spool := range a.dueSpools(pass, started) { a.flushSpool(ctx, spool, final, started) } } diff --git a/packages/agentvault/session_log_test.go b/packages/agentvault/session_log_test.go index c90658676..ce30d8fcc 100644 --- a/packages/agentvault/session_log_test.go +++ b/packages/agentvault/session_log_test.go @@ -325,7 +325,7 @@ func TestAFailedUploadRePostsTheSameChunkID(t *testing.T) { } func TestASessionThatIsGoneIsDropped(t *testing.T) { - shipper := &fakeShipper{postResults: []scriptedResult{{err: apiErr(http.StatusNotFound, "")}}} + shipper := &fakeShipper{postResults: []scriptedResult{{err: apiErr(http.StatusNotFound, infisicalNotFoundName)}}} log, _, tick := newTestLog(shipper) log.record(testGrant("s1"), aRecord("api.github.com")) @@ -341,6 +341,28 @@ func TestASessionThatIsGoneIsDropped(t *testing.T) { } } +func TestA404ThatIsNotInfisicalsNotFoundDropsOnlyThatChunk(t *testing.T) { + shipper := &fakeShipper{postResults: []scriptedResult{{err: apiErr(http.StatusNotFound, "")}}} + log, _, tick := newTestLog(shipper) + + log.record(testGrant("s1"), aRecord("api.github.com")) + tick() + + spool, ok := log.spools["s1"] + if !ok { + t.Fatal("an unnamed 404 dropped the whole session") + } + if len(spool.pending) != 0 || spool.ring.dropped != 1 { + t.Fatalf("the refused chunk was not dropped and counted (pending %d, dropped %d)", len(spool.pending), spool.ring.dropped) + } + + log.record(testGrant("s1"), aRecord("api.github.com")) + tick() + if len(shipper.puts()) != 1 { + t.Fatal("the session stopped shipping after an unnamed 404") + } +} + func TestARejectedProxyTokenKeepsEverything(t *testing.T) { shipper := &fakeShipper{postDefault: scriptedResult{err: apiErr(http.StatusUnauthorized, proxyTokenRejectedName)}} log, _, tick := newTestLog(shipper) @@ -816,7 +838,7 @@ func TestSequenceNumbersSurviveASpoolBeingForgotten(t *testing.T) { } func TestASessionThatIsGoneDoesNotReserveItsSequenceNumbers(t *testing.T) { - shipper := &fakeShipper{postResults: []scriptedResult{{err: apiErr(http.StatusNotFound, "")}}} + shipper := &fakeShipper{postResults: []scriptedResult{{err: apiErr(http.StatusNotFound, infisicalNotFoundName)}}} log, _, tick := newTestLog(shipper) log.record(testGrant("s1"), aRecord("api.github.com")) @@ -948,7 +970,7 @@ func TestEveryWayAChunkLeavesReleasesWhatItHeld(t *testing.T) { }{ {"uploaded", scriptedResult{}}, {"refused as bad", scriptedResult{err: apiErr(http.StatusUnprocessableEntity, "")}}, - {"session gone", scriptedResult{err: apiErr(http.StatusNotFound, "")}}, + {"session gone", scriptedResult{err: apiErr(http.StatusNotFound, infisicalNotFoundName)}}, } { shipper := &fakeShipper{postResults: []scriptedResult{outcome.post}} log, _, tick := newTestLog(shipper) @@ -1111,3 +1133,52 @@ func TestAChunkSpansItsEarliestAndLatestRecordWhenTheClockSteps(t *testing.T) { t.Fatalf("the chunk spans seq %d to %d, expected the first and last record", chunk.meta.FirstSeq, chunk.meta.LastSeq) } } + +func TestAWakeDoesNotResetTheBreakers(t *testing.T) { + shipper := &fakeShipper{putDefault: errors.New("bucket down")} + log, _, tick := newTestLog(shipper) + + log.record(testGrant("s1"), aRecord("api.github.com")) + tick() + if !log.s3Down { + t.Fatal("a failed upload did not trip the breaker, so this test proves nothing") + } + putsAfterTick := len(shipper.puts()) + + for i := 0; i < sessionLogFlushRecords; i++ { + log.record(testGrant("s1"), aRecord("api.github.com")) + } + log.flush(context.Background(), flushWake) + + if !log.s3Down { + t.Fatal("a wake reset the breaker") + } + if got := len(shipper.puts()); got != putsAfterTick { + t.Fatalf("a wake retried the bucket %d times while it was down", got-putsAfterTick) + } + + tick() + if got := len(shipper.puts()); got == putsAfterTick { + t.Fatal("the next tick did not retry the bucket") + } +} + +func TestAWakeShipsOnlyFullRings(t *testing.T) { + shipper := &fakeShipper{} + log, advance, _ := newTestLog(shipper) + + log.record(testGrant("s1"), aRecord("api.github.com")) + for i := 0; i < sessionLogFlushRecords; i++ { + log.record(testGrant("s2"), aRecord("api.github.com")) + } + advance(sessionLogFlushInterval) + log.flush(context.Background(), flushWake) + + posts := shipper.posts() + if len(posts) != 1 || posts[0].sessionID != "s2" { + t.Fatalf("a wake shipped %d chunks, expected only the full ring of s2", len(posts)) + } + if got := log.spools["s1"].ring.len(); got != 1 { + t.Fatalf("s1 holds %d records after the wake, expected it to wait for the tick", got) + } +} From 46bb280bb78fb6967b0930c6e162f688489a4708 Mon Sep 17 00:00:00 2001 From: Saif Ur Rahman Date: Mon, 28 Sep 2026 10:16:41 +0530 Subject: [PATCH 34/43] fix(agent-vault): retry a session log chunk on a 404 that isn't Infisical's NotFound --- packages/agentvault/session_log_ship.go | 4 +++- packages/agentvault/session_log_test.go | 10 +++++----- 2 files changed, 8 insertions(+), 6 deletions(-) diff --git a/packages/agentvault/session_log_ship.go b/packages/agentvault/session_log_ship.go index ca64b6b08..dd092042c 100644 --- a/packages/agentvault/session_log_ship.go +++ b/packages/agentvault/session_log_ship.go @@ -27,7 +27,9 @@ func isPoisonChunk(err error) bool { if !errors.As(err, &apiErr) { return false } - if apiErr.StatusCode == http.StatusRequestTimeout || apiErr.StatusCode == http.StatusTooManyRequests { + // Infisical's own NotFound is caught earlier as a gone session, so a 404 here is a route miss, as during a rollback. + if apiErr.StatusCode == http.StatusRequestTimeout || apiErr.StatusCode == http.StatusTooManyRequests || + apiErr.StatusCode == http.StatusNotFound { return false } return apiErr.StatusCode >= 400 && apiErr.StatusCode < 500 diff --git a/packages/agentvault/session_log_test.go b/packages/agentvault/session_log_test.go index ce30d8fcc..f2ec2b22b 100644 --- a/packages/agentvault/session_log_test.go +++ b/packages/agentvault/session_log_test.go @@ -341,7 +341,7 @@ func TestASessionThatIsGoneIsDropped(t *testing.T) { } } -func TestA404ThatIsNotInfisicalsNotFoundDropsOnlyThatChunk(t *testing.T) { +func TestA404ThatIsNotInfisicalsNotFoundIsRetried(t *testing.T) { shipper := &fakeShipper{postResults: []scriptedResult{{err: apiErr(http.StatusNotFound, "")}}} log, _, tick := newTestLog(shipper) @@ -352,14 +352,13 @@ func TestA404ThatIsNotInfisicalsNotFoundDropsOnlyThatChunk(t *testing.T) { if !ok { t.Fatal("an unnamed 404 dropped the whole session") } - if len(spool.pending) != 0 || spool.ring.dropped != 1 { - t.Fatalf("the refused chunk was not dropped and counted (pending %d, dropped %d)", len(spool.pending), spool.ring.dropped) + if len(spool.pending) != 1 || spool.ring.dropped != 0 { + t.Fatalf("the chunk was not kept for a retry (pending %d, dropped %d)", len(spool.pending), spool.ring.dropped) } - log.record(testGrant("s1"), aRecord("api.github.com")) tick() if len(shipper.puts()) != 1 { - t.Fatal("the session stopped shipping after an unnamed 404") + t.Fatal("the chunk was not shipped once the route answered again") } } @@ -995,6 +994,7 @@ func TestServerErrorsAreRetriedAndBadChunksAreDropped(t *testing.T) { {http.StatusServiceUnavailable, true}, {http.StatusTooManyRequests, true}, {http.StatusRequestTimeout, true}, + {http.StatusNotFound, true}, {http.StatusUnprocessableEntity, false}, {http.StatusConflict, false}, } { From 4fcb440cfc2b9ab7a3eeda14dfec7609765618dd Mon Sep 17 00:00:00 2001 From: Saif Ur Rahman Date: Mon, 28 Sep 2026 13:38:37 +0530 Subject: [PATCH 35/43] refactor(agent-vault): make the session log recorder easier to follow --- packages/agentvault/session_log.go | 434 ++++++++++--------- packages/agentvault/session_log_crypto.go | 5 +- packages/agentvault/session_log_ship.go | 52 ++- packages/agentvault/session_log_ship_test.go | 2 +- packages/agentvault/session_log_spool.go | 10 +- packages/agentvault/session_log_test.go | 74 +++- 6 files changed, 329 insertions(+), 248 deletions(-) diff --git a/packages/agentvault/session_log.go b/packages/agentvault/session_log.go index 5f7ecba91..1270be686 100644 --- a/packages/agentvault/session_log.go +++ b/packages/agentvault/session_log.go @@ -28,10 +28,6 @@ const ( sessionLogPutTimeout = 10 * time.Second sessionLogFinalTimeout = 3 * time.Second sessionLogCloseTimeout = 5 * time.Second - - sessionLogCeilingReachedName = "AgentVaultSessionLogCeilingReached" - sessionLogDisabledName = "AgentVaultSessionLogDisabled" - sessionLogClockSkewName = "AgentVaultSessionLogClockSkew" ) type sessionLogGrant struct { @@ -41,7 +37,8 @@ type sessionLogGrant struct { issued uint64 } -// Counts every grant the proxy is handed, so a refusal can be told apart from a key issued after it. +// Resolve only hands out a grant while session logs are on. So a grant issued after a refused request went +// out means logs came back on after that request, and the refusal is stale. var sessionLogGrantsIssued atomic.Uint64 func newSessionLogGrant(sessionID string, key []byte) *sessionLogGrant { @@ -49,8 +46,9 @@ func newSessionLogGrant(sessionID string, key []byte) *sessionLogGrant { } type forgottenSpool struct { - nextSeq uint64 - dropped uint64 + nextSeq uint64 + dropped uint64 + forgottenAt time.Time } type sessionLogShipper interface { @@ -58,12 +56,36 @@ type sessionLogShipper interface { putObject(ctx context.Context, url string, ciphertext []byte) error } +// Being off or paused also drops new records, while an outage only holds them until the next retry. +type sessionLogHold struct { + off bool + offThrough uint64 + + pausedUntil time.Time + + s3Down bool + infisicalDown bool +} + +func (h *sessionLogHold) dropsRecords(now time.Time) bool { + return h.off || now.Before(h.pausedUntil) +} + +func (h *sessionLogHold) canShip(now time.Time) bool { + return !h.dropsRecords(now) && !h.s3Down && !h.infisicalDown +} + +func (h *sessionLogHold) clearOutages() { + h.s3Down = false + h.infisicalDown = false +} + type sessionLogRecorder struct { proxyID string shipper sessionLogShipper now func() time.Time - // close's flushAll can overlap the run loop's; unguarded, one chunk ships twice. + // close's flush can overlap the run loop's; unguarded, one chunk ships twice. flushMu sync.Mutex mu sync.Mutex @@ -75,13 +97,7 @@ type sessionLogRecorder struct { sealedBytes int nextSealOrder uint64 - pauseUntil time.Time - - switchedOff bool - switchedOffThrough uint64 - - s3Down bool - infisicalDown bool + hold sessionLogHold clockSkewReported bool @@ -100,69 +116,69 @@ func newSessionLogRecorder(proxyID string, shipper sessionLogShipper) *sessionLo } } -func (a *sessionLogRecorder) record(g *sessionLogGrant, rec sessionLogRecord) { - if a == nil || g == nil { +func (r *sessionLogRecorder) record(g *sessionLogGrant, rec sessionLogRecord) { + if r == nil || g == nil { return } - a.mu.Lock() - defer a.mu.Unlock() - if a.closed { + r.mu.Lock() + defer r.mu.Unlock() + if r.closed { return } - spool, ok := a.spools[g.sessionID] + spool, ok := r.spools[g.sessionID] if !ok { - spool = newSessionLogSpool(g, a.now()) - if prior, ok := a.forgotten[g.sessionID]; ok { + spool = newSessionLogSpool(g, r.now()) + if prior, ok := r.forgotten[g.sessionID]; ok { spool.nextSeq, spool.ring.dropped = prior.nextSeq, prior.dropped - delete(a.forgotten, g.sessionID) + delete(r.forgotten, g.sessionID) } - a.spools[g.sessionID] = spool + r.spools[g.sessionID] = spool } rec.Seq = spool.nextSeq spool.nextSeq++ - rec.ProxyID = a.proxyID - now := a.now() + rec.ProxyID = r.proxyID + now := r.now() // Without the monotonic reading, Before and After compare wall time, which is what Ts shows. rec.at = now.Round(0) rec.Ts = rec.at.UTC().Format(time.RFC3339Nano) spool.lastRecordAt = now - if a.switchedOff { - if g.issued <= a.switchedOffThrough { + if r.hold.off { + if g.issued <= r.hold.offThrough { spool.ring.dropped++ return } - a.switchedOff = false + r.hold.off = false log.Info().Msg("agent-vault: session logs are back on, recording again") } - if !a.pauseUntil.IsZero() && a.now().Before(a.pauseUntil) { + if now.Before(r.hold.pausedUntil) { spool.ring.dropped++ return } - if a.total >= sessionLogTotalCapacity { + if r.total >= sessionLogTotalCapacity { spool.ring.dropped++ return } if evicted := spool.ring.push(rec); !evicted { - a.total++ + r.total++ } if spool.ring.len() >= sessionLogFlushRecords { select { - case a.wake <- struct{}{}: + case r.wake <- struct{}{}: default: } } } -func (a *sessionLogRecorder) run(stop <-chan struct{}) { - if a == nil { +func (r *sessionLogRecorder) run(stop <-chan struct{}) { + if r == nil { return } ticker := time.NewTicker(sessionLogFlushInterval) @@ -173,44 +189,41 @@ func (a *sessionLogRecorder) run(stop <-chan struct{}) { case <-stop: return case <-ticker.C: - a.flushAll(context.Background(), false) - case <-a.wake: - a.flush(context.Background(), flushWake) + r.flush(context.Background(), flushTick) + case <-r.wake: + r.flush(context.Background(), flushWake) } } } -func (a *sessionLogRecorder) close(ctx context.Context) { - if a == nil { +func (r *sessionLogRecorder) close(ctx context.Context) { + if r == nil { return } - a.mu.Lock() - a.closed = true - a.mu.Unlock() + r.mu.Lock() + r.closed = true + r.mu.Unlock() - a.flushAll(ctx, true) + r.flush(ctx, flushFinal) - a.mu.Lock() - defer a.mu.Unlock() + r.mu.Lock() + defer r.mu.Unlock() var lost int - for _, spool := range a.spools { - lost += spool.ring.len() - for _, chunk := range spool.pending { - lost += chunk.meta.RecordCount - } + for _, spool := range r.spools { + lost += spool.heldRecords() } if lost > 0 { log.Warn().Int("records", lost).Msg("agent-vault: session log records were not shipped before shutdown") } } -func (a *sessionLogRecorder) dueSpools(pass flushPass, now time.Time) []*sessionLogSpool { - a.mu.Lock() - defer a.mu.Unlock() +func (r *sessionLogRecorder) dueSpools(pass flushPass, now time.Time) []*sessionLogSpool { + r.mu.Lock() + defer r.mu.Unlock() final := pass == flushFinal - due := make([]*sessionLogSpool, 0, len(a.spools)) - for id, spool := range a.spools { + due := make([]*sessionLogSpool, 0, len(r.spools)) + for id, spool := range r.spools { if pass == flushWake { if spool.ring.len() >= sessionLogFlushRecords { due = append(due, spool) @@ -219,7 +232,7 @@ func (a *sessionLogRecorder) dueSpools(pass flushPass, now time.Time) []*session } if spool.ring.len() == 0 && len(spool.pending) == 0 { if !final && now.Sub(spool.lastRecordAt) > sessionLogIdleClose { - a.forgetSpoolLocked(id, spool) + r.forgetSpoolLocked(id, spool) } continue } @@ -233,56 +246,85 @@ func (a *sessionLogRecorder) dueSpools(pass flushPass, now time.Time) []*session } // Keeps the drop count too, so a spool forgotten while logging was off still reports what it lost. -func (a *sessionLogRecorder) forgetSpoolLocked(sessionID string, spool *sessionLogSpool) { - if len(a.forgotten) >= maxSessionCacheEntries { - a.forgotten = make(map[string]forgottenSpool) +func (r *sessionLogRecorder) forgetSpoolLocked(sessionID string, spool *sessionLogSpool) { + if len(r.forgotten) >= maxSessionCacheEntries { + r.evictLongestForgottenLocked() } - a.forgotten[sessionID] = forgottenSpool{nextSeq: spool.nextSeq, dropped: spool.ring.dropped} - delete(a.spools, sessionID) + r.forgotten[sessionID] = forgottenSpool{nextSeq: spool.nextSeq, dropped: spool.ring.dropped, forgottenAt: r.now()} + delete(r.spools, sessionID) } -func (a *sessionLogRecorder) holding() bool { - a.mu.Lock() - defer a.mu.Unlock() - return a.switchedOff || (!a.pauseUntil.IsZero() && a.now().Before(a.pauseUntil)) +func (r *sessionLogRecorder) evictLongestForgottenLocked() { + var oldestID string + var oldestAt time.Time + found := false + for id, prior := range r.forgotten { + if !found || prior.forgottenAt.Before(oldestAt) { + oldestID, oldestAt, found = id, prior.forgottenAt, true + } + } + delete(r.forgotten, oldestID) } -func (a *sessionLogRecorder) pause() { - a.mu.Lock() - defer a.mu.Unlock() - a.pauseUntil = a.now().Add(sessionLogPauseBackoff) +func (r *sessionLogRecorder) pause() { + r.mu.Lock() + defer r.mu.Unlock() + r.hold.pausedUntil = r.now().Add(sessionLogPauseBackoff) } -// Drops everything held and counts it. A key issued after the refused request went out means logging may already -// be back on, so then the refusal is stale and nothing is dropped. -func (a *sessionLogRecorder) switchOff(grantsIssuedAtSend uint64) { - a.mu.Lock() - defer a.mu.Unlock() +// Drops everything held and counts it, unless a grant was issued after the refused request went out, which +// makes the refusal stale (see sessionLogGrantsIssued). +func (r *sessionLogRecorder) switchOff(grantsIssuedAtSend uint64) { + r.mu.Lock() + defer r.mu.Unlock() if sessionLogGrantsIssued.Load() > grantsIssuedAtSend { return } var lost int - for _, spool := range a.spools { - held := spool.ring.len() - spool.ring.drain(held) + for _, spool := range r.spools { + lost += r.discardAllLocked(spool, true) + } + + if !r.hold.off { + log.Warn().Int("records", lost). + Msg("agent-vault: session logs are off for this project, dropping what was held until they are back on") + } + r.hold.off = true + r.hold.offThrough = grantsIssuedAtSend +} + +// A gone session passes countDropped false: its spool is deleted, so no later chunk could report the loss. +func (r *sessionLogRecorder) discardAllLocked(spool *sessionLogSpool, countDropped bool) (lost int) { + lost = spool.heldRecords() + + held := spool.ring.len() + spool.ring.drain(held) + r.total -= held + if countDropped { spool.ring.dropped += uint64(held) - a.total -= held - lost += held - for _, chunk := range spool.pending { + } + + for _, chunk := range spool.pending { + r.sealedBytes -= len(chunk.ciphertext) + if countDropped { spool.ring.dropped += chunk.lostCount() - a.sealedBytes -= len(chunk.ciphertext) - lost += chunk.meta.RecordCount } - spool.pending = nil } + spool.pending = nil + return lost +} - if !a.switchedOff { - log.Warn().Int("records", lost). - Msg("agent-vault: session logs are off for this project, dropping what was held until they are back on") +func (r *sessionLogRecorder) discardHeadLocked(spool *sessionLogSpool, chunk *sealedChunk, countDropped bool) bool { + if len(spool.pending) == 0 || spool.pending[0] != chunk { + return false + } + spool.popPending() + r.sealedBytes -= len(chunk.ciphertext) + if countDropped { + spool.ring.dropped += chunk.lostCount() } - a.switchedOff = true - a.switchedOffThrough = grantsIssuedAtSend + return true } type flushPass int @@ -293,55 +335,45 @@ const ( flushFinal ) -func (a *sessionLogRecorder) flushAll(ctx context.Context, final bool) { - pass := flushTick - if final { - pass = flushFinal - } - a.flush(ctx, pass) -} - -func (a *sessionLogRecorder) flush(ctx context.Context, pass flushPass) { - if a == nil { +func (r *sessionLogRecorder) flush(ctx context.Context, pass flushPass) { + if r == nil { return } - a.flushMu.Lock() - defer a.flushMu.Unlock() + r.flushMu.Lock() + defer r.flushMu.Unlock() // A busy session wakes the loop every thousand records, so resetting here would retry a down bucket or // Infisical that often instead of once a minute. if pass != flushWake { - a.mu.Lock() - a.s3Down = false - a.infisicalDown = false - a.mu.Unlock() + r.mu.Lock() + r.hold.clearOutages() + r.mu.Unlock() } - final := pass == flushFinal // One start time for every spool, so shipping the earlier ones doesn't make the later ones late for the next tick. - started := a.now() - for _, spool := range a.dueSpools(pass, started) { - a.flushSpool(ctx, spool, final, started) + started := r.now() + for _, spool := range r.dueSpools(pass, started) { + r.flushSpool(ctx, spool, pass, started) } } -func (a *sessionLogRecorder) sealRing(spool *sessionLogSpool, started time.Time) { +func (r *sessionLogRecorder) sealRing(spool *sessionLogSpool, started time.Time) { for { - a.mu.Lock() + r.mu.Lock() records := spool.ring.drain(sessionLogFlushRecords) if len(records) == 0 { spool.lastFlushAt = started - a.mu.Unlock() + r.mu.Unlock() return } - a.total -= len(records) + r.total -= len(records) dropped := spool.ring.takeDropped() - a.mu.Unlock() + r.mu.Unlock() groups, err := packSessionLogRecords(records) if err != nil { - a.dropUnsealed(spool, len(records), dropped, err) + r.dropUnsealed(spool, len(records), dropped, err) continue } for i, group := range groups { @@ -352,45 +384,45 @@ func (a *sessionLogRecorder) sealRing(spool *sessionLogSpool, started time.Time) chunk, err := spool.sealSlice(group.records, group.plaintext, groupDropped) if err != nil { - a.dropUnsealed(spool, len(group.records), groupDropped, err) + r.dropUnsealed(spool, len(group.records), groupDropped, err) continue } - a.mu.Lock() - chunk.sealOrder = a.nextSealOrder - a.nextSealOrder++ + r.mu.Lock() + chunk.sealOrder = r.nextSealOrder + r.nextSealOrder++ spool.pending = append(spool.pending, chunk) - a.sealedBytes += len(chunk.ciphertext) - a.enforcePendingCapsLocked(spool) - a.mu.Unlock() + r.sealedBytes += len(chunk.ciphertext) + r.enforcePendingCapsLocked(spool) + r.mu.Unlock() } } } -func (a *sessionLogRecorder) dropUnsealed(spool *sessionLogSpool, records int, dropped uint64, err error) { - a.mu.Lock() +func (r *sessionLogRecorder) dropUnsealed(spool *sessionLogSpool, records int, dropped uint64, err error) { + r.mu.Lock() spool.ring.dropped += dropped + uint64(records) - a.mu.Unlock() + r.mu.Unlock() log.Error().Err(err).Str("sessionId", spool.sessionID).Int("records", records). Msg("agent-vault: could not seal a session log chunk, dropping those records") } -func (a *sessionLogRecorder) enforcePendingCapsLocked(spool *sessionLogSpool) { +func (r *sessionLogRecorder) enforcePendingCapsLocked(spool *sessionLogSpool) { for len(spool.pending) > sessionLogPendingChunks { - a.evictOldestLocked(spool) + r.evictOldestLocked(spool) } - for a.sealedBytes > sessionLogTotalSealedBytes { - victim := a.oldestPendingLocked() + for r.sealedBytes > sessionLogTotalSealedBytes { + victim := r.oldestPendingLocked() if victim == nil { return } - a.evictOldestLocked(victim) + r.evictOldestLocked(victim) } } -func (a *sessionLogRecorder) oldestPendingLocked() *sessionLogSpool { +func (r *sessionLogRecorder) oldestPendingLocked() *sessionLogSpool { var oldest *sessionLogSpool - for _, spool := range a.spools { + for _, spool := range r.spools { if len(spool.pending) == 0 { continue } @@ -401,10 +433,9 @@ func (a *sessionLogRecorder) oldestPendingLocked() *sessionLogSpool { return oldest } -func (a *sessionLogRecorder) evictOldestLocked(spool *sessionLogSpool) { - oldest := spool.popPending() - a.sealedBytes -= len(oldest.ciphertext) - spool.ring.dropped += oldest.lostCount() +func (r *sessionLogRecorder) evictOldestLocked(spool *sessionLogSpool) { + oldest := spool.pending[0] + r.discardHeadLocked(spool, oldest, true) log.Warn(). Str("sessionId", spool.sessionID). Str("chunkId", oldest.meta.ChunkID). @@ -412,51 +443,46 @@ func (a *sessionLogRecorder) evictOldestLocked(spool *sessionLogSpool) { Msg("agent-vault: dropped an unshipped session log chunk, the buffer is full") } -func (a *sessionLogRecorder) flushSpool(ctx context.Context, spool *sessionLogSpool, final bool, started time.Time) { - a.sealRing(spool, started) - if a.holding() { - return - } +func (r *sessionLogRecorder) flushSpool(ctx context.Context, spool *sessionLogSpool, pass flushPass, started time.Time) { + r.sealRing(spool, started) for { - a.mu.Lock() - if len(spool.pending) == 0 || a.s3Down || a.infisicalDown { - a.mu.Unlock() + r.mu.Lock() + if len(spool.pending) == 0 || !r.hold.canShip(r.now()) { + r.mu.Unlock() return } chunk := spool.pending[0] - a.mu.Unlock() + r.mu.Unlock() - if !a.shipChunk(ctx, spool, chunk, final) { + if !r.shipChunk(ctx, spool, chunk, pass) { return } - a.mu.Lock() - if len(spool.pending) > 0 && spool.pending[0] == chunk { - spool.popPending() - a.sealedBytes -= len(chunk.ciphertext) - } - a.mu.Unlock() + r.mu.Lock() + r.discardHeadLocked(spool, chunk, false) + r.mu.Unlock() } } -func (a *sessionLogRecorder) shipChunk(ctx context.Context, spool *sessionLogSpool, chunk *sealedChunk, final bool) bool { - if chunk.uploadURL == "" || a.now().Add(10*time.Second).After(chunk.urlExpires) { +func (r *sessionLogRecorder) shipChunk(ctx context.Context, spool *sessionLogSpool, chunk *sealedChunk, pass flushPass) bool { + final := pass == flushFinal + if chunk.uploadURL == "" || r.now().Add(10*time.Second).After(chunk.urlExpires) { // Past the shutdown budget, a new row could only be written for an upload that can no longer happen. if ctx.Err() != nil { return false } grantsIssuedAtSend := sessionLogGrantsIssued.Load() - res, err := a.shipper.createChunk(ctx, final, spool.sessionID, chunk.meta) + res, err := r.shipper.createChunk(ctx, final, spool.sessionID, chunk.meta) if err != nil { - return a.handleCreateFailure(spool, chunk, err, grantsIssuedAtSend) + return r.handleCreateFailure(spool, chunk, err, grantsIssuedAtSend) } - a.mu.Lock() - a.clockSkewReported = false - a.mu.Unlock() + r.mu.Lock() + r.clockSkewReported = false + r.mu.Unlock() chunk.posted = true chunk.uploadURL = res.UploadURL - chunk.urlExpires = a.now().Add(time.Duration(res.ExpiresInSeconds) * time.Second) + chunk.urlExpires = r.now().Add(time.Duration(res.ExpiresInSeconds) * time.Second) } putCtx := ctx @@ -466,11 +492,11 @@ func (a *sessionLogRecorder) shipChunk(ctx context.Context, spool *sessionLogSpo defer cancel() } - if err := a.shipper.putObject(putCtx, chunk.uploadURL, chunk.ciphertext); err != nil { + if err := r.shipper.putObject(putCtx, chunk.uploadURL, chunk.ciphertext); err != nil { chunk.uploadURL = "" - a.mu.Lock() - a.s3Down = true - a.mu.Unlock() + r.mu.Lock() + r.hold.s3Down = true + r.mu.Unlock() log.Warn().Err(err).Str("sessionId", spool.sessionID).Str("chunkId", chunk.meta.ChunkID). Msg("agent-vault: could not upload a session log chunk, will retry") return false @@ -479,70 +505,54 @@ func (a *sessionLogRecorder) shipChunk(ctx context.Context, spool *sessionLogSpo return true } -func (a *sessionLogRecorder) handleCreateFailure(spool *sessionLogSpool, chunk *sealedChunk, err error, grantsIssuedAtSend uint64) bool { - switch { - case isProxyTokenRejected(err): +func (r *sessionLogRecorder) handleCreateFailure(spool *sessionLogSpool, chunk *sealedChunk, err error, grantsIssuedAtSend uint64) bool { + switch classifyChunkError(err) { + case chunkTokenRejected: log.Warn().Err(err).Msg("agent-vault: Infisical rejected this proxy's token, holding session logs") - return false - case isSessionGone(err): - a.mu.Lock() - lost := spool.ring.len() - a.total -= spool.ring.len() - for _, held := range spool.pending { - lost += held.meta.RecordCount - a.sealedBytes -= len(held.ciphertext) - } - a.forgetSpoolLocked(spool.sessionID, spool) - delete(a.forgotten, spool.sessionID) - a.mu.Unlock() + case chunkSessionGone: + r.mu.Lock() + lost := r.discardAllLocked(spool, false) + delete(r.spools, spool.sessionID) + r.mu.Unlock() log.Warn().Err(err).Str("sessionId", spool.sessionID).Int("records", lost). Msg("agent-vault: Infisical no longer accepts session logs for this session, dropping what was held") - return false - case isSessionLogErrorNamed(err, sessionLogCeilingReachedName): - a.pause() + case chunkOrgFull: + r.pause() log.Error().Err(err).Msg("agent-vault: session logs have reached their limit for this organization, retrying in 15m") - return false - case isSessionLogErrorNamed(err, sessionLogDisabledName): - a.switchOff(grantsIssuedAtSend) - return false + case chunkLoggingOff: + r.switchOff(grantsIssuedAtSend) - case isSessionLogErrorNamed(err, sessionLogClockSkewName): - a.dropRefused(spool, chunk) - a.mu.Lock() - reported := a.clockSkewReported - a.clockSkewReported = true - a.mu.Unlock() + case chunkClockSkew: + r.dropRefused(spool, chunk) + r.mu.Lock() + reported := r.clockSkewReported + r.clockSkewReported = true + r.mu.Unlock() if !reported { log.Error().Err(err).Msg("agent-vault: session logs are being refused because this machine's clock is wrong; fix the clock to resume recording") } - return false - case isPoisonChunk(err): - a.dropRefused(spool, chunk) + case chunkRefused: + r.dropRefused(spool, chunk) log.Error().Err(err).Str("chunkId", chunk.meta.ChunkID).Int("records", chunk.meta.RecordCount). - Msg("agent-vault: Infisical rejected a session log chunk as malformed, dropping it") - return false + Msg("agent-vault: Infisical refused a session log chunk, dropping it") - default: - a.mu.Lock() - a.infisicalDown = true - a.mu.Unlock() + case chunkRetry: + r.mu.Lock() + r.hold.infisicalDown = true + r.mu.Unlock() log.Warn().Err(err).Str("sessionId", spool.sessionID). Msg("agent-vault: could not record session logs, will retry") - return false } + return false } -func (a *sessionLogRecorder) dropRefused(spool *sessionLogSpool, chunk *sealedChunk) { - a.mu.Lock() - defer a.mu.Unlock() - if len(spool.pending) > 0 && spool.pending[0] == chunk { - spool.popPending() - a.sealedBytes -= len(chunk.ciphertext) - // Counted as dropped, or an agent that gets its own chunk refused could erase what it did. - spool.ring.dropped += chunk.lostCount() - } +func (r *sessionLogRecorder) dropRefused(spool *sessionLogSpool, chunk *sealedChunk) { + r.mu.Lock() + defer r.mu.Unlock() + // Counted as dropped, or an agent that gets its own chunk refused could erase what it did. + r.discardHeadLocked(spool, chunk, true) } diff --git a/packages/agentvault/session_log_crypto.go b/packages/agentvault/session_log_crypto.go index 1da5905ad..6554b432d 100644 --- a/packages/agentvault/session_log_crypto.go +++ b/packages/agentvault/session_log_crypto.go @@ -50,12 +50,13 @@ func encodeSessionLogIV(iv []byte) string { // The browser checks the downloaded object against this before decrypting, so an edited object reads as // changed rather than as a decryption failure. -func sessionLogCiphertextSHA256(ciphertext []byte) string { +func infisicalCiphertextSha256(ciphertext []byte) string { sum := sha256.Sum256(ciphertext) return base64.RawStdEncoding.EncodeToString(sum[:]) } -func sessionLogPaddedSHA256(ciphertext []byte) string { +// S3 wants the digest in padded base64, while Infisical stores it unpadded. +func s3ChecksumHeader(ciphertext []byte) string { sum := sha256.Sum256(ciphertext) return base64.StdEncoding.EncodeToString(sum[:]) } diff --git a/packages/agentvault/session_log_ship.go b/packages/agentvault/session_log_ship.go index dd092042c..17440bb59 100644 --- a/packages/agentvault/session_log_ship.go +++ b/packages/agentvault/session_log_ship.go @@ -17,22 +17,56 @@ import ( "github.com/go-resty/resty/v2" ) -func isSessionLogErrorNamed(err error, name string) bool { - var apiErr *api.APIError - return errors.As(err, &apiErr) && apiErr.Name == name -} +const ( + sessionLogCeilingReachedName = "AgentVaultSessionLogCeilingReached" + sessionLogDisabledName = "AgentVaultSessionLogDisabled" + sessionLogClockSkewName = "AgentVaultSessionLogClockSkew" +) + +type chunkRefusal int + +const ( + chunkRetry chunkRefusal = iota + chunkTokenRejected + chunkSessionGone + chunkOrgFull + chunkLoggingOff + chunkClockSkew + chunkRefused +) + +func classifyChunkError(err error) chunkRefusal { + switch { + case isProxyTokenRejected(err): + return chunkTokenRejected + case isSessionGone(err): + return chunkSessionGone + case isSessionLogErrorNamed(err, sessionLogCeilingReachedName): + return chunkOrgFull + case isSessionLogErrorNamed(err, sessionLogDisabledName): + return chunkLoggingOff + case isSessionLogErrorNamed(err, sessionLogClockSkewName): + return chunkClockSkew + } -func isPoisonChunk(err error) bool { var apiErr *api.APIError if !errors.As(err, &apiErr) { - return false + return chunkRetry } // Infisical's own NotFound is caught earlier as a gone session, so a 404 here is a route miss, as during a rollback. if apiErr.StatusCode == http.StatusRequestTimeout || apiErr.StatusCode == http.StatusTooManyRequests || apiErr.StatusCode == http.StatusNotFound { - return false + return chunkRetry } - return apiErr.StatusCode >= 400 && apiErr.StatusCode < 500 + if apiErr.StatusCode >= 400 && apiErr.StatusCode < 500 { + return chunkRefused + } + return chunkRetry +} + +func isSessionLogErrorNamed(err error, name string) bool { + var apiErr *api.APIError + return errors.As(err, &apiErr) && apiErr.Name == name } type sessionLogShipperClient struct { @@ -104,7 +138,7 @@ func (c *sessionLogShipperClient) putObject(ctx context.Context, uploadURL strin req.Header.Set("Content-Length", strconv.Itoa(len(ciphertext))) req.Header.Set("If-None-Match", "*") // Signed into the link, so S3 refuses any body whose digest isn't the one Infisical recorded. - req.Header.Set("X-Amz-Checksum-Sha256", sessionLogPaddedSHA256(ciphertext)) + req.Header.Set("X-Amz-Checksum-Sha256", s3ChecksumHeader(ciphertext)) res, err := c.put.Do(req) if err != nil { diff --git a/packages/agentvault/session_log_ship_test.go b/packages/agentvault/session_log_ship_test.go index 671033a64..8fef071ec 100644 --- a/packages/agentvault/session_log_ship_test.go +++ b/packages/agentvault/session_log_ship_test.go @@ -96,7 +96,7 @@ func TestTheChunkPostCarriesTheProxyTokenAndTheBucketPutDoesNot(t *testing.T) { if putIfNone != "*" { t.Fatalf("the upload sent If-None-Match %q; it must be create-only", putIfNone) } - if putSHA256 != sessionLogCiphertextSHA256(ciphertext)+"=" { + if putSHA256 != infisicalCiphertextSha256(ciphertext)+"=" { t.Fatalf("the upload sent X-Amz-Checksum-Sha256 %q; it must be the padded digest Infisical signed", putSHA256) } if string(putBody) != string(ciphertext) { diff --git a/packages/agentvault/session_log_spool.go b/packages/agentvault/session_log_spool.go index 73ec19fe2..d39ee4764 100644 --- a/packages/agentvault/session_log_spool.go +++ b/packages/agentvault/session_log_spool.go @@ -133,6 +133,14 @@ func (s *sessionLogSpool) popPending() *sealedChunk { return chunk } +func (s *sessionLogSpool) heldRecords() int { + held := s.ring.len() + for _, chunk := range s.pending { + held += chunk.meta.RecordCount + } + return held +} + func newSessionLogSpool(g *sessionLogGrant, now time.Time) *sessionLogSpool { return &sessionLogSpool{ sessionID: g.sessionID, @@ -213,7 +221,7 @@ func (s *sessionLogSpool) sealSlice(records []sessionLogRecord, plaintext []byte DroppedCount: dropped, CiphertextBytes: len(ciphertext), IV: encodeSessionLogIV(iv), - CiphertextSha256: sessionLogCiphertextSHA256(ciphertext), + CiphertextSha256: infisicalCiphertextSha256(ciphertext), }, ciphertext: ciphertext, }, nil diff --git a/packages/agentvault/session_log_test.go b/packages/agentvault/session_log_test.go index f2ec2b22b..7291e8d9c 100644 --- a/packages/agentvault/session_log_test.go +++ b/packages/agentvault/session_log_test.go @@ -139,7 +139,7 @@ func newTestLog(shipper sessionLogShipper) (log *sessionLogRecorder, advance fun } tick = func() { advance(sessionLogFlushInterval) - log.flushAll(context.Background(), false) + log.flush(context.Background(), flushTick) } return log, advance, tick } @@ -247,7 +247,7 @@ func TestTheDropCountIsReportedOnceAndRidesTheFirstChunk(t *testing.T) { for i := 0; i < sessionLogSpoolCapacity+50; i++ { log.record(grant, aRecord("api.github.com")) } - log.flushAll(context.Background(), true) + log.flush(context.Background(), flushFinal) posts := shipper.posts() if len(posts) == 0 { @@ -291,7 +291,7 @@ func TestAChunkIsPostedBeforeItIsUploaded(t *testing.T) { log, _, _ := newTestLog(shipper) log.record(testGrant("s1"), aRecord("api.github.com")) - log.flushAll(context.Background(), true) + log.flush(context.Background(), flushFinal) if got := shipper.kinds(); len(got) != 2 || got[0] != "post" || got[1] != "put" { t.Fatalf("call order was %v, expected post then put", got) @@ -385,7 +385,7 @@ func TestTheCeilingPausesTheWholeProxyAndLiftsAfterTheBackoff(t *testing.T) { log.record(testGrant("s1"), aRecord("api.github.com")) tick() - if !log.holding() || log.switchedOff { + if !log.hold.dropsRecords(log.now()) || log.hold.off { t.Fatal("expected a ceiling pause") } @@ -396,7 +396,7 @@ func TestTheCeilingPausesTheWholeProxyAndLiftsAfterTheBackoff(t *testing.T) { } advance(sessionLogPauseBackoff + time.Second) - log.flushAll(context.Background(), false) + log.flush(context.Background(), flushTick) if len(shipper.posts()) < 2 { t.Fatal("nothing was retried after the pause lifted") } @@ -412,9 +412,9 @@ func TestBeingSwitchedOffDropsWhatWasHeldAndCountsIt(t *testing.T) { tick() spool := log.spools["s1"] - if !log.switchedOff || len(spool.pending) != 0 || spool.ring.len() != 0 { + if !log.hold.off || len(spool.pending) != 0 || spool.ring.len() != 0 { t.Fatalf("switched off=%v, pending=%d, ring=%d; expected everything held to be dropped", - log.switchedOff, len(spool.pending), spool.ring.len()) + log.hold.off, len(spool.pending), spool.ring.len()) } if spool.ring.dropped != 2 { t.Fatalf("%d records were counted as dropped, expected 2", spool.ring.dropped) @@ -441,7 +441,7 @@ func TestAKeyIssuedAfterTheSwitchOffResumesRecordingAtOnce(t *testing.T) { tick() log.record(testGrant("s1"), aRecord("api.github.com")) - if log.switchedOff { + if log.hold.off { t.Fatal("a key issued after the refusal did not end the switch-off") } tick() @@ -463,7 +463,7 @@ func TestARefusalRacingANewKeyDropsNothing(t *testing.T) { log.record(grant, aRecord("api.github.com")) log.switchOff(sessionLogGrantsIssued.Load() - 1) - if log.switchedOff || log.spools["s1"].ring.len() != 1 { + if log.hold.off || log.spools["s1"].ring.len() != 1 { t.Fatal("a refusal sent before a newer key was issued still dropped what was held") } } @@ -476,7 +476,7 @@ func TestDropsAreStillReportedAfterAnIdleSpoolIsForgotten(t *testing.T) { tick() advance(sessionLogIdleClose + time.Minute) - log.flushAll(context.Background(), false) + log.flush(context.Background(), flushTick) if _, ok := log.spools["s1"]; ok { t.Fatal("the idle spool was not forgotten, so this test proves nothing") } @@ -730,7 +730,7 @@ func TestAnIdleSpoolIsForgotten(t *testing.T) { tick() advance(sessionLogIdleClose + time.Minute) - log.flushAll(context.Background(), false) + log.flush(context.Background(), flushTick) if _, ok := log.spools["s1"]; ok { t.Fatal("an idle spool was kept; a long-lived proxy would grow without bound") @@ -759,7 +759,7 @@ func TestCloseFlushesAndThenStopsRecording(t *testing.T) { } log.record(grant, aRecord("api.github.com")) - log.flushAll(context.Background(), true) + log.flush(context.Background(), flushFinal) if len(shipper.puts()) != 1 { t.Fatal("a record was accepted after close") } @@ -772,7 +772,7 @@ func TestABlockedHostIsStillRecorded(t *testing.T) { log.record(testGrant("s1"), sessionLogRecord{ Method: "POST", Host: "evil.example", Port: "443", Path: "/collect", Status: 403, Decision: decisionBlocked, }) - log.flushAll(context.Background(), true) + log.flush(context.Background(), flushFinal) if len(shipper.puts()) != 1 { t.Fatal("a blocked request was not recorded") @@ -824,7 +824,7 @@ func TestSequenceNumbersSurviveASpoolBeingForgotten(t *testing.T) { tick() advance(sessionLogIdleClose + time.Minute) - log.flushAll(context.Background(), false) + log.flush(context.Background(), flushTick) if _, ok := log.spools["s1"]; ok { t.Fatal("the idle spool was not forgotten, so this test proves nothing") } @@ -836,6 +836,34 @@ func TestSequenceNumbersSurviveASpoolBeingForgotten(t *testing.T) { } } +func TestAFullForgottenListDropsOnlyTheLongestForgottenSession(t *testing.T) { + log, _, _ := newTestLog(&fakeShipper{}) + now := log.now() + log.forgotten["oldest"] = forgottenSpool{nextSeq: 1, forgottenAt: now.Add(-2 * time.Hour)} + for i := 1; i < maxSessionCacheEntries; i++ { + log.forgotten[fmt.Sprintf("s%d", i)] = forgottenSpool{nextSeq: 1, forgottenAt: now.Add(-time.Hour)} + } + + spool := newSessionLogSpool(testGrant("newest"), now) + log.spools["newest"] = spool + log.mu.Lock() + log.forgetSpoolLocked("newest", spool) + log.mu.Unlock() + + if len(log.forgotten) != maxSessionCacheEntries { + t.Fatalf("the forgotten list holds %d sessions, want the cap of %d", len(log.forgotten), maxSessionCacheEntries) + } + if _, ok := log.forgotten["oldest"]; ok { + t.Fatal("the longest forgotten session was kept") + } + if _, ok := log.forgotten["s1"]; !ok { + t.Fatal("a more recently forgotten session was evicted too") + } + if _, ok := log.forgotten["newest"]; !ok { + t.Fatal("the session just forgotten was not kept") + } +} + func TestASessionThatIsGoneDoesNotReserveItsSequenceNumbers(t *testing.T) { shipper := &fakeShipper{postResults: []scriptedResult{{err: apiErr(http.StatusNotFound, infisicalNotFoundName)}}} log, _, tick := newTestLog(shipper) @@ -924,7 +952,7 @@ func TestEverySessionShipsOnEveryTickWhileEarlierOnesTakeTimeToUpload(t *testing log.record(testGrant("s2"), aRecord("api.github.com")) tickAt = tickAt.Add(sessionLogFlushInterval) advance(tickAt.Sub(log.now())) - log.flushAll(context.Background(), false) + log.flush(context.Background(), flushTick) if got := len(shipper.puts()); got != 2*i { t.Fatalf("after tick %d there were %d uploads; every session should ship on every tick", i, got) @@ -940,7 +968,7 @@ func TestASessionShipsAtATickThatLandsJustShortOfAMinute(t *testing.T) { tick() log.record(testGrant("s1"), aRecord("api.github.com")) advance(sessionLogFlushInterval - 10*time.Millisecond) - log.flushAll(context.Background(), false) + log.flush(context.Background(), flushTick) if got := len(shipper.puts()); got != 2 { t.Fatalf("there were %d uploads; a tick a few milliseconds early skipped the session", got) @@ -955,7 +983,7 @@ func TestASessionDoesNotShipAgainHalfwayToTheNextTick(t *testing.T) { tick() log.record(testGrant("s1"), aRecord("api.github.com")) advance(sessionLogFlushInterval / 2) - log.flushAll(context.Background(), false) + log.flush(context.Background(), flushTick) if got := len(shipper.puts()); got != 1 { t.Fatalf("there were %d uploads; a session shipped twice within one interval", got) @@ -1006,8 +1034,8 @@ func TestServerErrorsAreRetriedAndBadChunksAreDropped(t *testing.T) { spool := log.spools["s1"] if tc.kept { - if len(spool.pending) != 1 || !log.infisicalDown { - t.Fatalf("%d: the chunk was not kept for a retry (pending %d, infisicalDown %v)", tc.status, len(spool.pending), log.infisicalDown) + if len(spool.pending) != 1 || !log.hold.infisicalDown { + t.Fatalf("%d: the chunk was not kept for a retry (pending %d, infisicalDown %v)", tc.status, len(spool.pending), log.hold.infisicalDown) } continue } @@ -1024,7 +1052,7 @@ func TestTheRecorderShipsNoChunkOverTheServersRecordLimit(t *testing.T) { for i := 0; i < 2500; i++ { log.record(testGrant("s1"), aRecord("api.github.com")) } - log.flushAll(context.Background(), true) + log.flush(context.Background(), flushFinal) posts := shipper.posts() if len(posts) != 3 { @@ -1075,7 +1103,7 @@ func TestShutdownWaitsForAFlushInProgressSoNoChunkShipsTwice(t *testing.T) { tickDone := make(chan struct{}) go func() { defer close(tickDone) - log.flushAll(context.Background(), false) + log.flush(context.Background(), flushTick) }() <-shipper.entered @@ -1140,7 +1168,7 @@ func TestAWakeDoesNotResetTheBreakers(t *testing.T) { log.record(testGrant("s1"), aRecord("api.github.com")) tick() - if !log.s3Down { + if !log.hold.s3Down { t.Fatal("a failed upload did not trip the breaker, so this test proves nothing") } putsAfterTick := len(shipper.puts()) @@ -1150,7 +1178,7 @@ func TestAWakeDoesNotResetTheBreakers(t *testing.T) { } log.flush(context.Background(), flushWake) - if !log.s3Down { + if !log.hold.s3Down { t.Fatal("a wake reset the breaker") } if got := len(shipper.puts()); got != putsAfterTick { From d282da457586b2363bb03b1c781a9f445d0c64be Mon Sep 17 00:00:00 2001 From: Saif Ur Rahman Date: Mon, 28 Sep 2026 14:14:34 +0530 Subject: [PATCH 36/43] refactor(agent-vault): order the session log recorder by stage and put all its data under one lock --- packages/agentvault/session_log.go | 288 ++++++++++++----------- packages/agentvault/session_log_spool.go | 18 +- packages/agentvault/session_log_test.go | 52 ++-- 3 files changed, 189 insertions(+), 169 deletions(-) diff --git a/packages/agentvault/session_log.go b/packages/agentvault/session_log.go index 1270be686..0dd091ecb 100644 --- a/packages/agentvault/session_log.go +++ b/packages/agentvault/session_log.go @@ -24,10 +24,11 @@ const ( sessionLogIdleClose = 15 * time.Minute - sessionLogPauseBackoff = 15 * time.Minute - sessionLogPutTimeout = 10 * time.Second - sessionLogFinalTimeout = 3 * time.Second - sessionLogCloseTimeout = 5 * time.Second + sessionLogPauseBackoff = 15 * time.Minute + sessionLogPutTimeout = 10 * time.Second + sessionLogFinalTimeout = 3 * time.Second + sessionLogCloseTimeout = 5 * time.Second + sessionLogUploadURLMargin = 10 * time.Second // get a fresh link if the current one expires within this ) type sessionLogGrant struct { @@ -46,9 +47,9 @@ func newSessionLogGrant(sessionID string, key []byte) *sessionLogGrant { } type forgottenSpool struct { - nextSeq uint64 - dropped uint64 - forgottenAt time.Time + nextSeq uint64 + unreportedDrops uint64 + forgottenAt time.Time } type sessionLogShipper interface { @@ -80,29 +81,41 @@ func (h *sessionLogHold) clearOutages() { h.infisicalDown = false } +type flushPass int + +const ( + flushTick flushPass = iota + flushWake + flushFinal +) + +// record() appends each request to its session's ring. A flush seals the ring into encrypted chunks and +// queues them on the spool. Each queued chunk then gets its index row (createChunk) and is uploaded to S3 +// (putObject). When Infisical refuses a chunk, handleCreateFailure classifies the refusal and handles it. type sessionLogRecorder struct { proxyID string shipper sessionLogShipper now func() time.Time + wake chan struct{} - // close's flush can overlap the run loop's; unguarded, one chunk ships twice. + // flushMu makes flush passes take turns, so close() and the run loop never ship one chunk twice. flushMu sync.Mutex + // mu guards everything below, and each queued chunk's upload fields. mu sync.Mutex spools map[string]*sessionLogSpool forgotten map[string]forgottenSpool - total int - sealedBytes int - nextSealOrder uint64 + unsealedRecords int + sealedBytes int + nextSealOrder uint64 hold sessionLogHold clockSkewReported bool closed bool - wake chan struct{} } func newSessionLogRecorder(proxyID string, shipper sessionLogShipper) *sessionLogRecorder { @@ -131,7 +144,7 @@ func (r *sessionLogRecorder) record(g *sessionLogGrant, rec sessionLogRecord) { if !ok { spool = newSessionLogSpool(g, r.now()) if prior, ok := r.forgotten[g.sessionID]; ok { - spool.nextSeq, spool.ring.dropped = prior.nextSeq, prior.dropped + spool.nextSeq, spool.ring.unreportedDrops = prior.nextSeq, prior.unreportedDrops delete(r.forgotten, g.sessionID) } r.spools[g.sessionID] = spool @@ -148,7 +161,7 @@ func (r *sessionLogRecorder) record(g *sessionLogGrant, rec sessionLogRecord) { if r.hold.off { if g.issued <= r.hold.offThrough { - spool.ring.dropped++ + spool.ring.unreportedDrops++ return } r.hold.off = false @@ -156,17 +169,17 @@ func (r *sessionLogRecorder) record(g *sessionLogGrant, rec sessionLogRecord) { } if now.Before(r.hold.pausedUntil) { - spool.ring.dropped++ + spool.ring.unreportedDrops++ return } - if r.total >= sessionLogTotalCapacity { - spool.ring.dropped++ + if r.unsealedRecords >= sessionLogTotalCapacity { + spool.ring.unreportedDrops++ return } if evicted := spool.ring.push(rec); !evicted { - r.total++ + r.unsealedRecords++ } if spool.ring.len() >= sessionLogFlushRecords { @@ -217,32 +230,40 @@ func (r *sessionLogRecorder) close(ctx context.Context) { } } -func (r *sessionLogRecorder) dueSpools(pass flushPass, now time.Time) []*sessionLogSpool { - r.mu.Lock() - defer r.mu.Unlock() +func (r *sessionLogRecorder) flush(ctx context.Context, pass flushPass) { + if r == nil { + return + } - final := pass == flushFinal - due := make([]*sessionLogSpool, 0, len(r.spools)) + r.flushMu.Lock() + defer r.flushMu.Unlock() + + // A busy session wakes the loop every thousand records, so resetting here would retry a down bucket or + // Infisical that often instead of once a minute. + if pass != flushWake { + r.mu.Lock() + r.hold.clearOutages() + r.mu.Unlock() + } + + // One start time for every spool, so shipping the earlier ones doesn't make the later ones late for the next tick. + started := r.now() + if pass == flushTick { + r.mu.Lock() + r.forgetIdleSpoolsLocked(started) + r.mu.Unlock() + } + for _, spool := range r.dueSpools(pass, started) { + r.flushSpool(ctx, spool, pass, started) + } +} + +func (r *sessionLogRecorder) forgetIdleSpoolsLocked(now time.Time) { for id, spool := range r.spools { - if pass == flushWake { - if spool.ring.len() >= sessionLogFlushRecords { - due = append(due, spool) - } - continue - } - if spool.ring.len() == 0 && len(spool.pending) == 0 { - if !final && now.Sub(spool.lastRecordAt) > sessionLogIdleClose { - r.forgetSpoolLocked(id, spool) - } - continue - } - // The slack absorbs ticker jitter: without it a spool stamped at one tick is a hair short of due at the next. - if final || spool.ring.len() >= sessionLogFlushRecords || len(spool.pending) > 0 || - (spool.ring.len() > 0 && now.Sub(spool.lastFlushAt) >= sessionLogFlushInterval-sessionLogFlushSlack) { - due = append(due, spool) + if spool.ring.len() == 0 && len(spool.pending) == 0 && now.Sub(spool.lastRecordAt) > sessionLogIdleClose { + r.forgetSpoolLocked(id, spool) } } - return due } // Keeps the drop count too, so a spool forgotten while logging was off still reports what it lost. @@ -250,7 +271,7 @@ func (r *sessionLogRecorder) forgetSpoolLocked(sessionID string, spool *sessionL if len(r.forgotten) >= maxSessionCacheEntries { r.evictLongestForgottenLocked() } - r.forgotten[sessionID] = forgottenSpool{nextSeq: spool.nextSeq, dropped: spool.ring.dropped, forgottenAt: r.now()} + r.forgotten[sessionID] = forgottenSpool{nextSeq: spool.nextSeq, unreportedDrops: spool.ring.unreportedDrops, forgottenAt: r.now()} delete(r.spools, sessionID) } @@ -266,96 +287,29 @@ func (r *sessionLogRecorder) evictLongestForgottenLocked() { delete(r.forgotten, oldestID) } -func (r *sessionLogRecorder) pause() { - r.mu.Lock() - defer r.mu.Unlock() - r.hold.pausedUntil = r.now().Add(sessionLogPauseBackoff) -} - -// Drops everything held and counts it, unless a grant was issued after the refused request went out, which -// makes the refusal stale (see sessionLogGrantsIssued). -func (r *sessionLogRecorder) switchOff(grantsIssuedAtSend uint64) { +func (r *sessionLogRecorder) dueSpools(pass flushPass, now time.Time) []*sessionLogSpool { r.mu.Lock() defer r.mu.Unlock() - if sessionLogGrantsIssued.Load() > grantsIssuedAtSend { - return - } - var lost int + final := pass == flushFinal + due := make([]*sessionLogSpool, 0, len(r.spools)) for _, spool := range r.spools { - lost += r.discardAllLocked(spool, true) - } - - if !r.hold.off { - log.Warn().Int("records", lost). - Msg("agent-vault: session logs are off for this project, dropping what was held until they are back on") - } - r.hold.off = true - r.hold.offThrough = grantsIssuedAtSend -} - -// A gone session passes countDropped false: its spool is deleted, so no later chunk could report the loss. -func (r *sessionLogRecorder) discardAllLocked(spool *sessionLogSpool, countDropped bool) (lost int) { - lost = spool.heldRecords() - - held := spool.ring.len() - spool.ring.drain(held) - r.total -= held - if countDropped { - spool.ring.dropped += uint64(held) - } - - for _, chunk := range spool.pending { - r.sealedBytes -= len(chunk.ciphertext) - if countDropped { - spool.ring.dropped += chunk.lostCount() + if pass == flushWake { + if spool.ring.len() >= sessionLogFlushRecords { + due = append(due, spool) + } + continue + } + if spool.ring.len() == 0 && len(spool.pending) == 0 { + continue + } + // The slack absorbs ticker jitter: without it a spool stamped at one tick is a hair short of due at the next. + if final || spool.ring.len() >= sessionLogFlushRecords || len(spool.pending) > 0 || + (spool.ring.len() > 0 && now.Sub(spool.lastFlushAt) >= sessionLogFlushInterval-sessionLogFlushSlack) { + due = append(due, spool) } } - spool.pending = nil - return lost -} - -func (r *sessionLogRecorder) discardHeadLocked(spool *sessionLogSpool, chunk *sealedChunk, countDropped bool) bool { - if len(spool.pending) == 0 || spool.pending[0] != chunk { - return false - } - spool.popPending() - r.sealedBytes -= len(chunk.ciphertext) - if countDropped { - spool.ring.dropped += chunk.lostCount() - } - return true -} - -type flushPass int - -const ( - flushTick flushPass = iota - flushWake - flushFinal -) - -func (r *sessionLogRecorder) flush(ctx context.Context, pass flushPass) { - if r == nil { - return - } - - r.flushMu.Lock() - defer r.flushMu.Unlock() - - // A busy session wakes the loop every thousand records, so resetting here would retry a down bucket or - // Infisical that often instead of once a minute. - if pass != flushWake { - r.mu.Lock() - r.hold.clearOutages() - r.mu.Unlock() - } - - // One start time for every spool, so shipping the earlier ones doesn't make the later ones late for the next tick. - started := r.now() - for _, spool := range r.dueSpools(pass, started) { - r.flushSpool(ctx, spool, pass, started) - } + return due } func (r *sessionLogRecorder) sealRing(spool *sessionLogSpool, started time.Time) { @@ -367,8 +321,8 @@ func (r *sessionLogRecorder) sealRing(spool *sessionLogSpool, started time.Time) r.mu.Unlock() return } - r.total -= len(records) - dropped := spool.ring.takeDropped() + r.unsealedRecords -= len(records) + dropped := spool.ring.takeUnreportedDrops() r.mu.Unlock() groups, err := packSessionLogRecords(records) @@ -401,7 +355,7 @@ func (r *sessionLogRecorder) sealRing(spool *sessionLogSpool, started time.Time) func (r *sessionLogRecorder) dropUnsealed(spool *sessionLogSpool, records int, dropped uint64, err error) { r.mu.Lock() - spool.ring.dropped += dropped + uint64(records) + spool.ring.unreportedDrops += dropped + uint64(records) r.mu.Unlock() log.Error().Err(err).Str("sessionId", spool.sessionID).Int("records", records). Msg("agent-vault: could not seal a session log chunk, dropping those records") @@ -467,7 +421,10 @@ func (r *sessionLogRecorder) flushSpool(ctx context.Context, spool *sessionLogSp func (r *sessionLogRecorder) shipChunk(ctx context.Context, spool *sessionLogSpool, chunk *sealedChunk, pass flushPass) bool { final := pass == flushFinal - if chunk.uploadURL == "" || r.now().Add(10*time.Second).After(chunk.urlExpires) { + r.mu.Lock() + uploadURL, urlExpires := chunk.uploadURL, chunk.urlExpires + r.mu.Unlock() + if uploadURL == "" || r.now().Add(sessionLogUploadURLMargin).After(urlExpires) { // Past the shutdown budget, a new row could only be written for an upload that can no longer happen. if ctx.Err() != nil { return false @@ -475,14 +432,16 @@ func (r *sessionLogRecorder) shipChunk(ctx context.Context, spool *sessionLogSpo grantsIssuedAtSend := sessionLogGrantsIssued.Load() res, err := r.shipper.createChunk(ctx, final, spool.sessionID, chunk.meta) if err != nil { - return r.handleCreateFailure(spool, chunk, err, grantsIssuedAtSend) + r.handleCreateFailure(spool, chunk, err, grantsIssuedAtSend) + return false } r.mu.Lock() r.clockSkewReported = false - r.mu.Unlock() chunk.posted = true chunk.uploadURL = res.UploadURL chunk.urlExpires = r.now().Add(time.Duration(res.ExpiresInSeconds) * time.Second) + uploadURL = chunk.uploadURL + r.mu.Unlock() } putCtx := ctx @@ -492,9 +451,9 @@ func (r *sessionLogRecorder) shipChunk(ctx context.Context, spool *sessionLogSpo defer cancel() } - if err := r.shipper.putObject(putCtx, chunk.uploadURL, chunk.ciphertext); err != nil { - chunk.uploadURL = "" + if err := r.shipper.putObject(putCtx, uploadURL, chunk.ciphertext); err != nil { r.mu.Lock() + chunk.uploadURL = "" r.hold.s3Down = true r.mu.Unlock() log.Warn().Err(err).Str("sessionId", spool.sessionID).Str("chunkId", chunk.meta.ChunkID). @@ -505,7 +464,7 @@ func (r *sessionLogRecorder) shipChunk(ctx context.Context, spool *sessionLogSpo return true } -func (r *sessionLogRecorder) handleCreateFailure(spool *sessionLogSpool, chunk *sealedChunk, err error, grantsIssuedAtSend uint64) bool { +func (r *sessionLogRecorder) handleCreateFailure(spool *sessionLogSpool, chunk *sealedChunk, err error, grantsIssuedAtSend uint64) { switch classifyChunkError(err) { case chunkTokenRejected: log.Warn().Err(err).Msg("agent-vault: Infisical rejected this proxy's token, holding session logs") @@ -547,7 +506,34 @@ func (r *sessionLogRecorder) handleCreateFailure(spool *sessionLogSpool, chunk * log.Warn().Err(err).Str("sessionId", spool.sessionID). Msg("agent-vault: could not record session logs, will retry") } - return false +} + +// Drops everything held and counts it, unless a grant was issued after the refused request went out, which +// makes the refusal stale (see sessionLogGrantsIssued). +func (r *sessionLogRecorder) switchOff(grantsIssuedAtSend uint64) { + r.mu.Lock() + defer r.mu.Unlock() + if sessionLogGrantsIssued.Load() > grantsIssuedAtSend { + return + } + + var lost int + for _, spool := range r.spools { + lost += r.discardAllLocked(spool, true) + } + + if !r.hold.off { + log.Warn().Int("records", lost). + Msg("agent-vault: session logs are off for this project, dropping what was held until they are back on") + } + r.hold.off = true + r.hold.offThrough = grantsIssuedAtSend +} + +func (r *sessionLogRecorder) pause() { + r.mu.Lock() + defer r.mu.Unlock() + r.hold.pausedUntil = r.now().Add(sessionLogPauseBackoff) } func (r *sessionLogRecorder) dropRefused(spool *sessionLogSpool, chunk *sealedChunk) { @@ -556,3 +542,35 @@ func (r *sessionLogRecorder) dropRefused(spool *sessionLogSpool, chunk *sealedCh // Counted as dropped, or an agent that gets its own chunk refused could erase what it did. r.discardHeadLocked(spool, chunk, true) } + +// A gone session passes countDropped false: its spool is deleted, so no later chunk could report the loss. +func (r *sessionLogRecorder) discardAllLocked(spool *sessionLogSpool, countDropped bool) (lost int) { + lost = spool.heldRecords() + + held := spool.ring.len() + spool.ring.drain(held) + r.unsealedRecords -= held + if countDropped { + spool.ring.unreportedDrops += uint64(held) + } + + for _, chunk := range spool.pending { + r.sealedBytes -= len(chunk.ciphertext) + if countDropped { + spool.ring.unreportedDrops += chunk.lostCount() + } + } + spool.pending = nil + return lost +} + +func (r *sessionLogRecorder) discardHeadLocked(spool *sessionLogSpool, chunk *sealedChunk, countDropped bool) { + if len(spool.pending) == 0 || spool.pending[0] != chunk { + return + } + spool.popPending() + r.sealedBytes -= len(chunk.ciphertext) + if countDropped { + spool.ring.unreportedDrops += chunk.lostCount() + } +} diff --git a/packages/agentvault/session_log_spool.go b/packages/agentvault/session_log_spool.go index d39ee4764..cf0d0b030 100644 --- a/packages/agentvault/session_log_spool.go +++ b/packages/agentvault/session_log_spool.go @@ -31,7 +31,8 @@ type sessionLogRing struct { capacity int head int n int - dropped uint64 + + unreportedDrops uint64 } func newSessionLogRing(capacity int) sessionLogRing { @@ -47,7 +48,7 @@ func (r *sessionLogRing) push(rec sessionLogRecord) (evicted bool) { if r.n == r.capacity { r.buf[r.head] = rec r.head = (r.head + 1) % len(r.buf) - r.dropped++ + r.unreportedDrops++ return true } r.buf[(r.head+r.n)%len(r.buf)] = rec @@ -86,20 +87,21 @@ func (r *sessionLogRing) drain(max int) []sessionLogRecord { return out } -func (r *sessionLogRing) takeDropped() uint64 { - dropped := r.dropped - r.dropped = 0 +func (r *sessionLogRing) takeUnreportedDrops() uint64 { + dropped := r.unreportedDrops + r.unreportedDrops = 0 return dropped } type sealedChunk struct { meta api.CreateAgentVaultSessionLogChunkRequest ciphertext []byte + sealOrder uint64 + + // uploadURL, urlExpires and posted are guarded by the recorder's mu. uploadURL string urlExpires time.Time - - sealOrder uint64 - posted bool + posted bool } // A posted chunk's row already reports its records and drops, so counting them here would double-report. diff --git a/packages/agentvault/session_log_test.go b/packages/agentvault/session_log_test.go index 7291e8d9c..849d7bb82 100644 --- a/packages/agentvault/session_log_test.go +++ b/packages/agentvault/session_log_test.go @@ -168,8 +168,8 @@ func TestTheRingDropsTheOldestAndCountsIt(t *testing.T) { if ring.len() != 3 { t.Fatalf("ring holds %d, capacity is 3", ring.len()) } - if ring.dropped != 2 { - t.Fatalf("ring counted %d drops, expected 2", ring.dropped) + if ring.unreportedDrops != 2 { + t.Fatalf("ring counted %d drops, expected 2", ring.unreportedDrops) } drained := ring.drain(10) @@ -215,8 +215,8 @@ func TestTheRingKeepsItsOrderWhileItGrowsPastAWrap(t *testing.T) { t.Fatalf("record %d has seq %d, expected %d; growth reordered the ring", i, rec.Seq, 10+i) } } - if ring.dropped != 0 { - t.Fatalf("growth counted %d drops; nothing was over capacity", ring.dropped) + if ring.unreportedDrops != 0 { + t.Fatalf("growth counted %d drops; nothing was over capacity", ring.unreportedDrops) } } @@ -253,7 +253,7 @@ func TestTheDropCountIsReportedOnceAndRidesTheFirstChunk(t *testing.T) { if len(posts) == 0 { t.Fatal("nothing was shipped") } - if log.spools["s1"].ring.dropped != 0 { + if log.spools["s1"].ring.unreportedDrops != 0 { t.Fatal("the drop count was not reset after being reported") } } @@ -281,8 +281,8 @@ func TestTheProxyWideFuseDropsTheNewest(t *testing.T) { } } - if log.total > sessionLogTotalCapacity { - t.Fatalf("the proxy holds %d records, past the %d fuse", log.total, sessionLogTotalCapacity) + if log.unsealedRecords > sessionLogTotalCapacity { + t.Fatalf("the proxy holds %d records, past the %d fuse", log.unsealedRecords, sessionLogTotalCapacity) } } @@ -352,8 +352,8 @@ func TestA404ThatIsNotInfisicalsNotFoundIsRetried(t *testing.T) { if !ok { t.Fatal("an unnamed 404 dropped the whole session") } - if len(spool.pending) != 1 || spool.ring.dropped != 0 { - t.Fatalf("the chunk was not kept for a retry (pending %d, dropped %d)", len(spool.pending), spool.ring.dropped) + if len(spool.pending) != 1 || spool.ring.unreportedDrops != 0 { + t.Fatalf("the chunk was not kept for a retry (pending %d, dropped %d)", len(spool.pending), spool.ring.unreportedDrops) } tick() @@ -416,11 +416,11 @@ func TestBeingSwitchedOffDropsWhatWasHeldAndCountsIt(t *testing.T) { t.Fatalf("switched off=%v, pending=%d, ring=%d; expected everything held to be dropped", log.hold.off, len(spool.pending), spool.ring.len()) } - if spool.ring.dropped != 2 { - t.Fatalf("%d records were counted as dropped, expected 2", spool.ring.dropped) + if spool.ring.unreportedDrops != 2 { + t.Fatalf("%d records were counted as dropped, expected 2", spool.ring.unreportedDrops) } - if log.total != 0 || log.sealedBytes != 0 { - t.Fatalf("totals not restored: records=%d sealed bytes=%d", log.total, log.sealedBytes) + if log.unsealedRecords != 0 || log.sealedBytes != 0 { + t.Fatalf("totals not restored: records=%d sealed bytes=%d", log.unsealedRecords, log.sealedBytes) } log.record(grant, aRecord("api.github.com")) @@ -428,8 +428,8 @@ func TestBeingSwitchedOffDropsWhatWasHeldAndCountsIt(t *testing.T) { if len(shipper.posts()) != 1 { t.Fatal("the proxy kept sending while logging was switched off") } - if spool.ring.dropped != 3 { - t.Fatalf("a record made with the old key was not counted as dropped, got %d", spool.ring.dropped) + if spool.ring.unreportedDrops != 3 { + t.Fatalf("a record made with the old key was not counted as dropped, got %d", spool.ring.unreportedDrops) } } @@ -498,11 +498,11 @@ func TestRecordsArePausedAsCountedGapsNotSilentLosses(t *testing.T) { log.record(grant, aRecord("api.github.com")) tick() - before := log.spools["s1"].ring.dropped + before := log.spools["s1"].ring.unreportedDrops for i := 0; i < 5; i++ { log.record(grant, aRecord("api.github.com")) } - if got := log.spools["s1"].ring.dropped - before; got != 5 { + if got := log.spools["s1"].ring.unreportedDrops - before; got != 5 { t.Fatalf("%d records were counted as dropped while paused, expected 5", got) } } @@ -633,7 +633,7 @@ func TestThePendingCapEvictsTheOldestAndCountsIt(t *testing.T) { if len(spool.pending) > sessionLogPendingChunks { t.Fatalf("pending holds %d chunks, the cap is %d", len(spool.pending), sessionLogPendingChunks) } - if spool.ring.dropped == 0 { + if spool.ring.unreportedDrops == 0 { t.Fatal("evicted chunks were not counted as dropped records") } } @@ -679,10 +679,10 @@ func TestTheByteCapEvictsTheOldestChunkOnTheProxy(t *testing.T) { t.Fatalf("s%d lost its chunk; only the oldest should go", i) } } - if got := log.spools["oldest"].ring.dropped; got != 107 { + if got := log.spools["oldest"].ring.unreportedDrops; got != 107 { t.Fatalf("the unposted chunk counted %d dropped, expected 107", got) } - if got := log.spools["posted"].ring.dropped; got != 0 { + if got := log.spools["posted"].ring.unreportedDrops; got != 0 { t.Fatalf("the posted chunk counted %d dropped, expected 0", got) } } @@ -890,8 +890,8 @@ func TestRecordsLostToASealFailureAreStillCounted(t *testing.T) { if spool == nil { t.Fatal("the spool disappeared") } - if spool.ring.dropped != 3 { - t.Fatalf("%d records were counted as dropped after a seal failure, expected 3", spool.ring.dropped) + if spool.ring.unreportedDrops != 3 { + t.Fatalf("%d records were counted as dropped after a seal failure, expected 3", spool.ring.unreportedDrops) } if len(shipper.posts()) != 0 { t.Fatal("a chunk was shipped despite the seal failing") @@ -1005,9 +1005,9 @@ func TestEveryWayAChunkLeavesReleasesWhatItHeld(t *testing.T) { log.record(testGrant("s1"), aRecord("api.github.com")) tick() - if log.total != 0 || log.sealedBytes != 0 { + if log.unsealedRecords != 0 || log.sealedBytes != 0 { t.Fatalf("%s: the proxy still counts %d records and %d sealed bytes; the pending cap would fill and stop recording", - outcome.name, log.total, log.sealedBytes) + outcome.name, log.unsealedRecords, log.sealedBytes) } } } @@ -1039,8 +1039,8 @@ func TestServerErrorsAreRetriedAndBadChunksAreDropped(t *testing.T) { } continue } - if len(spool.pending) != 0 || spool.ring.dropped != 1 { - t.Fatalf("%d: a refused chunk was not dropped and counted (pending %d, dropped %d)", tc.status, len(spool.pending), spool.ring.dropped) + if len(spool.pending) != 0 || spool.ring.unreportedDrops != 1 { + t.Fatalf("%d: a refused chunk was not dropped and counted (pending %d, dropped %d)", tc.status, len(spool.pending), spool.ring.unreportedDrops) } } } From 0e62b8f8c35f8d69138a749b18f1799701226db7 Mon Sep 17 00:00:00 2001 From: Saif Ur Rahman Date: Mon, 28 Sep 2026 19:29:15 +0530 Subject: [PATCH 37/43] refactor(agent-vault): name the session log recorder's decisions and limits --- packages/agentvault/session_log.go | 137 ++++++++++++++--------- packages/agentvault/session_log_spool.go | 13 ++- packages/agentvault/session_log_test.go | 10 +- 3 files changed, 97 insertions(+), 63 deletions(-) diff --git a/packages/agentvault/session_log.go b/packages/agentvault/session_log.go index 0dd091ecb..777983251 100644 --- a/packages/agentvault/session_log.go +++ b/packages/agentvault/session_log.go @@ -16,11 +16,11 @@ const ( sessionLogFlushRecords = 1000 sessionLogMaxChunkPlaintext = 4 << 20 - sessionLogSpoolCapacity = 5000 - sessionLogTotalCapacity = 200_000 - - sessionLogPendingChunks = 10 - sessionLogTotalSealedBytes = 64 << 20 + sessionLogSpoolCapacity = 5000 // one session's ring; full: overwrite the oldest line, count it + sessionLogTotalCapacity = 200_000 // all rings together; full: drop new lines, count them + sessionLogPendingChunks = 10 // one session's sealed chunks; full: drop the oldest, count it + sessionLogTotalSealedBytes = 64 << 20 // all sealed chunks; full: drop the oldest chunk of any session + sessionLogForgottenCapacity = maxSessionCacheEntries // idle sessions remembered; full: forget the one forgotten longest sessionLogIdleClose = 15 * time.Minute @@ -42,6 +42,8 @@ type sessionLogGrant struct { // out means logs came back on after that request, and the refusal is stale. var sessionLogGrantsIssued atomic.Uint64 +func grantIssuedSince(snapshot uint64) bool { return sessionLogGrantsIssued.Load() > snapshot } + func newSessionLogGrant(sessionID string, key []byte) *sessionLogGrant { return &sessionLogGrant{sessionID: sessionID, key: key, issued: sessionLogGrantsIssued.Add(1)} } @@ -159,21 +161,7 @@ func (r *sessionLogRecorder) record(g *sessionLogGrant, rec sessionLogRecord) { rec.Ts = rec.at.UTC().Format(time.RFC3339Nano) spool.lastRecordAt = now - if r.hold.off { - if g.issued <= r.hold.offThrough { - spool.ring.unreportedDrops++ - return - } - r.hold.off = false - log.Info().Msg("agent-vault: session logs are back on, recording again") - } - - if now.Before(r.hold.pausedUntil) { - spool.ring.unreportedDrops++ - return - } - - if r.unsealedRecords >= sessionLogTotalCapacity { + if !r.admitLocked(g, now) { spool.ring.unreportedDrops++ return } @@ -190,6 +178,24 @@ func (r *sessionLogRecorder) record(g *sessionLogGrant, rec sessionLogRecord) { } } +// Also switches recording back on when a grant issued after logs went off arrives. +func (r *sessionLogRecorder) admitLocked(g *sessionLogGrant, now time.Time) bool { + if r.hold.off { + if g.issued <= r.hold.offThrough { + return false + } + r.hold.off = false + log.Info().Msg("agent-vault: session logs are back on, recording again") + } + if now.Before(r.hold.pausedUntil) { + return false + } + if r.unsealedRecords >= sessionLogTotalCapacity { + return false + } + return true +} + func (r *sessionLogRecorder) run(stop <-chan struct{}) { if r == nil { return @@ -268,7 +274,7 @@ func (r *sessionLogRecorder) forgetIdleSpoolsLocked(now time.Time) { // Keeps the drop count too, so a spool forgotten while logging was off still reports what it lost. func (r *sessionLogRecorder) forgetSpoolLocked(sessionID string, spool *sessionLogSpool) { - if len(r.forgotten) >= maxSessionCacheEntries { + if len(r.forgotten) >= sessionLogForgottenCapacity { r.evictLongestForgottenLocked() } r.forgotten[sessionID] = forgottenSpool{nextSeq: spool.nextSeq, unreportedDrops: spool.ring.unreportedDrops, forgottenAt: r.now()} @@ -291,27 +297,35 @@ func (r *sessionLogRecorder) dueSpools(pass flushPass, now time.Time) []*session r.mu.Lock() defer r.mu.Unlock() - final := pass == flushFinal due := make([]*sessionLogSpool, 0, len(r.spools)) for _, spool := range r.spools { - if pass == flushWake { - if spool.ring.len() >= sessionLogFlushRecords { - due = append(due, spool) - } - continue - } - if spool.ring.len() == 0 && len(spool.pending) == 0 { - continue - } - // The slack absorbs ticker jitter: without it a spool stamped at one tick is a hair short of due at the next. - if final || spool.ring.len() >= sessionLogFlushRecords || len(spool.pending) > 0 || - (spool.ring.len() > 0 && now.Sub(spool.lastFlushAt) >= sessionLogFlushInterval-sessionLogFlushSlack) { + if r.isDueLocked(spool, pass, now) { due = append(due, spool) } } return due } +func (r *sessionLogRecorder) isDueLocked(spool *sessionLogSpool, pass flushPass, now time.Time) bool { + if pass == flushWake { + return spool.ring.len() >= sessionLogFlushRecords + } + if spool.ring.len() == 0 && len(spool.pending) == 0 { + return false + } + if pass == flushFinal { + return true + } + if spool.ring.len() >= sessionLogFlushRecords { + return true + } + if len(spool.pending) > 0 { + return true + } + // The slack absorbs ticker jitter: without it a spool stamped at one tick is a hair short of due at the next. + return now.Sub(spool.lastFlushAt) >= sessionLogFlushInterval-sessionLogFlushSlack +} + func (r *sessionLogRecorder) sealRing(spool *sessionLogSpool, started time.Time) { for { r.mu.Lock() @@ -332,6 +346,7 @@ func (r *sessionLogRecorder) sealRing(spool *sessionLogSpool, started time.Time) } for i, group := range groups { var groupDropped uint64 + // Only the first chunk of a split batch carries the drop count, so drops aren't reported twice. if i == 0 { groupDropped = dropped } @@ -420,32 +435,44 @@ func (r *sessionLogRecorder) flushSpool(ctx context.Context, spool *sessionLogSp } func (r *sessionLogRecorder) shipChunk(ctx context.Context, spool *sessionLogSpool, chunk *sealedChunk, pass flushPass) bool { - final := pass == flushFinal + uploadURL, ok := r.ensureUploadLink(ctx, spool, chunk, pass) + if !ok { + return false + } + return r.upload(ctx, spool, chunk, pass, uploadURL) +} + +func (r *sessionLogRecorder) ensureUploadLink(ctx context.Context, spool *sessionLogSpool, chunk *sealedChunk, pass flushPass) (uploadURL string, ok bool) { r.mu.Lock() uploadURL, urlExpires := chunk.uploadURL, chunk.urlExpires r.mu.Unlock() - if uploadURL == "" || r.now().Add(sessionLogUploadURLMargin).After(urlExpires) { - // Past the shutdown budget, a new row could only be written for an upload that can no longer happen. - if ctx.Err() != nil { - return false - } - grantsIssuedAtSend := sessionLogGrantsIssued.Load() - res, err := r.shipper.createChunk(ctx, final, spool.sessionID, chunk.meta) - if err != nil { - r.handleCreateFailure(spool, chunk, err, grantsIssuedAtSend) - return false - } - r.mu.Lock() - r.clockSkewReported = false - chunk.posted = true - chunk.uploadURL = res.UploadURL - chunk.urlExpires = r.now().Add(time.Duration(res.ExpiresInSeconds) * time.Second) - uploadURL = chunk.uploadURL - r.mu.Unlock() + if uploadURL != "" && !r.now().Add(sessionLogUploadURLMargin).After(urlExpires) { + return uploadURL, true + } + + // Past the shutdown budget, a new row could only be written for an upload that can no longer happen. + if ctx.Err() != nil { + return "", false + } + grantsIssuedAtSend := sessionLogGrantsIssued.Load() + res, err := r.shipper.createChunk(ctx, pass == flushFinal, spool.sessionID, chunk.meta) + if err != nil { + r.handleCreateFailure(spool, chunk, err, grantsIssuedAtSend) + return "", false } + r.mu.Lock() + defer r.mu.Unlock() + r.clockSkewReported = false + chunk.state = chunkPosted + chunk.uploadURL = res.UploadURL + chunk.urlExpires = r.now().Add(time.Duration(res.ExpiresInSeconds) * time.Second) + return chunk.uploadURL, true +} + +func (r *sessionLogRecorder) upload(ctx context.Context, spool *sessionLogSpool, chunk *sealedChunk, pass flushPass, uploadURL string) bool { putCtx := ctx - if !final { + if pass != flushFinal { var cancel context.CancelFunc putCtx, cancel = context.WithTimeout(ctx, sessionLogPutTimeout) defer cancel() @@ -513,7 +540,7 @@ func (r *sessionLogRecorder) handleCreateFailure(spool *sessionLogSpool, chunk * func (r *sessionLogRecorder) switchOff(grantsIssuedAtSend uint64) { r.mu.Lock() defer r.mu.Unlock() - if sessionLogGrantsIssued.Load() > grantsIssuedAtSend { + if grantIssuedSince(grantsIssuedAtSend) { return } diff --git a/packages/agentvault/session_log_spool.go b/packages/agentvault/session_log_spool.go index cf0d0b030..119953a01 100644 --- a/packages/agentvault/session_log_spool.go +++ b/packages/agentvault/session_log_spool.go @@ -93,20 +93,27 @@ func (r *sessionLogRing) takeUnreportedDrops() uint64 { return dropped } +type chunkState int + +const ( + chunkSealed chunkState = iota + chunkPosted +) + type sealedChunk struct { meta api.CreateAgentVaultSessionLogChunkRequest ciphertext []byte sealOrder uint64 - // uploadURL, urlExpires and posted are guarded by the recorder's mu. + // uploadURL, urlExpires and state are guarded by the recorder's mu. uploadURL string urlExpires time.Time - posted bool + state chunkState } // A posted chunk's row already reports its records and drops, so counting them here would double-report. func (c *sealedChunk) lostCount() uint64 { - if c.posted { + if c.state == chunkPosted { return 0 } return c.meta.DroppedCount + uint64(c.meta.RecordCount) diff --git a/packages/agentvault/session_log_test.go b/packages/agentvault/session_log_test.go index 849d7bb82..cf91700c7 100644 --- a/packages/agentvault/session_log_test.go +++ b/packages/agentvault/session_log_test.go @@ -642,7 +642,7 @@ func TestTheByteCapEvictsTheOldestChunkOnTheProxy(t *testing.T) { log, _, _ := newTestLog(&fakeShipper{}) blob := make([]byte, 12<<20) - add := func(sessionID string, order uint64, posted bool, carried uint64) *sessionLogSpool { + add := func(sessionID string, order uint64, state chunkState, carried uint64) *sessionLogSpool { spool, ok := log.spools[sessionID] if !ok { spool = newSessionLogSpool(testGrant(sessionID), log.now()) @@ -652,18 +652,18 @@ func TestTheByteCapEvictsTheOldestChunkOnTheProxy(t *testing.T) { meta: api.CreateAgentVaultSessionLogChunkRequest{ChunkID: fmt.Sprintf("c%d", order), RecordCount: 100, DroppedCount: carried}, ciphertext: blob, sealOrder: order, - posted: posted, + state: state, }) log.sealedBytes += len(blob) return spool } log.mu.Lock() - add("oldest", 0, false, 7) - add("posted", 1, true, 3) + add("oldest", 0, chunkSealed, 7) + add("posted", 1, chunkPosted, 3) var newest *sessionLogSpool for i := 2; i < 7; i++ { - newest = add(fmt.Sprintf("s%d", i), uint64(i), false, 0) + newest = add(fmt.Sprintf("s%d", i), uint64(i), chunkSealed, 0) } log.enforcePendingCapsLocked(newest) log.mu.Unlock() From 37633bab985f77f8b9bded9b79f7717c9404878c Mon Sep 17 00:00:00 2001 From: Saif Ur Rahman Date: Mon, 28 Sep 2026 20:02:13 +0530 Subject: [PATCH 38/43] fix(agent-vault): keep forgotten session log spools in order so a full list drops the oldest without a scan --- packages/agentvault/session_log.go | 68 +++++++++++++++++-------- packages/agentvault/session_log_test.go | 18 +++---- 2 files changed, 56 insertions(+), 30 deletions(-) diff --git a/packages/agentvault/session_log.go b/packages/agentvault/session_log.go index 777983251..aa9dc6560 100644 --- a/packages/agentvault/session_log.go +++ b/packages/agentvault/session_log.go @@ -1,6 +1,7 @@ package agentvault import ( + "container/list" "context" "sync" "sync/atomic" @@ -51,9 +52,50 @@ func newSessionLogGrant(sessionID string, key []byte) *sessionLogGrant { type forgottenSpool struct { nextSeq uint64 unreportedDrops uint64 - forgottenAt time.Time } +// Idle sessions in the order they were forgotten, so a full list drops the oldest without a scan. +type forgottenSpools struct { + capacity int + order *list.List // of forgottenEntry, oldest first + byID map[string]*list.Element +} + +type forgottenEntry struct { + sessionID string + spool forgottenSpool +} + +func newForgottenSpools(capacity int) *forgottenSpools { + return &forgottenSpools{capacity: capacity, order: list.New(), byID: make(map[string]*list.Element)} +} + +func (f *forgottenSpools) remember(sessionID string, spool forgottenSpool) { + f.drop(sessionID) + if f.order.Len() >= f.capacity { + f.drop(f.order.Front().Value.(forgottenEntry).sessionID) + } + f.byID[sessionID] = f.order.PushBack(forgottenEntry{sessionID: sessionID, spool: spool}) +} + +func (f *forgottenSpools) take(sessionID string) (forgottenSpool, bool) { + el, ok := f.byID[sessionID] + if !ok { + return forgottenSpool{}, false + } + f.drop(sessionID) + return el.Value.(forgottenEntry).spool, true +} + +func (f *forgottenSpools) drop(sessionID string) { + if el, ok := f.byID[sessionID]; ok { + f.order.Remove(el) + delete(f.byID, sessionID) + } +} + +func (f *forgottenSpools) len() int { return f.order.Len() } + type sessionLogShipper interface { createChunk(ctx context.Context, final bool, sessionID string, req api.CreateAgentVaultSessionLogChunkRequest) (api.CreateAgentVaultSessionLogChunkResponse, error) putObject(ctx context.Context, url string, ciphertext []byte) error @@ -107,7 +149,7 @@ type sessionLogRecorder struct { mu sync.Mutex spools map[string]*sessionLogSpool - forgotten map[string]forgottenSpool + forgotten *forgottenSpools unsealedRecords int sealedBytes int @@ -126,7 +168,7 @@ func newSessionLogRecorder(proxyID string, shipper sessionLogShipper) *sessionLo shipper: shipper, now: time.Now, spools: make(map[string]*sessionLogSpool), - forgotten: make(map[string]forgottenSpool), + forgotten: newForgottenSpools(sessionLogForgottenCapacity), wake: make(chan struct{}, 1), } } @@ -145,9 +187,8 @@ func (r *sessionLogRecorder) record(g *sessionLogGrant, rec sessionLogRecord) { spool, ok := r.spools[g.sessionID] if !ok { spool = newSessionLogSpool(g, r.now()) - if prior, ok := r.forgotten[g.sessionID]; ok { + if prior, ok := r.forgotten.take(g.sessionID); ok { spool.nextSeq, spool.ring.unreportedDrops = prior.nextSeq, prior.unreportedDrops - delete(r.forgotten, g.sessionID) } r.spools[g.sessionID] = spool } @@ -274,25 +315,10 @@ func (r *sessionLogRecorder) forgetIdleSpoolsLocked(now time.Time) { // Keeps the drop count too, so a spool forgotten while logging was off still reports what it lost. func (r *sessionLogRecorder) forgetSpoolLocked(sessionID string, spool *sessionLogSpool) { - if len(r.forgotten) >= sessionLogForgottenCapacity { - r.evictLongestForgottenLocked() - } - r.forgotten[sessionID] = forgottenSpool{nextSeq: spool.nextSeq, unreportedDrops: spool.ring.unreportedDrops, forgottenAt: r.now()} + r.forgotten.remember(sessionID, forgottenSpool{nextSeq: spool.nextSeq, unreportedDrops: spool.ring.unreportedDrops}) delete(r.spools, sessionID) } -func (r *sessionLogRecorder) evictLongestForgottenLocked() { - var oldestID string - var oldestAt time.Time - found := false - for id, prior := range r.forgotten { - if !found || prior.forgottenAt.Before(oldestAt) { - oldestID, oldestAt, found = id, prior.forgottenAt, true - } - } - delete(r.forgotten, oldestID) -} - func (r *sessionLogRecorder) dueSpools(pass flushPass, now time.Time) []*sessionLogSpool { r.mu.Lock() defer r.mu.Unlock() diff --git a/packages/agentvault/session_log_test.go b/packages/agentvault/session_log_test.go index cf91700c7..686992cc9 100644 --- a/packages/agentvault/session_log_test.go +++ b/packages/agentvault/session_log_test.go @@ -839,9 +839,9 @@ func TestSequenceNumbersSurviveASpoolBeingForgotten(t *testing.T) { func TestAFullForgottenListDropsOnlyTheLongestForgottenSession(t *testing.T) { log, _, _ := newTestLog(&fakeShipper{}) now := log.now() - log.forgotten["oldest"] = forgottenSpool{nextSeq: 1, forgottenAt: now.Add(-2 * time.Hour)} - for i := 1; i < maxSessionCacheEntries; i++ { - log.forgotten[fmt.Sprintf("s%d", i)] = forgottenSpool{nextSeq: 1, forgottenAt: now.Add(-time.Hour)} + log.forgotten.remember("oldest", forgottenSpool{nextSeq: 1}) + for i := 1; i < sessionLogForgottenCapacity; i++ { + log.forgotten.remember(fmt.Sprintf("s%d", i), forgottenSpool{nextSeq: 1}) } spool := newSessionLogSpool(testGrant("newest"), now) @@ -850,16 +850,16 @@ func TestAFullForgottenListDropsOnlyTheLongestForgottenSession(t *testing.T) { log.forgetSpoolLocked("newest", spool) log.mu.Unlock() - if len(log.forgotten) != maxSessionCacheEntries { - t.Fatalf("the forgotten list holds %d sessions, want the cap of %d", len(log.forgotten), maxSessionCacheEntries) + if log.forgotten.len() != sessionLogForgottenCapacity { + t.Fatalf("the forgotten list holds %d sessions, want the cap of %d", log.forgotten.len(), sessionLogForgottenCapacity) } - if _, ok := log.forgotten["oldest"]; ok { + if _, ok := log.forgotten.byID["oldest"]; ok { t.Fatal("the longest forgotten session was kept") } - if _, ok := log.forgotten["s1"]; !ok { + if _, ok := log.forgotten.byID["s1"]; !ok { t.Fatal("a more recently forgotten session was evicted too") } - if _, ok := log.forgotten["newest"]; !ok { + if _, ok := log.forgotten.byID["newest"]; !ok { t.Fatal("the session just forgotten was not kept") } } @@ -871,7 +871,7 @@ func TestASessionThatIsGoneDoesNotReserveItsSequenceNumbers(t *testing.T) { log.record(testGrant("s1"), aRecord("api.github.com")) tick() - if _, ok := log.forgotten["s1"]; ok { + if _, ok := log.forgotten.byID["s1"]; ok { t.Fatal("a session the server has forgotten is still holding a sequence number") } } From cd8e8bd8a541aba49e58494898f28375bac40359 Mon Sep 17 00:00:00 2001 From: Saif Ur Rahman Date: Tue, 29 Sep 2026 02:28:06 +0530 Subject: [PATCH 39/43] fix(agent-vault): let stop win over a due session log pass, ship sessions in parallel, and hold chunks on foreign 401s and TokenError --- packages/agentvault/cache.go | 17 +- packages/agentvault/cache_test.go | 5 +- packages/agentvault/proxy_gate_test.go | 2 +- packages/agentvault/session_log.go | 210 +++++++++++++++++------- packages/agentvault/session_log_ship.go | 9 +- packages/agentvault/session_log_test.go | 123 ++++++++++++-- 6 files changed, 285 insertions(+), 81 deletions(-) diff --git a/packages/agentvault/cache.go b/packages/agentvault/cache.go index f32b086f4..947c6a0a0 100644 --- a/packages/agentvault/cache.go +++ b/packages/agentvault/cache.go @@ -138,20 +138,25 @@ func isProxyTokenRejected(err error) bool { return errors.As(err, &apiErr) && apiErr.Name == proxyTokenRejectedName } -// The name on Infisical's own 404s. A route miss or a middlebox answers 404 under another name or none. -const infisicalNotFoundName = "NotFound" +// The names on Infisical's own 404s and 401s. A route miss, a middlebox or an auth proxy in front of Infisical +// answers under another name or none. +const ( + infisicalNotFoundName = "NotFound" + infisicalUnauthorizedName = "UnauthorizedError" +) // Resolve answers 200, 401 or 404 by contract, and 401 with a name when the proxy's own token is the -// problem. Only a 401 or a 404 named NotFound is a verdict on the session; any other 4xx, an unnamed 404 -// included, is a proxy-side fault or a middlebox, and the heartbeat classifier reads a 4xx the same way, so -// the two agree. A rejected proxy token is not a verdict on the session and is reported separately. +// problem. Only Infisical's own 401 (UnauthorizedError) or 404 (NotFound) is a verdict on the session; any +// other 4xx, an unnamed one included, is a proxy-side fault or a middlebox, and the heartbeat classifier reads a +// 4xx the same way, so the two agree. A rejected proxy token is not a verdict on the session and is reported +// separately. func isSessionGone(err error) bool { var apiErr *api.APIError if errors.As(err, &apiErr) { if isProxyTokenRejected(err) { return false } - return apiErr.StatusCode == http.StatusUnauthorized || + return (apiErr.StatusCode == http.StatusUnauthorized && apiErr.Name == infisicalUnauthorizedName) || (apiErr.StatusCode == http.StatusNotFound && apiErr.Name == infisicalNotFoundName) } return errors.Is(err, errSessionGone) diff --git a/packages/agentvault/cache_test.go b/packages/agentvault/cache_test.go index a01a2c822..d15abb588 100644 --- a/packages/agentvault/cache_test.go +++ b/packages/agentvault/cache_test.go @@ -97,7 +97,7 @@ func TestRefreshDropsAGoneSessionImmediately(t *testing.T) { t.Fatalf("get: %v", err) } - resolver.err = &api.APIError{StatusCode: status, Name: map[int]string{404: infisicalNotFoundName}[status]} + resolver.err = &api.APIError{StatusCode: status, Name: map[int]string{401: infisicalUnauthorizedName, 404: infisicalNotFoundName}[status]} cache.refresh() if len(cache.entries) != 0 { @@ -207,7 +207,8 @@ func TestRefreshTreatsOnlyTheContractRefusalsAsTerminal(t *testing.T) { name string kept bool }{ - {401, "", false}, {404, infisicalNotFoundName, false}, + {401, infisicalUnauthorizedName, false}, {404, infisicalNotFoundName, false}, + {401, "", true}, {401, "Unauthorized", true}, {404, "", true}, {404, "Not Found", true}, {400, "", true}, {403, "", true}, {405, "", true}, {407, "", true}, {422, "", true}, {408, "", true}, {429, "", true}, {500, "", true}, {502, "", true}, diff --git a/packages/agentvault/proxy_gate_test.go b/packages/agentvault/proxy_gate_test.go index ffc088bde..c5e1d21ac 100644 --- a/packages/agentvault/proxy_gate_test.go +++ b/packages/agentvault/proxy_gate_test.go @@ -24,7 +24,7 @@ func TestGateDenialsAreLogged(t *testing.T) { level string decision string }{ - {"revoked or expired session", &api.APIError{StatusCode: 401}, 403, "warn", decisionBlocked}, + {"revoked or expired session", &api.APIError{StatusCode: 401, Name: infisicalUnauthorizedName}, 403, "warn", decisionBlocked}, {"infisical unreachable", errors.New("dial tcp: connection refused"), 502, "error", decisionError}, } { t.Run(tc.name, func(t *testing.T) { diff --git a/packages/agentvault/session_log.go b/packages/agentvault/session_log.go index aa9dc6560..3a6d91fc1 100644 --- a/packages/agentvault/session_log.go +++ b/packages/agentvault/session_log.go @@ -3,6 +3,7 @@ package agentvault import ( "container/list" "context" + "sort" "sync" "sync/atomic" "time" @@ -30,6 +31,8 @@ const ( sessionLogFinalTimeout = 3 * time.Second sessionLogCloseTimeout = 5 * time.Second sessionLogUploadURLMargin = 10 * time.Second // get a fresh link if the current one expires within this + + sessionLogShipParallelism = 8 // chunks sent at once, one per session, like refreshParallelism ) type sessionLogGrant struct { @@ -241,17 +244,31 @@ func (r *sessionLogRecorder) run(stop <-chan struct{}) { if r == nil { return } + // Cancelled on stop, so a pass in flight ends at once and close() can start the final flush. + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + go func() { + <-stop + cancel() + }() + ticker := time.NewTicker(sessionLogFlushInterval) defer ticker.Stop() for { + // Checked on its own first: select picks at random among ready cases, so a due tick could win over stop. + select { + case <-stop: + return + default: + } select { case <-stop: return case <-ticker.C: - r.flush(context.Background(), flushTick) + r.flush(ctx, flushTick) case <-r.wake: - r.flush(context.Background(), flushWake) + r.flush(ctx, flushWake) } } } @@ -300,8 +317,19 @@ func (r *sessionLogRecorder) flush(ctx context.Context, pass flushPass) { r.forgetIdleSpoolsLocked(started) r.mu.Unlock() } - for _, spool := range r.dueSpools(pass, started) { - r.flushSpool(ctx, spool, pass, started) + due := r.dueSpools(pass, started) + // Everything is sealed before anything is sent, so evicting for the byte cap never hits a chunk in flight. + for _, spool := range due { + r.sealRing(spool, started) + } + + stopped := make(map[*sessionLogSpool]bool) + for ctx.Err() == nil { + batch := r.nextShipments(due, stopped) + if len(batch) == 0 { + return + } + r.applyShipments(ctx, r.sendShipments(ctx, batch, pass), stopped) } } @@ -438,83 +466,147 @@ func (r *sessionLogRecorder) evictOldestLocked(spool *sessionLogSpool) { Msg("agent-vault: dropped an unshipped session log chunk, the buffer is full") } -func (r *sessionLogRecorder) flushSpool(ctx context.Context, spool *sessionLogSpool, pass flushPass, started time.Time) { - r.sealRing(spool, started) +// One chunk picked for a round, with the upload link it held when picked. +type shipment struct { + spool *sessionLogSpool + chunk *sealedChunk + uploadURL string + urlExpires time.Time +} - for { - r.mu.Lock() - if len(spool.pending) == 0 || !r.hold.canShip(r.now()) { - r.mu.Unlock() - return - } - chunk := spool.pending[0] - r.mu.Unlock() +// What the network said about one shipment. Only applyShipments turns it into state. +type shipmentResult struct { + shipment - if !r.shipChunk(ctx, spool, chunk, pass) { - return - } + skipped bool - r.mu.Lock() - r.discardHeadLocked(spool, chunk, false) - r.mu.Unlock() - } + posted bool + postedURL string + postedExpires time.Time + grantsIssuedAtSend uint64 + createErr error + + putErr error } -func (r *sessionLogRecorder) shipChunk(ctx context.Context, spool *sessionLogSpool, chunk *sealedChunk, pass flushPass) bool { - uploadURL, ok := r.ensureUploadLink(ctx, spool, chunk, pass) - if !ok { - return false - } - return r.upload(ctx, spool, chunk, pass, uploadURL) +func (res shipmentResult) shipped() bool { + return !res.skipped && res.createErr == nil && res.putErr == nil } -func (r *sessionLogRecorder) ensureUploadLink(ctx context.Context, spool *sessionLogSpool, chunk *sealedChunk, pass flushPass) (uploadURL string, ok bool) { +// The next chunk of up to sessionLogShipParallelism sessions. A session that failed this pass is left for the next. +func (r *sessionLogRecorder) nextShipments(due []*sessionLogSpool, stopped map[*sessionLogSpool]bool) []shipment { r.mu.Lock() - uploadURL, urlExpires := chunk.uploadURL, chunk.urlExpires - r.mu.Unlock() - if uploadURL != "" && !r.now().Add(sessionLogUploadURLMargin).After(urlExpires) { - return uploadURL, true + defer r.mu.Unlock() + if !r.hold.canShip(r.now()) { + return nil } - // Past the shutdown budget, a new row could only be written for an upload that can no longer happen. - if ctx.Err() != nil { - return "", false - } - grantsIssuedAtSend := sessionLogGrantsIssued.Load() - res, err := r.shipper.createChunk(ctx, pass == flushFinal, spool.sessionID, chunk.meta) - if err != nil { - r.handleCreateFailure(spool, chunk, err, grantsIssuedAtSend) - return "", false + batch := make([]shipment, 0, sessionLogShipParallelism) + for _, spool := range due { + if stopped[spool] || len(spool.pending) == 0 { + continue + } + chunk := spool.pending[0] + batch = append(batch, shipment{spool: spool, chunk: chunk, uploadURL: chunk.uploadURL, urlExpires: chunk.urlExpires}) + if len(batch) == sessionLogShipParallelism { + break + } } + return batch +} - r.mu.Lock() - defer r.mu.Unlock() - r.clockSkewReported = false - chunk.state = chunkPosted - chunk.uploadURL = res.UploadURL - chunk.urlExpires = r.now().Add(time.Duration(res.ExpiresInSeconds) * time.Second) - return chunk.uploadURL, true +// Network only: the recorder's state is never touched here, so the rest of it stays single-threaded. +func (r *sessionLogRecorder) sendShipments(ctx context.Context, batch []shipment, pass flushPass) []shipmentResult { + results := make([]shipmentResult, len(batch)) + var wg sync.WaitGroup + for i, next := range batch { + wg.Add(1) + go func(i int, next shipment) { + defer wg.Done() + results[i] = r.sendShipment(ctx, next, pass) + }(i, next) + } + wg.Wait() + return results } -func (r *sessionLogRecorder) upload(ctx context.Context, spool *sessionLogSpool, chunk *sealedChunk, pass flushPass, uploadURL string) bool { +func (r *sessionLogRecorder) sendShipment(ctx context.Context, next shipment, pass flushPass) shipmentResult { + res := shipmentResult{shipment: next} + final := pass == flushFinal + + uploadURL := next.uploadURL + if uploadURL == "" || r.now().Add(sessionLogUploadURLMargin).After(next.urlExpires) { + // Past the shutdown budget, a new row could only be written for an upload that can no longer happen. + if ctx.Err() != nil { + res.skipped = true + return res + } + res.grantsIssuedAtSend = sessionLogGrantsIssued.Load() + created, err := r.shipper.createChunk(ctx, final, next.spool.sessionID, next.chunk.meta) + if err != nil { + res.createErr = err + return res + } + res.posted = true + res.postedURL = created.UploadURL + res.postedExpires = r.now().Add(time.Duration(created.ExpiresInSeconds) * time.Second) + uploadURL = created.UploadURL + } + putCtx := ctx - if pass != flushFinal { + if !final { var cancel context.CancelFunc putCtx, cancel = context.WithTimeout(ctx, sessionLogPutTimeout) defer cancel() } + res.putErr = r.shipper.putObject(putCtx, uploadURL, next.chunk.ciphertext) + return res +} - if err := r.shipper.putObject(putCtx, uploadURL, chunk.ciphertext); err != nil { - r.mu.Lock() - chunk.uploadURL = "" - r.hold.s3Down = true - r.mu.Unlock() - log.Warn().Err(err).Str("sessionId", spool.sessionID).Str("chunkId", chunk.meta.ChunkID). - Msg("agent-vault: could not upload a session log chunk, will retry") - return false +func (r *sessionLogRecorder) applyShipments(ctx context.Context, results []shipmentResult, stopped map[*sessionLogSpool]bool) { + // Rows first, then successes, then failures: a refusal can switch logging off and discard every queue, + // and a chunk that already has its row, or has landed, must not be counted as dropped by that. + r.mu.Lock() + for _, res := range results { + if res.posted { + r.clockSkewReported = false + res.chunk.state = chunkPosted + res.chunk.uploadURL = res.postedURL + res.chunk.urlExpires = res.postedExpires + } } + r.mu.Unlock() + sort.SliceStable(results, func(i, j int) bool { return results[i].shipped() && !results[j].shipped() }) - return true + for _, res := range results { + if res.shipped() { + r.mu.Lock() + r.discardHeadLocked(res.spool, res.chunk, false) + r.mu.Unlock() + continue + } + stopped[res.spool] = true + + switch { + case res.skipped: + case res.createErr != nil && ctx.Err() != nil: + // Cancelled by shutdown: the chunk stays queued for the final flush, which is no outage to report. + r.mu.Lock() + r.hold.infisicalDown = true + r.mu.Unlock() + case res.createErr != nil: + r.handleCreateFailure(res.spool, res.chunk, res.createErr, res.grantsIssuedAtSend) + default: + r.mu.Lock() + res.chunk.uploadURL = "" + r.hold.s3Down = true + r.mu.Unlock() + if ctx.Err() == nil { + log.Warn().Err(res.putErr).Str("sessionId", res.spool.sessionID).Str("chunkId", res.chunk.meta.ChunkID). + Msg("agent-vault: could not upload a session log chunk, will retry") + } + } + } } func (r *sessionLogRecorder) handleCreateFailure(spool *sessionLogSpool, chunk *sealedChunk, err error, grantsIssuedAtSend uint64) { diff --git a/packages/agentvault/session_log_ship.go b/packages/agentvault/session_log_ship.go index 17440bb59..d071297fc 100644 --- a/packages/agentvault/session_log_ship.go +++ b/packages/agentvault/session_log_ship.go @@ -21,6 +21,8 @@ const ( sessionLogCeilingReachedName = "AgentVaultSessionLogCeilingReached" sessionLogDisabledName = "AgentVaultSessionLogDisabled" sessionLogClockSkewName = "AgentVaultSessionLogClockSkew" + // Infisical's 403 for a proxy JWT that no longer verifies, e.g. after its signing secret rotated. + sessionLogTokenErrorName = "TokenError" ) type chunkRefusal int @@ -37,7 +39,7 @@ const ( func classifyChunkError(err error) chunkRefusal { switch { - case isProxyTokenRejected(err): + case isProxyTokenRejected(err), isSessionLogErrorNamed(err, sessionLogTokenErrorName): return chunkTokenRejected case isSessionGone(err): return chunkSessionGone @@ -53,9 +55,10 @@ func classifyChunkError(err error) chunkRefusal { if !errors.As(err, &apiErr) { return chunkRetry } - // Infisical's own NotFound is caught earlier as a gone session, so a 404 here is a route miss, as during a rollback. + // Infisical's own 401s and NotFound are caught above, so a 401 or 404 here came from a route miss (as during + // a rollback) or something in front of Infisical. if apiErr.StatusCode == http.StatusRequestTimeout || apiErr.StatusCode == http.StatusTooManyRequests || - apiErr.StatusCode == http.StatusNotFound { + apiErr.StatusCode == http.StatusNotFound || apiErr.StatusCode == http.StatusUnauthorized { return chunkRetry } if apiErr.StatusCode >= 400 && apiErr.StatusCode < 500 { diff --git a/packages/agentvault/session_log_test.go b/packages/agentvault/session_log_test.go index 686992cc9..d2c75628b 100644 --- a/packages/agentvault/session_log_test.go +++ b/packages/agentvault/session_log_test.go @@ -43,9 +43,12 @@ type fakeShipper struct { putDefault error nextURL int + + delay time.Duration } func (f *fakeShipper) createChunk(_ context.Context, final bool, sessionID string, req api.CreateAgentVaultSessionLogChunkRequest) (api.CreateAgentVaultSessionLogChunkResponse, error) { + time.Sleep(f.delay) f.mu.Lock() defer f.mu.Unlock() @@ -68,6 +71,7 @@ func (f *fakeShipper) createChunk(_ context.Context, final bool, sessionID strin } func (f *fakeShipper) putObject(_ context.Context, url string, ciphertext []byte) error { + time.Sleep(f.delay) f.mu.Lock() defer f.mu.Unlock() @@ -691,18 +695,20 @@ func TestTheTickBreakerStopsHammeringADeadBucket(t *testing.T) { shipper := &fakeShipper{putDefault: errors.New("i/o timeout")} log, _, tick := newTestLog(shipper) - for i := 0; i < 5; i++ { + sessions := 2*sessionLogShipParallelism + 1 + for i := 0; i < sessions; i++ { log.record(testGrant(fmt.Sprintf("s%d", i)), aRecord("api.github.com")) } tick() - if got := len(shipper.puts()); got != 1 { - t.Fatalf("%d uploads were attempted in one tick after the first failed", got) + // One round goes out at once; the breaker has to stop the rounds after it. + if got := len(shipper.puts()); got != sessionLogShipParallelism { + t.Fatalf("%d uploads were attempted in one tick, want one round of %d", got, sessionLogShipParallelism) } - if got := len(shipper.posts()); got != 1 { - t.Fatalf("%d rows were written for objects that could not be uploaded", got) + if got := len(shipper.posts()); got != sessionLogShipParallelism { + t.Fatalf("%d rows were written for objects that could not be uploaded, want one round of %d", got, sessionLogShipParallelism) } - for i := 0; i < 5; i++ { + for i := 0; i < sessions; i++ { if len(log.spools[fmt.Sprintf("s%d", i)].pending) == 0 { t.Fatalf("spool s%d sealed nothing during the outage", i) } @@ -902,15 +908,16 @@ func TestAnUnreachableControlPlaneStopsTheTickAfterOneTimeout(t *testing.T) { shipper := &fakeShipper{postDefault: scriptedResult{err: errors.New("i/o timeout")}} log, _, tick := newTestLog(shipper) - for i := 0; i < 5; i++ { + sessions := 2*sessionLogShipParallelism + 1 + for i := 0; i < sessions; i++ { log.record(testGrant(fmt.Sprintf("s%d", i)), aRecord("api.github.com")) } tick() - if got := len(shipper.posts()); got != 1 { - t.Fatalf("%d chunk POSTs were attempted in one tick after the first timed out", got) + if got := len(shipper.posts()); got != sessionLogShipParallelism { + t.Fatalf("%d chunk POSTs were attempted in one tick, want one round of %d", got, sessionLogShipParallelism) } - for i := 0; i < 5; i++ { + for i := 0; i < sessions; i++ { if len(log.spools[fmt.Sprintf("s%d", i)].pending) == 0 { t.Fatalf("spool s%d sealed nothing during the outage", i) } @@ -1023,6 +1030,7 @@ func TestServerErrorsAreRetriedAndBadChunksAreDropped(t *testing.T) { {http.StatusTooManyRequests, true}, {http.StatusRequestTimeout, true}, {http.StatusNotFound, true}, + {http.StatusUnauthorized, true}, {http.StatusUnprocessableEntity, false}, {http.StatusConflict, false}, } { @@ -1210,3 +1218,98 @@ func TestAWakeShipsOnlyFullRings(t *testing.T) { t.Fatalf("s1 holds %d records after the wake, expected it to wait for the tick", got) } } + +func TestAPassShipsSessionsInParallel(t *testing.T) { + shipper := &fakeShipper{delay: 100 * time.Millisecond} + log, _, tick := newTestLog(shipper) + + for i := 0; i < sessionLogShipParallelism; i++ { + log.record(testGrant(fmt.Sprintf("s%d", i)), aRecord("api.github.com")) + } + started := time.Now() + tick() + took := time.Since(started) + + if got := len(shipper.puts()); got != sessionLogShipParallelism { + t.Fatalf("%d chunks were uploaded, want %d", got, sessionLogShipParallelism) + } + // In series this is 16 calls of 100ms; one round is two. + if took > 800*time.Millisecond { + t.Fatalf("shipping %d sessions took %s, which is serial; want about one round trip", sessionLogShipParallelism, took) + } +} + +func TestStopWinsOverAReadyWake(t *testing.T) { + shipper := &fakeShipper{} + log, _, _ := newTestLog(shipper) + grant := testGrant("s1") + for i := 0; i < sessionLogFlushRecords; i++ { + log.record(grant, aRecord("api.github.com")) + } + + stop := make(chan struct{}) + close(stop) + log.run(stop) + + if got := len(shipper.posts()); got != 0 { + t.Fatalf("a pass started after stop: %d chunks were posted", got) + } +} + +type passBlockingShipper struct { + fakeShipper + entered chan struct{} +} + +func (b *passBlockingShipper) createChunk(ctx context.Context, final bool, sessionID string, req api.CreateAgentVaultSessionLogChunkRequest) (api.CreateAgentVaultSessionLogChunkResponse, error) { + if !final { + close(b.entered) + <-ctx.Done() + return api.CreateAgentVaultSessionLogChunkResponse{}, ctx.Err() + } + return b.fakeShipper.createChunk(ctx, final, sessionID, req) +} + +func TestStopCancelsAPassInFlightSoTheFinalFlushShipsIt(t *testing.T) { + shipper := &passBlockingShipper{entered: make(chan struct{})} + log, _, _ := newTestLog(shipper) + grant := testGrant("s1") + for i := 0; i < sessionLogFlushRecords; i++ { + log.record(grant, aRecord("api.github.com")) + } + + stop := make(chan struct{}) + done := make(chan struct{}) + go func() { + log.run(stop) + close(done) + }() + <-shipper.entered + close(stop) + + select { + case <-done: + case <-time.After(2 * time.Second): + t.Fatal("the pass in flight was not cancelled by stop") + } + + ctx, cancel := context.WithTimeout(context.Background(), sessionLogCloseTimeout) + defer cancel() + log.close(ctx) + if got := len(shipper.puts()); got != 1 { + t.Fatalf("the final flush uploaded %d chunks, want the one the cancelled pass left", got) + } +} + +func TestATokenErrorHoldsTheChunk(t *testing.T) { + shipper := &fakeShipper{postResults: []scriptedResult{{err: apiErr(http.StatusForbidden, sessionLogTokenErrorName)}}} + log, _, tick := newTestLog(shipper) + + log.record(testGrant("s1"), aRecord("api.github.com")) + tick() + + spool := log.spools["s1"] + if len(spool.pending) != 1 || spool.ring.unreportedDrops != 0 { + t.Fatalf("a TokenError dropped the chunk (pending %d, dropped %d); it should be held until the proxy logs in again", len(spool.pending), spool.ring.unreportedDrops) + } +} From 93d56dd5db8cad34c6cec982658cda76e80fd803 Mon Sep 17 00:00:00 2001 From: Saif Ur Rahman Date: Tue, 29 Sep 2026 07:18:09 +0530 Subject: [PATCH 40/43] fix(agent-vault): seal session log chunks a round at a time, so a healthy pass never evicts chunks it just sealed --- packages/agentvault/session_log.go | 105 ++++++++++++++++-------- packages/agentvault/session_log_test.go | 48 +++++++++++ 2 files changed, 118 insertions(+), 35 deletions(-) diff --git a/packages/agentvault/session_log.go b/packages/agentvault/session_log.go index 3a6d91fc1..cc3f022da 100644 --- a/packages/agentvault/session_log.go +++ b/packages/agentvault/session_log.go @@ -318,19 +318,23 @@ func (r *sessionLogRecorder) flush(ctx context.Context, pass flushPass) { r.mu.Unlock() } due := r.dueSpools(pass, started) - // Everything is sealed before anything is sent, so evicting for the byte cap never hits a chunk in flight. - for _, spool := range due { - r.sealRing(spool, started) - } + // Sealed a round at a time, between rounds, so a healthy pass never holds more than one round against the + // byte cap, and evicting for it never hits a chunk in flight. stopped := make(map[*sessionLogSpool]bool) for ctx.Err() == nil { + r.sealNextRound(due, stopped, started) batch := r.nextShipments(due, stopped) if len(batch) == 0 { - return + break } r.applyShipments(ctx, r.sendShipments(ctx, batch, pass), stopped) } + + // Whatever this pass couldn't ship waits sealed, where the byte cap drops the oldest first. + for _, spool := range due { + r.sealRing(spool, started) + } } func (r *sessionLogRecorder) forgetIdleSpoolsLocked(now time.Time) { @@ -380,46 +384,77 @@ func (r *sessionLogRecorder) isDueLocked(spool *sessionLogSpool, pass flushPass, return now.Sub(spool.lastFlushAt) >= sessionLogFlushInterval-sessionLogFlushSlack } -func (r *sessionLogRecorder) sealRing(spool *sessionLogSpool, started time.Time) { - for { +// The next slice of up to sessionLogShipParallelism sessions, sealed only where nothing is queued, so each +// session has a chunk for the round that is about to ship. +func (r *sessionLogRecorder) sealNextRound(due []*sessionLogSpool, stopped map[*sessionLogSpool]bool, started time.Time) { + ready := 0 + for _, spool := range due { + if ready == sessionLogShipParallelism { + return + } + if stopped[spool] { + continue + } r.mu.Lock() - records := spool.ring.drain(sessionLogFlushRecords) - if len(records) == 0 { - spool.lastFlushAt = started + queued := len(spool.pending) > 0 + r.mu.Unlock() + if !queued { + r.sealNextSlice(spool, started) + r.mu.Lock() + queued = len(spool.pending) > 0 r.mu.Unlock() - return } - r.unsealedRecords -= len(records) - dropped := spool.ring.takeUnreportedDrops() + if queued { + ready++ + } + } +} + +func (r *sessionLogRecorder) sealRing(spool *sessionLogSpool, started time.Time) { + for r.sealNextSlice(spool, started) { + } +} + +// False once the ring is empty, which is when the spool counts as flushed. +func (r *sessionLogRecorder) sealNextSlice(spool *sessionLogSpool, started time.Time) bool { + r.mu.Lock() + records := spool.ring.drain(sessionLogFlushRecords) + if len(records) == 0 { + spool.lastFlushAt = started r.mu.Unlock() + return false + } + r.unsealedRecords -= len(records) + dropped := spool.ring.takeUnreportedDrops() + r.mu.Unlock() + + groups, err := packSessionLogRecords(records) + if err != nil { + r.dropUnsealed(spool, len(records), dropped, err) + return true + } + for i, group := range groups { + var groupDropped uint64 + // Only the first chunk of a split batch carries the drop count, so drops aren't reported twice. + if i == 0 { + groupDropped = dropped + } - groups, err := packSessionLogRecords(records) + chunk, err := spool.sealSlice(group.records, group.plaintext, groupDropped) if err != nil { - r.dropUnsealed(spool, len(records), dropped, err) + r.dropUnsealed(spool, len(group.records), groupDropped, err) continue } - for i, group := range groups { - var groupDropped uint64 - // Only the first chunk of a split batch carries the drop count, so drops aren't reported twice. - if i == 0 { - groupDropped = dropped - } - chunk, err := spool.sealSlice(group.records, group.plaintext, groupDropped) - if err != nil { - r.dropUnsealed(spool, len(group.records), groupDropped, err) - continue - } - - r.mu.Lock() - chunk.sealOrder = r.nextSealOrder - r.nextSealOrder++ - spool.pending = append(spool.pending, chunk) - r.sealedBytes += len(chunk.ciphertext) - r.enforcePendingCapsLocked(spool) - r.mu.Unlock() - } + r.mu.Lock() + chunk.sealOrder = r.nextSealOrder + r.nextSealOrder++ + spool.pending = append(spool.pending, chunk) + r.sealedBytes += len(chunk.ciphertext) + r.enforcePendingCapsLocked(spool) + r.mu.Unlock() } + return true } func (r *sessionLogRecorder) dropUnsealed(spool *sessionLogSpool, records int, dropped uint64, err error) { diff --git a/packages/agentvault/session_log_test.go b/packages/agentvault/session_log_test.go index d2c75628b..a84119e17 100644 --- a/packages/agentvault/session_log_test.go +++ b/packages/agentvault/session_log_test.go @@ -1239,6 +1239,54 @@ func TestAPassShipsSessionsInParallel(t *testing.T) { } } +type observingShipper struct { + *fakeShipper + onPost func() +} + +func (o *observingShipper) createChunk(ctx context.Context, final bool, sessionID string, req api.CreateAgentVaultSessionLogChunkRequest) (api.CreateAgentVaultSessionLogChunkResponse, error) { + o.onPost() + return o.fakeShipper.createChunk(ctx, final, sessionID, req) +} + +// Sealing every busy session before the first upload could pass the byte cap and evict chunks while the +// bucket is fine, so a healthy pass holds no more than the round it is shipping. +func TestAHealthyPassSealsOnlyTheRoundItShips(t *testing.T) { + var log *sessionLogRecorder + var mostQueued int + shipper := &observingShipper{fakeShipper: &fakeShipper{}, onPost: func() { + log.mu.Lock() + defer log.mu.Unlock() + queued := 0 + for _, spool := range log.spools { + queued += len(spool.pending) + } + mostQueued = max(mostQueued, queued) + }} + log, _, tick := newTestLog(shipper) + + sessions := 3 * sessionLogShipParallelism + for i := 0; i < sessions; i++ { + grant := testGrant(fmt.Sprintf("s%d", i)) + for j := 0; j < 2*sessionLogFlushRecords; j++ { + log.record(grant, aRecord("api.github.com")) + } + } + tick() + + if mostQueued > sessionLogShipParallelism { + t.Fatalf("%d chunks were sealed while a round was shipping, want at most one round of %d", mostQueued, sessionLogShipParallelism) + } + if got, want := len(shipper.puts()), 2*sessions; got != want { + t.Fatalf("%d chunks were uploaded, want %d", got, want) + } + for id, spool := range log.spools { + if spool.heldRecords() != 0 || spool.ring.unreportedDrops != 0 { + t.Fatalf("spool %s still holds %d records and %d drops after a healthy pass", id, spool.heldRecords(), spool.ring.unreportedDrops) + } + } +} + func TestStopWinsOverAReadyWake(t *testing.T) { shipper := &fakeShipper{} log, _, _ := newTestLog(shipper) From 0c4691d6bc4e714023fdcfba4a0337e8533672ea Mon Sep 17 00:00:00 2001 From: Saif Ur Rahman Date: Tue, 29 Sep 2026 08:15:15 +0530 Subject: [PATCH 41/43] fix(agent-vault): seal only what a session log pass started with, so requests arriving mid-upload wait for the next pass instead of shipping one per round --- packages/agentvault/session_log.go | 44 +++++++++++------- packages/agentvault/session_log_test.go | 59 +++++++++++++++++++++++++ 2 files changed, 88 insertions(+), 15 deletions(-) diff --git a/packages/agentvault/session_log.go b/packages/agentvault/session_log.go index cc3f022da..df8dde635 100644 --- a/packages/agentvault/session_log.go +++ b/packages/agentvault/session_log.go @@ -318,12 +318,15 @@ func (r *sessionLogRecorder) flush(ctx context.Context, pass flushPass) { r.mu.Unlock() } due := r.dueSpools(pass, started) + // Only what each ring held at the start, so requests arriving mid-pass wait for the next wake or tick instead + // of going out one tiny chunk per round and keeping the pass open. + unsealed := r.startSealing(due, started) // Sealed a round at a time, between rounds, so a healthy pass never holds more than one round against the // byte cap, and evicting for it never hits a chunk in flight. stopped := make(map[*sessionLogSpool]bool) for ctx.Err() == nil { - r.sealNextRound(due, stopped, started) + r.sealNextRound(due, stopped, unsealed) batch := r.nextShipments(due, stopped) if len(batch) == 0 { break @@ -331,9 +334,10 @@ func (r *sessionLogRecorder) flush(ctx context.Context, pass flushPass) { r.applyShipments(ctx, r.sendShipments(ctx, batch, pass), stopped) } - // Whatever this pass couldn't ship waits sealed, where the byte cap drops the oldest first. + // Only a failed or cancelled round leaves any: they wait sealed, where the byte cap drops the oldest first. for _, spool := range due { - r.sealRing(spool, started) + for unsealed[spool] > 0 && r.sealNext(spool, unsealed) { + } } } @@ -384,9 +388,22 @@ func (r *sessionLogRecorder) isDueLocked(spool *sessionLogSpool, pass flushPass, return now.Sub(spool.lastFlushAt) >= sessionLogFlushInterval-sessionLogFlushSlack } +// How many records each due ring holds as the pass starts, which is all the pass seals. The spools count as +// flushed now, so the next tick is a full interval away however long the pass takes. +func (r *sessionLogRecorder) startSealing(due []*sessionLogSpool, started time.Time) map[*sessionLogSpool]int { + r.mu.Lock() + defer r.mu.Unlock() + unsealed := make(map[*sessionLogSpool]int, len(due)) + for _, spool := range due { + unsealed[spool] = spool.ring.len() + spool.lastFlushAt = started + } + return unsealed +} + // The next slice of up to sessionLogShipParallelism sessions, sealed only where nothing is queued, so each // session has a chunk for the round that is about to ship. -func (r *sessionLogRecorder) sealNextRound(due []*sessionLogSpool, stopped map[*sessionLogSpool]bool, started time.Time) { +func (r *sessionLogRecorder) sealNextRound(due []*sessionLogSpool, stopped map[*sessionLogSpool]bool, unsealed map[*sessionLogSpool]int) { ready := 0 for _, spool := range due { if ready == sessionLogShipParallelism { @@ -398,8 +415,8 @@ func (r *sessionLogRecorder) sealNextRound(due []*sessionLogSpool, stopped map[* r.mu.Lock() queued := len(spool.pending) > 0 r.mu.Unlock() - if !queued { - r.sealNextSlice(spool, started) + if !queued && unsealed[spool] > 0 { + r.sealNext(spool, unsealed) r.mu.Lock() queued = len(spool.pending) > 0 r.mu.Unlock() @@ -410,20 +427,17 @@ func (r *sessionLogRecorder) sealNextRound(due []*sessionLogSpool, stopped map[* } } -func (r *sessionLogRecorder) sealRing(spool *sessionLogSpool, started time.Time) { - for r.sealNextSlice(spool, started) { - } -} - -// False once the ring is empty, which is when the spool counts as flushed. -func (r *sessionLogRecorder) sealNextSlice(spool *sessionLogSpool, started time.Time) bool { +// Seals the next slice of what the pass took on for this spool. False once the ring has none of it left, which +// happens early only if session logs were switched off mid-pass and the ring was discarded. +func (r *sessionLogRecorder) sealNext(spool *sessionLogSpool, unsealed map[*sessionLogSpool]int) bool { r.mu.Lock() - records := spool.ring.drain(sessionLogFlushRecords) + records := spool.ring.drain(min(unsealed[spool], sessionLogFlushRecords)) if len(records) == 0 { - spool.lastFlushAt = started + unsealed[spool] = 0 r.mu.Unlock() return false } + unsealed[spool] -= len(records) r.unsealedRecords -= len(records) dropped := spool.ring.takeUnreportedDrops() r.mu.Unlock() diff --git a/packages/agentvault/session_log_test.go b/packages/agentvault/session_log_test.go index a84119e17..392637f28 100644 --- a/packages/agentvault/session_log_test.go +++ b/packages/agentvault/session_log_test.go @@ -1287,6 +1287,65 @@ func TestAHealthyPassSealsOnlyTheRoundItShips(t *testing.T) { } } +// Resealing whatever arrived since the last round would ship a busy session one request per round, and never end +// the pass while its agent keeps working. +func TestRequestsArrivingDuringAPassWaitForTheNext(t *testing.T) { + var log *sessionLogRecorder + grant := testGrant("s1") + shipper := &observingShipper{fakeShipper: &fakeShipper{}, onPost: func() { + log.record(grant, aRecord("api.github.com")) + }} + log, _, tick := newTestLog(shipper) + + for i := 0; i < 10; i++ { + log.record(grant, aRecord("api.github.com")) + } + tick() + + posts := shipper.posts() + if len(posts) != 1 || posts[0].records != 10 { + t.Fatalf("the pass sent %d chunks, want one with the 10 records it started with", len(posts)) + } + if got := log.spools["s1"].ring.len(); got != 1 { + t.Fatalf("the ring holds %d records, want the 1 that arrived during the upload", got) + } +} + +func TestAWakeShipsWhatArrivedDuringThePassBeforeIt(t *testing.T) { + var log *sessionLogRecorder + var once sync.Once + grant := testGrant("s1") + shipper := &observingShipper{fakeShipper: &fakeShipper{}, onPost: func() { + once.Do(func() { + for i := 0; i < sessionLogFlushRecords; i++ { + log.record(grant, aRecord("api.github.com")) + } + }) + }} + log, _, tick := newTestLog(shipper) + + for i := 0; i < sessionLogFlushRecords; i++ { + log.record(grant, aRecord("api.github.com")) + } + <-log.wake + tick() + if got := len(shipper.puts()); got != 1 { + t.Fatalf("the tick uploaded %d chunks, want only the one it started with", got) + } + if len(log.wake) != 1 { + t.Fatal("the records that arrived during the upload didn't queue a wake") + } + + <-log.wake + log.flush(context.Background(), flushWake) + if got := len(shipper.puts()); got != 2 { + t.Fatalf("%d chunks were uploaded after the wake, want 2", got) + } + if held := log.spools["s1"].heldRecords(); held != 0 { + t.Fatalf("the spool still holds %d records after the wake", held) + } +} + func TestStopWinsOverAReadyWake(t *testing.T) { shipper := &fakeShipper{} log, _, _ := newTestLog(shipper) From 810aec6f3307d8ac39c00682ca0cf21dbac1341c Mon Sep 17 00:00:00 2001 From: Saif Ur Rahman Date: Tue, 29 Sep 2026 09:04:10 +0530 Subject: [PATCH 42/43] fix(agent-vault): say session logs are disabled, without naming a project, when the proxy drops what it held --- packages/agentvault/session_log.go | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/packages/agentvault/session_log.go b/packages/agentvault/session_log.go index df8dde635..98523af9f 100644 --- a/packages/agentvault/session_log.go +++ b/packages/agentvault/session_log.go @@ -718,7 +718,7 @@ func (r *sessionLogRecorder) switchOff(grantsIssuedAtSend uint64) { if !r.hold.off { log.Warn().Int("records", lost). - Msg("agent-vault: session logs are off for this project, dropping what was held until they are back on") + Msg("agent-vault: session logs are disabled, dropping what was held until they're turned back on") } r.hold.off = true r.hold.offThrough = grantsIssuedAtSend From 0689d1547ca8a7d0f346bfca7024a5e057dbf920 Mon Sep 17 00:00:00 2001 From: Saif Ur Rahman Date: Tue, 29 Sep 2026 23:12:51 +0530 Subject: [PATCH 43/43] fix(agent-vault): ship every session in one round at shutdown, so one stuck upload can't hold the rest back --- packages/agentvault/session_log.go | 27 +++++--- packages/agentvault/session_log_test.go | 89 +++++++++++++++++++++++++ 2 files changed, 106 insertions(+), 10 deletions(-) diff --git a/packages/agentvault/session_log.go b/packages/agentvault/session_log.go index 98523af9f..92c85805a 100644 --- a/packages/agentvault/session_log.go +++ b/packages/agentvault/session_log.go @@ -33,6 +33,9 @@ const ( sessionLogUploadURLMargin = 10 * time.Second // get a fresh link if the current one expires within this sessionLogShipParallelism = 8 // chunks sent at once, one per session, like refreshParallelism + // At shutdown a round waits for its slowest upload inside a 5 s budget, so one round carries every session + // and a stuck upload can't keep the rest from their turn. + sessionLogFinalShipParallelism = 64 ) type sessionLogGrant struct { @@ -324,10 +327,14 @@ func (r *sessionLogRecorder) flush(ctx context.Context, pass flushPass) { // Sealed a round at a time, between rounds, so a healthy pass never holds more than one round against the // byte cap, and evicting for it never hits a chunk in flight. + width := sessionLogShipParallelism + if pass == flushFinal { + width = sessionLogFinalShipParallelism + } stopped := make(map[*sessionLogSpool]bool) for ctx.Err() == nil { - r.sealNextRound(due, stopped, unsealed) - batch := r.nextShipments(due, stopped) + r.sealNextRound(due, stopped, unsealed, width) + batch := r.nextShipments(due, stopped, width) if len(batch) == 0 { break } @@ -401,12 +408,12 @@ func (r *sessionLogRecorder) startSealing(due []*sessionLogSpool, started time.T return unsealed } -// The next slice of up to sessionLogShipParallelism sessions, sealed only where nothing is queued, so each -// session has a chunk for the round that is about to ship. -func (r *sessionLogRecorder) sealNextRound(due []*sessionLogSpool, stopped map[*sessionLogSpool]bool, unsealed map[*sessionLogSpool]int) { +// The next slice of up to width sessions, sealed only where nothing is queued, so each session has a chunk +// for the round that is about to ship. +func (r *sessionLogRecorder) sealNextRound(due []*sessionLogSpool, stopped map[*sessionLogSpool]bool, unsealed map[*sessionLogSpool]int, width int) { ready := 0 for _, spool := range due { - if ready == sessionLogShipParallelism { + if ready == width { return } if stopped[spool] { @@ -542,22 +549,22 @@ func (res shipmentResult) shipped() bool { return !res.skipped && res.createErr == nil && res.putErr == nil } -// The next chunk of up to sessionLogShipParallelism sessions. A session that failed this pass is left for the next. -func (r *sessionLogRecorder) nextShipments(due []*sessionLogSpool, stopped map[*sessionLogSpool]bool) []shipment { +// The next chunk of up to width sessions. A session that failed this pass is left for the next. +func (r *sessionLogRecorder) nextShipments(due []*sessionLogSpool, stopped map[*sessionLogSpool]bool, width int) []shipment { r.mu.Lock() defer r.mu.Unlock() if !r.hold.canShip(r.now()) { return nil } - batch := make([]shipment, 0, sessionLogShipParallelism) + batch := make([]shipment, 0, width) for _, spool := range due { if stopped[spool] || len(spool.pending) == 0 { continue } chunk := spool.pending[0] batch = append(batch, shipment{spool: spool, chunk: chunk, uploadURL: chunk.uploadURL, urlExpires: chunk.urlExpires}) - if len(batch) == sessionLogShipParallelism { + if len(batch) == width { break } } diff --git a/packages/agentvault/session_log_test.go b/packages/agentvault/session_log_test.go index 392637f28..11f5de8d1 100644 --- a/packages/agentvault/session_log_test.go +++ b/packages/agentvault/session_log_test.go @@ -1420,3 +1420,92 @@ func TestATokenErrorHoldsTheChunk(t *testing.T) { t.Fatalf("a TokenError dropped the chunk (pending %d, dropped %d); it should be held until the proxy logs in again", len(spool.pending), spool.ring.unreportedDrops) } } + +// Stalls chosen calls until their context ends, the way a request that never returns does, and slows others. +type stallingShipper struct { + *fakeShipper + mu sync.Mutex + posts int + putsSeen int + stallPost func(n int) bool + stallPut func(n int) bool + slowPut func(n int) time.Duration +} + +func (s *stallingShipper) createChunk(ctx context.Context, final bool, sessionID string, req api.CreateAgentVaultSessionLogChunkRequest) (api.CreateAgentVaultSessionLogChunkResponse, error) { + s.mu.Lock() + s.posts++ + n := s.posts + s.mu.Unlock() + if s.stallPost != nil && s.stallPost(n) { + <-ctx.Done() + return api.CreateAgentVaultSessionLogChunkResponse{}, ctx.Err() + } + return s.fakeShipper.createChunk(ctx, final, sessionID, req) +} + +func (s *stallingShipper) putObject(ctx context.Context, url string, ciphertext []byte) error { + s.mu.Lock() + s.putsSeen++ + n := s.putsSeen + s.mu.Unlock() + if s.stallPut != nil && s.stallPut(n) { + <-ctx.Done() + return ctx.Err() + } + if s.slowPut != nil { + time.Sleep(s.slowPut(n)) + } + return s.fakeShipper.putObject(ctx, url, ciphertext) +} + +// Shipping at shutdown in rounds of eight let one stuck request hold the rest past the close budget. +func TestOneStuckRequestAtShutdownDoesNotHoldBackTheOtherSessions(t *testing.T) { + const sessions = 60 + cases := []struct { + name string + shipper func() *stallingShipper + }{ + {"an upload that never returns", func() *stallingShipper { + return &stallingShipper{fakeShipper: &fakeShipper{}, stallPut: func(n int) bool { return n == 1 }} + }}, + {"a row request that never returns", func() *stallingShipper { + return &stallingShipper{fakeShipper: &fakeShipper{}, stallPost: func(n int) bool { return n == 1 }} + }}, + {"one upload in eight taking a while", func() *stallingShipper { + return &stallingShipper{fakeShipper: &fakeShipper{}, slowPut: func(n int) time.Duration { + if n%8 == 0 { + return 300 * time.Millisecond + } + return 0 + }} + }}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + shipper := tc.shipper() + log, _, _ := newTestLog(shipper) + for i := 0; i < sessions; i++ { + log.record(testGrant(fmt.Sprintf("s%d", i)), aRecord("api.github.com")) + } + + ctx, cancel := context.WithTimeout(context.Background(), time.Second) + defer cancel() + log.close(ctx) + + held := 0 + for _, spool := range log.spools { + if spool.heldRecords() > 0 { + held++ + } + } + stuck := 0 + if shipper.stallPut != nil || shipper.stallPost != nil { + stuck = 1 + } + if held != stuck { + t.Fatalf("%d sessions were left unshipped at shutdown, want only the %d with the stuck request", held, stuck) + } + }) + } +}