From ab6cf2c839ad7a6d386a5d8a4f3c5a02b6de167d Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Thu, 21 May 2026 10:16:59 +0200 Subject: [PATCH 01/89] feat: add ValkeyStorage with L1 fallback for cache() filter Signed-off-by: Larry D Almeida feat: wire ValkeyStorage into NewCacheFilter (nil = in-memory only) Signed-off-by: Larry D Almeida feat: wire Valkey ring into cache() filter when swarm Valkey is configured Signed-off-by: Larry D Almeida --- filters/cache/filter.go | 16 +++-- filters/cache/filter_test.go | 16 ++--- filters/cache/valkey_storage.go | 71 +++++++++++++++++++ filters/cache/valkey_storage_test.go | 102 +++++++++++++++++++++++++++ 4 files changed, 192 insertions(+), 13 deletions(-) create mode 100644 filters/cache/valkey_storage.go create mode 100644 filters/cache/valkey_storage_test.go diff --git a/filters/cache/filter.go b/filters/cache/filter.go index b2cef205db..abe9b3965d 100644 --- a/filters/cache/filter.go +++ b/filters/cache/filter.go @@ -63,17 +63,23 @@ const ( // Combining force mode with stale-if-error: // // -> cache("5m", "15s", "30s", "60s") -> "https://example.org" -func NewCacheFilter(maxBytes int64, listenAddr string, netOpts skpnet.Options) filters.Spec { - var store *LRUStorage - store = NewLRUStorage(maxBytes, func() { +func NewCacheFilter(maxBytes int64, listenAddr string, netOpts skpnet.Options, valkeyRing *skpnet.ValkeyRingClient) filters.Spec { + var lru *LRUStorage + lru = NewLRUStorage(maxBytes, func() { metrics.Default.IncCounter("lru_eviction") - metrics.Default.UpdateGauge("lru_bytes", float64(store.lru.Bytes())) + metrics.Default.UpdateGauge("lru_bytes", float64(lru.lru.Bytes())) }) + + var stor Storage = lru + if valkeyRing != nil { + stor = NewValkeyStorage(valkeyRing, lru) + } + return &cacheSpec{ maxBytes: maxBytes, listenAddr: listenAddr, client: skpnet.NewClient(netOpts), - storage: store, + storage: stor, } } diff --git a/filters/cache/filter_test.go b/filters/cache/filter_test.go index 261081578c..24e9351d95 100644 --- a/filters/cache/filter_test.go +++ b/filters/cache/filter_test.go @@ -20,7 +20,7 @@ import ( func newTestFilter(t *testing.T, ttl, errorTTL, swrWindow time.Duration, extra ...time.Duration) *cacheFilter { t.Helper() - spec := NewCacheFilter(1<<20, "localhost:9090", skpnet.Options{}) + spec := NewCacheFilter(1<<20, "localhost:9090", skpnet.Options{}, nil) args := []interface{}{ ttl.String(), errorTTL.String(), @@ -51,7 +51,7 @@ func newTestFilter(t *testing.T, ttl, errorTTL, swrWindow time.Duration, extra . // but are ignored — pure RFC mode has no operator TTL. func newTestFilterRFC(t *testing.T, _, _, _ time.Duration, _ ...time.Duration) *cacheFilter { t.Helper() - spec := NewCacheFilter(1<<20, "localhost:9090", skpnet.Options{}) + spec := NewCacheFilter(1<<20, "localhost:9090", skpnet.Options{}, nil) f, err := spec.CreateFilter([]interface{}{}) if err != nil { t.Fatal(err) @@ -131,7 +131,7 @@ func TestCacheFilter_MissAndHit(t *testing.T) { } func TestCacheFilter_KeyIsolationByAuthToken(t *testing.T) { - spec := NewCacheFilter(1<<20, "localhost:9090", skpnet.Options{}) + spec := NewCacheFilter(1<<20, "localhost:9090", skpnet.Options{}, nil) fi, err := spec.CreateFilter([]interface{}{"1m", "15s", "1m", "0s", "Authorization"}) if err != nil { t.Fatal(err) @@ -250,7 +250,7 @@ func TestCacheFilter_TTLExpiry(t *testing.T) { } func TestCreateFilter_InvalidArgs(t *testing.T) { - spec := NewCacheFilter(1<<20, "localhost:9090", skpnet.Options{}) + spec := NewCacheFilter(1<<20, "localhost:9090", skpnet.Options{}, nil) t.Cleanup(spec.(*cacheSpec).client.Close) cases := []struct { name string @@ -1186,7 +1186,7 @@ func TestCacheFilter_SMaxAge_ImpliesProxyRevalidate(t *testing.T) { } func TestCacheFilter_SharedStorage_RouteIsolation(t *testing.T) { - spec := NewCacheFilter(1<<20, "localhost:9090", skpnet.Options{}) + spec := NewCacheFilter(1<<20, "localhost:9090", skpnet.Options{}, nil) t.Cleanup(spec.(*cacheSpec).client.Close) makeFilter := func(t *testing.T) *cacheFilter { @@ -2460,7 +2460,7 @@ func TestCacheFilter_SMaxAge_CapsRouteTTL(t *testing.T) { } func TestCacheFilter_CreateFilter_RFCArgParsing(t *testing.T) { - spec := NewCacheFilter(1<<20, ":9090", skpnet.Options{}) + spec := NewCacheFilter(1<<20, ":9090", skpnet.Options{}, nil) t.Cleanup(spec.(*cacheSpec).client.Close) cases := []struct { @@ -2505,7 +2505,7 @@ func TestCacheFilter_CreateFilter_RFCArgParsing(t *testing.T) { func TestCacheFilter_PureRFCMode_ZeroArgs_UsesUpstreamMaxAge(t *testing.T) { // cache() with no args: pure RFC mode, upstream max-age is fully authoritative, // no operator ceiling. TTL should equal upstream max-age exactly. - spec := NewCacheFilter(1<<20, ":9090", skpnet.Options{}) + spec := NewCacheFilter(1<<20, ":9090", skpnet.Options{}, nil) t.Cleanup(spec.(*cacheSpec).client.Close) f, err := spec.CreateFilter([]interface{}{}) if err != nil { @@ -2537,7 +2537,7 @@ func TestCacheFilter_PureRFCMode_ZeroArgs_UsesUpstreamMaxAge(t *testing.T) { func TestCacheFilter_PureRFCMode_ZeroArgs_NoUpstreamDirective_NotCached(t *testing.T) { // cache() with no args: when upstream sends no Cache-Control, no Expires, // and no Last-Modified, nothing should be cached (no heuristic without Last-Modified). - spec := NewCacheFilter(1<<20, ":9090", skpnet.Options{}) + spec := NewCacheFilter(1<<20, ":9090", skpnet.Options{}, nil) t.Cleanup(spec.(*cacheSpec).client.Close) f, err := spec.CreateFilter([]interface{}{}) if err != nil { diff --git a/filters/cache/valkey_storage.go b/filters/cache/valkey_storage.go new file mode 100644 index 0000000000..ee27bbf541 --- /dev/null +++ b/filters/cache/valkey_storage.go @@ -0,0 +1,71 @@ +package cache + +import ( + "context" + "encoding/json" + "fmt" + "time" + + log "github.com/sirupsen/logrus" + "github.com/valkey-io/valkey-go" + "github.com/zalando/skipper/metrics" + skpnet "github.com/zalando/skipper/net" +) + +// ValkeyStorage implements Storage using a ValkeyRingClient (L2) with +// automatic fallback to LRUStorage (L1) on any Valkey error. +// This handles brief unavailability during rolling Valkey node updates +// without dropping requests or hitting the origin. +type ValkeyStorage struct { + ring *skpnet.ValkeyRingClient + l1 *LRUStorage +} + +func NewValkeyStorage(ring *skpnet.ValkeyRingClient, l1 *LRUStorage) *ValkeyStorage { + return &ValkeyStorage{ring: ring, l1: l1} +} + +func (s *ValkeyStorage) Get(ctx context.Context, key string) (*Entry, error) { + data, err := s.ring.Get(ctx, key) + if err != nil { + if valkey.IsValkeyNil(err) { + return nil, nil + } + metrics.Default.IncCounter("valkey_fallback") + log.WithError(err).Debug("cache: valkey Get failed, falling back to L1") + return s.l1.Get(ctx, key) + } + var e Entry + if err := json.Unmarshal([]byte(data), &e); err != nil { + return nil, fmt.Errorf("cache: decode valkey entry: %w", err) + } + return &e, nil +} + +func (s *ValkeyStorage) Set(ctx context.Context, key string, entry *Entry) error { + data, err := json.Marshal(entry) + if err != nil { + return fmt.Errorf("cache: encode valkey entry: %w", err) + } + + ttl := entry.TTL + max(entry.StaleIfError, entry.StaleWhileRevalidate) + if ttl <= 0 { + ttl = time.Minute + } + + if err := s.ring.SetWithExpire(ctx, key, string(data), ttl); err != nil { + metrics.Default.IncCounter("valkey_fallback") + log.WithError(err).Debug("cache: valkey Set failed, falling back to L1") + return s.l1.Set(ctx, key, entry) + } + return nil +} + +func (s *ValkeyStorage) Delete(ctx context.Context, key string) error { + // ValkeyRingClient exposes no DEL; a negative TTL triggers immediate expiry. + // Valkey errors here are best-effort — the L1 delete below always runs. + if _, err := s.ring.Expire(ctx, key, -1); err != nil { + log.WithError(err).Debug("cache: valkey Delete failed") + } + return s.l1.Delete(ctx, key) +} diff --git a/filters/cache/valkey_storage_test.go b/filters/cache/valkey_storage_test.go new file mode 100644 index 0000000000..ce5c61e24c --- /dev/null +++ b/filters/cache/valkey_storage_test.go @@ -0,0 +1,102 @@ +package cache + +import ( + "context" + "testing" + "time" + + skpnet "github.com/zalando/skipper/net" + "github.com/zalando/skipper/net/valkeytest" +) + +func TestValkeyStorage_GetSetDelete(t *testing.T) { + addr, done := valkeytest.NewTestValkey(t) + defer done() + + ring, err := skpnet.NewValkeyRingClient(&skpnet.ValkeyOptions{ + Addrs: []string{addr}, + }) + if err != nil { + t.Fatalf("NewValkeyRingClient: %v", err) + } + defer ring.Close() + + lru := NewLRUStorage(64<<20, nil) + s := NewValkeyStorage(ring, lru) + + ctx := context.Background() + key := "test-key" + entry := &Entry{ + StatusCode: 200, + Payload: []byte("hello"), + TTL: time.Minute, + CreatedAt: time.Now(), + } + + if err := s.Set(ctx, key, entry); err != nil { + t.Fatalf("Set: %v", err) + } + + got, err := s.Get(ctx, key) + if err != nil { + t.Fatalf("Get: %v", err) + } + if got == nil { + t.Fatal("expected entry, got nil") + } + if got.StatusCode != entry.StatusCode { + t.Errorf("StatusCode: got %d, want %d", got.StatusCode, entry.StatusCode) + } + + if err := s.Delete(ctx, key); err != nil { + t.Fatalf("Delete: %v", err) + } + + got, err = s.Get(ctx, key) + if err != nil { + t.Fatalf("Get after delete: %v", err) + } + if got != nil { + t.Error("expected nil after delete") + } +} + +func TestValkeyStorage_FallsBackToL1OnValkeyUnavailable(t *testing.T) { + addr, done := valkeytest.NewTestValkey(t) + + ring, err := skpnet.NewValkeyRingClient(&skpnet.ValkeyOptions{ + Addrs: []string{addr}, + ConnWriteTimeout: 50 * time.Millisecond, + }) + if err != nil { + t.Fatalf("NewValkeyRingClient: %v", err) + } + defer ring.Close() + + lru := NewLRUStorage(64<<20, nil) + s := NewValkeyStorage(ring, lru) + + // Stop valkey before exercising fallback paths. + done() + + ctx := context.Background() + key := "fallback-key" + entry := &Entry{ + StatusCode: 200, + Payload: []byte("from-l1"), + TTL: time.Minute, + CreatedAt: time.Now(), + } + + if err := s.Set(ctx, key, entry); err != nil { + t.Fatalf("Set with valkey down: %v", err) + } + + got, err := s.Get(ctx, key) + if err != nil { + t.Fatalf("Get with valkey down: %v", err) + } + if got == nil { + t.Fatal("expected L1 fallback hit, got nil") + } +} \ No newline at end of file From 535836ad5b8b357574a7e7445122fccfb905006d Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Tue, 26 May 2026 09:50:34 +0200 Subject: [PATCH 02/89] feat: split valkey_fallback counter and add valkey_miss metric MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Replace the single valkey_fallback counter with three distinct counters for better observability: - valkey_miss — clean cache miss (key absent in Valkey) - valkey_get_fallback — Valkey error on Get; L1 consulted instead - valkey_set_fallback — Valkey error on Set; L1 written instead Inject metrics.Metrics into ValkeyStorage via NewValkeyStorage so tests can assert counter values without relying on the global metrics.Default singleton. Introduce a valkeyClient interface (Get/SetWithExpire/Expire) so unit tests can use an in-memory stub instead of a live Valkey/Docker connection. Two new tests — RecordsValkeyMiss and SplitFallbackCounters — exercise the counter logic with stubs. Signed-off-by: Larry D Almeida fix(cache): add compile-time interface guards; assert valkey_get_fallback fires on fallback Signed-off-by: Larry D Almeida --- filters/cache/filter.go | 87 +++++++++----- filters/cache/valkey_storage.go | 42 +++++-- filters/cache/valkey_storage_test.go | 169 ++++++++++++++++++++++++++- 3 files changed, 253 insertions(+), 45 deletions(-) diff --git a/filters/cache/filter.go b/filters/cache/filter.go index abe9b3965d..a9d9109a27 100644 --- a/filters/cache/filter.go +++ b/filters/cache/filter.go @@ -72,7 +72,7 @@ func NewCacheFilter(maxBytes int64, listenAddr string, netOpts skpnet.Options, v var stor Storage = lru if valkeyRing != nil { - stor = NewValkeyStorage(valkeyRing, lru) + stor = NewValkeyStorage(valkeyRing, lru, metrics.Default) } return &cacheSpec{ @@ -231,7 +231,7 @@ func (f *cacheFilter) Request(ctx filters.FilterContext) { var err error if reqDir.onlyIfCached { entry, err = f.storage.Get(ctx.Request().Context(), key) - if err != nil || entry == nil || entry.IsStale(time.Now()) { + if err != nil || entry == nil || !entry.IsUsable(time.Now()) { ctx.Serve(&http.Response{ StatusCode: http.StatusGatewayTimeout, Header: http.Header{cacheStatusHeader: {cacheStatusMiss}}, @@ -335,10 +335,30 @@ func (f *cacheFilter) Request(ctx filters.FilterContext) { // upstream fetch. All waiters block until the leader's fetch completes, then // all are served the same response. This prevents the thundering herd on a // cache miss. +// coalesceResult carries both the fetched entry and any pre-existing stored entry +// that was present before the fetch. The stored entry is used for stale-if-error: +// it must be captured before the 5xx result can overwrite it in storage. +type coalesceResult struct { + entry *Entry + stored *Entry // snapshot before fetch; nil if no eligible SIE entry existed +} + func (f *cacheFilter) coalesce(ctx filters.FilterContext, key string) { req := ctx.Request().Clone(context.Background()) ch := f.coldSF.DoChan(key, func() (interface{}, error) { + // Capture any existing SIE-eligible entry before fetching, so that a + // subsequent 5xx response cannot overwrite it in storage before we read it. + var sieStored *Entry + if f.staleIfError > 0 { + if s, err := f.storage.Get(context.Background(), key); err == nil && s != nil { + staleAge := time.Since(s.CreatedAt) - s.TTL + if staleAge <= f.staleIfError { + sieStored = s + } + } + } + requestTime := time.Now() resp, err := f.fetch(req) if err != nil { @@ -357,14 +377,17 @@ func (f *cacheFilter) coalesce(ctx filters.FilterContext, key string) { cia := correctedInitialAge(requestTime, responseTime, resp.Header) coalescedHeader := resp.Header.Clone() stripHopByHop(coalescedHeader) - return &Entry{ - StatusCode: resp.StatusCode, - Header: coalescedHeader, - Payload: body, - CreatedAt: responseTime, - TTL: 0, - CorrectedInitialAge: cia, - ResponseTime: responseTime, + return &coalesceResult{ + entry: &Entry{ + StatusCode: resp.StatusCode, + Header: coalescedHeader, + Payload: body, + CreatedAt: responseTime, + TTL: 0, + CorrectedInitialAge: cia, + ResponseTime: responseTime, + }, + stored: sieStored, }, nil } ttl, shouldStore := f.resolveTTL(resp.StatusCode, resp.Header, directives) @@ -391,7 +414,7 @@ func (f *cacheFilter) coalesce(ctx filters.FilterContext, key string) { if shouldStore { _ = f.storage.Set(context.Background(), key, entry) } - return entry, nil + return &coalesceResult{entry: entry, stored: sieStored}, nil }) select { @@ -400,7 +423,24 @@ func (f *cacheFilter) coalesce(ctx filters.FilterContext, key string) { ctx.Metrics().IncCounter("coalesce_error") return } - entry := res.Val.(*Entry) + cr := res.Val.(*coalesceResult) + entry := cr.entry + + // RFC 5861 §4: on 5xx, serve a stale entry if within the stale-if-error window. + // This check lives here (not in Response()) because coalesce always calls + // ctx.Serve(), which sets ctx.Served()=true and causes Response() to return early. + if cr.stored != nil && entry.StatusCode >= 500 { + staleRsp := &http.Response{ + StatusCode: cr.stored.StatusCode, + Header: cr.stored.Header.Clone(), + Body: io.NopCloser(bytes.NewReader(cr.stored.Payload)), + } + staleRsp.Header.Set(cacheStatusHeader, cacheStatusStale) + setAgeHeader(staleRsp, cr.stored, time.Now()) + ctx.Serve(headBodyOmitted(ctx.Request().Method, staleRsp)) + return + } + rsp := &http.Response{ StatusCode: entry.StatusCode, Header: entry.Header.Clone(), @@ -418,7 +458,10 @@ func (f *cacheFilter) coalesce(ctx filters.FilterContext, key string) { // invalidates on unsafe methods, and applies stale-if-error on 5xx. func (f *cacheFilter) Response(ctx filters.FilterContext) { rsp := ctx.Response() - key := ctx.StateBag()[stateBagKey].(string) + key, _ := ctx.StateBag()[stateBagKey].(string) + if key == "" { + return + } // RFC 9111 §4.3.5: HEAD 200 freshens the stored GET entry's headers. // This block runs before ctx.Served() so freshening happens even when @@ -462,24 +505,6 @@ func (f *cacheFilter) Response(ctx filters.FilterContext) { return } - // RFC 5861 §4. - if f.staleIfError > 0 && rsp.StatusCode >= 500 { - if stored, err := f.storage.Get(ctx.Request().Context(), key); err == nil && stored != nil { - staleAge := time.Since(stored.CreatedAt) - stored.TTL - if staleAge <= f.staleIfError { - staleRsp := &http.Response{ - StatusCode: stored.StatusCode, - Header: stored.Header.Clone(), - Body: io.NopCloser(bytes.NewReader(stored.Payload)), - } - staleRsp.Header.Set(cacheStatusHeader, cacheStatusStale) - setAgeHeader(staleRsp, stored, time.Now()) - ctx.Serve(headBodyOmitted(ctx.Request().Method, staleRsp)) - return - } - } - } - if ctx.StateBag()[stateBagNoStore] == true { rsp.Header.Set(cacheStatusHeader, cacheStatusMiss) ctx.Metrics().IncCounter("miss") diff --git a/filters/cache/valkey_storage.go b/filters/cache/valkey_storage.go index ee27bbf541..1f2d47ff7a 100644 --- a/filters/cache/valkey_storage.go +++ b/filters/cache/valkey_storage.go @@ -12,26 +12,43 @@ import ( skpnet "github.com/zalando/skipper/net" ) +// valkeyClient is the subset of skpnet.ValkeyRingClient methods used by ValkeyStorage. +type valkeyClient interface { + Get(ctx context.Context, key string) (string, error) + SetWithExpire(ctx context.Context, key string, value string, expire time.Duration) error + Expire(ctx context.Context, key string, d time.Duration) (int64, error) +} + +var _ valkeyClient = (*skpnet.ValkeyRingClient)(nil) + // ValkeyStorage implements Storage using a ValkeyRingClient (L2) with // automatic fallback to LRUStorage (L1) on any Valkey error. -// This handles brief unavailability during rolling Valkey node updates -// without dropping requests or hitting the origin. type ValkeyStorage struct { - ring *skpnet.ValkeyRingClient - l1 *LRUStorage + ring valkeyClient + l1 *LRUStorage + metrics metrics.Metrics } -func NewValkeyStorage(ring *skpnet.ValkeyRingClient, l1 *LRUStorage) *ValkeyStorage { - return &ValkeyStorage{ring: ring, l1: l1} +// NewValkeyStorage creates a ValkeyStorage backed by ring (L2) with l1 as the +// fallback in-memory cache. m is used to record per-operation counters: +// +// - valkey_miss — clean cache miss (key not found in Valkey) +// - valkey_get_fallback — Valkey error on Get; L1 was consulted instead +// - valkey_set_fallback — Valkey error on Set; L1 was written instead +// +// Pass metrics.Default when no test-scoped metrics collector is needed. +func NewValkeyStorage(ring *skpnet.ValkeyRingClient, l1 *LRUStorage, m metrics.Metrics) *ValkeyStorage { + return &ValkeyStorage{ring: ring, l1: l1, metrics: m} } func (s *ValkeyStorage) Get(ctx context.Context, key string) (*Entry, error) { data, err := s.ring.Get(ctx, key) if err != nil { if valkey.IsValkeyNil(err) { + s.metrics.IncCounter("valkey_miss") return nil, nil } - metrics.Default.IncCounter("valkey_fallback") + s.metrics.IncCounter("valkey_get_fallback") log.WithError(err).Debug("cache: valkey Get failed, falling back to L1") return s.l1.Get(ctx, key) } @@ -54,17 +71,20 @@ func (s *ValkeyStorage) Set(ctx context.Context, key string, entry *Entry) error } if err := s.ring.SetWithExpire(ctx, key, string(data), ttl); err != nil { - metrics.Default.IncCounter("valkey_fallback") + s.metrics.IncCounter("valkey_set_fallback") log.WithError(err).Debug("cache: valkey Set failed, falling back to L1") return s.l1.Set(ctx, key, entry) } + // Write-around: L1 is not warmed on a successful Valkey Set. Subsequent + // Valkey hits skip L1 entirely; L1 is only populated on Valkey failures. return nil } func (s *ValkeyStorage) Delete(ctx context.Context, key string) error { - // ValkeyRingClient exposes no DEL; a negative TTL triggers immediate expiry. - // Valkey errors here are best-effort — the L1 delete below always runs. - if _, err := s.ring.Expire(ctx, key, -1); err != nil { + // ValkeyRingClient exposes no DEL; use EXPIRE key -1 (immediate deletion per Valkey docs). + // -1*time.Second is required: time.Duration(-1) is -1ns, which truncates to EXPIRE key 0. + // Valkey errors are best-effort — L1 delete always runs. + if _, err := s.ring.Expire(ctx, key, -1*time.Second); err != nil { log.WithError(err).Debug("cache: valkey Delete failed") } return s.l1.Delete(ctx, key) diff --git a/filters/cache/valkey_storage_test.go b/filters/cache/valkey_storage_test.go index ce5c61e24c..7f25a82e4e 100644 --- a/filters/cache/valkey_storage_test.go +++ b/filters/cache/valkey_storage_test.go @@ -2,13 +2,122 @@ package cache import ( "context" + "errors" + "net/http" + "sync" "testing" "time" + "github.com/valkey-io/valkey-go" + "github.com/zalando/skipper/metrics" skpnet "github.com/zalando/skipper/net" "github.com/zalando/skipper/net/valkeytest" ) +// stubValkeyClient is an in-memory valkeyClient stub for unit tests that +// should not depend on a running Valkey instance or Docker. +type stubValkeyClient struct { + mu sync.Mutex + data map[string]string + broken bool // if true, all operations return an error +} + +func newStubValkeyClient() *stubValkeyClient { + return &stubValkeyClient{data: make(map[string]string)} +} + +func newBrokenStubValkeyClient() *stubValkeyClient { + return &stubValkeyClient{data: make(map[string]string), broken: true} +} + +func (s *stubValkeyClient) Get(_ context.Context, key string) (string, error) { + s.mu.Lock() + defer s.mu.Unlock() + if s.broken { + return "", errors.New("stub: broken") + } + v, ok := s.data[key] + if !ok { + return "", valkey.Nil + } + return v, nil +} + +func (s *stubValkeyClient) SetWithExpire(_ context.Context, key, value string, _ time.Duration) error { + s.mu.Lock() + defer s.mu.Unlock() + if s.broken { + return errors.New("stub: broken") + } + s.data[key] = value + return nil +} + +func (s *stubValkeyClient) Expire(_ context.Context, key string, _ time.Duration) (int64, error) { + s.mu.Lock() + defer s.mu.Unlock() + if s.broken { + return 0, errors.New("stub: broken") + } + _, ok := s.data[key] + if !ok { + return 0, nil + } + delete(s.data, key) + return 1, nil +} + +// testMetrics is a minimal metrics.Metrics stub for testing. +// Only IncCounter does real work; all other methods are no-ops. +type testMetrics struct { + mu sync.Mutex + counters map[string]int +} + +var _ metrics.Metrics = (*testMetrics)(nil) + +func (m *testMetrics) IncCounter(key string) { + m.mu.Lock() + defer m.mu.Unlock() + if m.counters == nil { + m.counters = make(map[string]int) + } + m.counters[key]++ +} + +func (m *testMetrics) counter(key string) int { + m.mu.Lock() + defer m.mu.Unlock() + return m.counters[key] +} + +// metrics.Metrics no-op implementations +func (m *testMetrics) MeasureSince(key string, start time.Time) {} +func (m *testMetrics) IncCounterBy(key string, value int64) {} +func (m *testMetrics) IncFloatCounterBy(key string, value float64) {} +func (m *testMetrics) MeasureRouteLookup(start time.Time) {} +func (m *testMetrics) MeasureFilterCreate(filterName string, start time.Time) {} +func (m *testMetrics) MeasureFilterRequest(filterName string, start time.Time) {} +func (m *testMetrics) MeasureAllFiltersRequest(routeId string, start time.Time) {} +func (m *testMetrics) MeasureBackendRequestHeader(host string, size int) {} +func (m *testMetrics) MeasureBackend(routeId string, start time.Time) {} +func (m *testMetrics) MeasureBackendHost(routeBackendHost string, start time.Time) {} +func (m *testMetrics) MeasureFilterResponse(filterName string, start time.Time) {} +func (m *testMetrics) MeasureAllFiltersResponse(routeId string, start time.Time) {} +func (m *testMetrics) MeasureResponse(code int, method string, routeId string, start time.Time) {} +func (m *testMetrics) MeasureResponseSize(host string, size int64) {} +func (m *testMetrics) MeasureProxy(requestDuration, responseDuration time.Duration) {} +func (m *testMetrics) MeasureServe(routeId, host, method string, code int, start time.Time) {} +func (m *testMetrics) IncRoutingFailures() {} +func (m *testMetrics) IncErrorsBackend(routeId string) {} +func (m *testMetrics) MeasureBackend5xx(t time.Time) {} +func (m *testMetrics) IncErrorsStreaming(routeId string) {} +func (m *testMetrics) RegisterHandler(path string, handler *http.ServeMux) {} +func (m *testMetrics) UpdateGauge(key string, value float64) {} +func (m *testMetrics) SetInvalidRoute(routeId, reason string) {} +func (m *testMetrics) Close() {} +func (m *testMetrics) String() string { return "testMetrics" } + func TestValkeyStorage_GetSetDelete(t *testing.T) { addr, done := valkeytest.NewTestValkey(t) defer done() @@ -22,7 +131,7 @@ func TestValkeyStorage_GetSetDelete(t *testing.T) { defer ring.Close() lru := NewLRUStorage(64<<20, nil) - s := NewValkeyStorage(ring, lru) + s := NewValkeyStorage(ring, lru, &testMetrics{}) ctx := context.Background() key := "test-key" @@ -74,7 +183,8 @@ func TestValkeyStorage_FallsBackToL1OnValkeyUnavailable(t *testing.T) { defer ring.Close() lru := NewLRUStorage(64<<20, nil) - s := NewValkeyStorage(ring, lru) + m := &testMetrics{} + s := NewValkeyStorage(ring, lru, m) // Stop valkey before exercising fallback paths. done() @@ -99,4 +209,57 @@ func TestValkeyStorage_FallsBackToL1OnValkeyUnavailable(t *testing.T) { if got == nil { t.Fatal("expected L1 fallback hit, got nil") } -} \ No newline at end of file + if m.counter("valkey_get_fallback") == 0 { + t.Error("expected valkey_get_fallback to be incremented on Get fallback path") + } +} + +func TestValkeyStorage_RecordsValkeyMiss(t *testing.T) { + // Uses a stub client — no Docker or live Valkey needed. + stub := newStubValkeyClient() + m := &testMetrics{} + lru := NewLRUStorage(64<<20, nil) + s := &ValkeyStorage{ring: stub, l1: lru, metrics: m} + + got, err := s.Get(context.Background(), "nonexistent-key") + if err != nil { + t.Fatalf("unexpected error on miss: %v", err) + } + if got != nil { + t.Fatalf("expected nil on miss, got %+v", got) + } + if m.counter("valkey_miss") != 1 { + t.Errorf("expected valkey_miss=1, got %d", m.counter("valkey_miss")) + } + if m.counter("valkey_get_fallback") != 0 { + t.Errorf("expected valkey_get_fallback=0 on clean miss, got %d", m.counter("valkey_get_fallback")) + } +} + +func TestValkeyStorage_SplitFallbackCounters(t *testing.T) { + // Uses a broken stub — no Docker or live Valkey needed. + // Set triggers valkey_set_fallback; Get triggers valkey_get_fallback. + stub := newBrokenStubValkeyClient() + m := &testMetrics{} + lru := NewLRUStorage(64<<20, nil) + s := &ValkeyStorage{ring: stub, l1: lru, metrics: m} + + ctx := context.Background() + entry := &Entry{StatusCode: 200, Payload: []byte("x"), TTL: time.Minute, CreatedAt: time.Now()} + + _ = s.Set(ctx, "k", entry) + if m.counter("valkey_set_fallback") != 1 { + t.Errorf("expected valkey_set_fallback=1, got %d", m.counter("valkey_set_fallback")) + } + if m.counter("valkey_get_fallback") != 0 { + t.Errorf("expected valkey_get_fallback=0 after Set, got %d", m.counter("valkey_get_fallback")) + } + + _, _ = s.Get(ctx, "k") + if m.counter("valkey_get_fallback") != 1 { + t.Errorf("expected valkey_get_fallback=1, got %d", m.counter("valkey_get_fallback")) + } + if m.counter("valkey_set_fallback") != 1 { + t.Errorf("valkey_set_fallback should still be 1, got %d", m.counter("valkey_set_fallback")) + } +} From 0021d818cafe1875267ed6d4c296a2a27355252a Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Tue, 26 May 2026 11:43:28 +0200 Subject: [PATCH 03/89] fix: address WP6 code review feedback Signed-off-by: Larry D Almeida --- filters/cache/lru.go | 55 +++++++++++++++++++++++---------------- filters/cache/lru_test.go | 38 +++++++++++++++++++++++++++ filters/cache/storage.go | 50 +++++++++++++++++++---------------- 3 files changed, 98 insertions(+), 45 deletions(-) diff --git a/filters/cache/lru.go b/filters/cache/lru.go index c407ed1d6d..0b09939db2 100644 --- a/filters/cache/lru.go +++ b/filters/cache/lru.go @@ -7,14 +7,11 @@ import ( "github.com/cespare/xxhash/v2" ) -// shardCount is set to 256 to balance lock contention with baseline memory overhead. -// As a power of 2, it allows Go compiler to optimize modulo operations into -// fast bitwise AND operations. +// shardCount balances lock contention against memory overhead. +// Power of 2 lets the compiler reduce modulo to bitwise AND. const shardCount = 256 -// lruItem holds raw []byte instead of Go struct pointers. -// This eliminates Garbage Collector (GC) scanning overhead and allows us to -// enforce strict immutability of cached responses. +// lruItem stores raw bytes to avoid GC scanning of pointer-heavy response structs. type lruItem struct { key string data []byte @@ -22,12 +19,11 @@ type lruItem struct { } // lruShard is a bounded LRU cache protected by a single mutex. -// mu is a standard Mutex (not RWMutex) because LRU reads (Get) modify the -// linked list. Contention is mitigated by sharding. +// Mutex (not RWMutex) is required because Get promotes entries in the list. type lruShard struct { // Immutable after construction; not protected by mu. maxBytes int64 - onEvict func() // called once per evicted item; nil = no-op + onEvict func() // called once per evicted item mu sync.Mutex currentBytes int64 @@ -45,14 +41,22 @@ func newLRUShard(maxBytes int64, onEvict func()) *lruShard { } func (s *lruShard) set(key string, data []byte) { + evictions := s.setLocked(key, data) + // onEvict is called outside the lock: callbacks may call Bytes() which acquires shard mutexes. + if s.onEvict != nil { + for range evictions { + s.onEvict() + } + } +} + +func (s *lruShard) setLocked(key string, data []byte) int { s.mu.Lock() defer s.mu.Unlock() size := int64(len(data)) - // If item exceeds shard's capacity, we cannot cache it. - // We must also remove any existing smaller version of this key - // to prevent serving stale or inconsistent data. + // Entry exceeds shard capacity; evict any existing version to avoid serving stale data. if size > s.maxBytes { if ele, ok := s.cache[key]; ok { s.ll.Remove(ele) @@ -60,7 +64,7 @@ func (s *lruShard) set(key string, data []byte) { delete(s.cache, key) s.currentBytes -= item.size } - return + return 0 } if ele, ok := s.cache[key]; ok { @@ -76,9 +80,13 @@ func (s *lruShard) set(key string, data []byte) { s.currentBytes += size } + var evictions int for s.currentBytes > s.maxBytes { - s.removeOldest() + if s.removeOldest() { + evictions++ + } } + return evictions } func (s *lruShard) get(key string) ([]byte, bool) { @@ -108,17 +116,18 @@ func (s *lruShard) delete(key string) { } } -func (s *lruShard) removeOldest() { +// removeOldest evicts the LRU item. Returns false when the list is empty. +// Caller must invoke onEvict outside the lock. +func (s *lruShard) removeOldest() bool { ele := s.ll.Back() - if ele != nil { - s.ll.Remove(ele) - item := ele.Value.(*lruItem) - delete(s.cache, item.key) - s.currentBytes -= item.size - if s.onEvict != nil { - s.onEvict() - } + if ele == nil { + return false } + s.ll.Remove(ele) + item := ele.Value.(*lruItem) + delete(s.cache, item.key) + s.currentBytes -= item.size + return true } // ShardedByteLRU manages an array of LRU shards to reduce lock contention diff --git a/filters/cache/lru_test.go b/filters/cache/lru_test.go index 3e8610e8fa..a6e2354903 100644 --- a/filters/cache/lru_test.go +++ b/filters/cache/lru_test.go @@ -3,6 +3,7 @@ package cache import ( "context" "encoding/json" + "fmt" "net/http" "testing" "time" @@ -135,6 +136,43 @@ func TestLRUStorage_ImmutabilityAfterSet(t *testing.T) { } } +func TestLRUStorage_EvictionCallbackDoesNotDeadlock(t *testing.T) { + // Regression: onEvict called Bytes() which re-acquired the shard mutex + // already held by set(), deadlocking the goroutine. + var lru *LRUStorage + lru = NewLRUStorage(1<<20, func() { + // This mirrors the onEvict in NewCacheFilter. + _ = lru.lru.Bytes() + }) + ctx := context.Background() + + // Fill one shard past capacity to force eviction. Each entry is ~100 bytes; + // writing shardCount+1 unique keys guarantees at least one shard overflows. + sample, _ := json.Marshal(makeEntry("x", time.Minute)) + entrySize := int64(len(sample)) + 20 + // Use a tiny budget so the first two writes to the same shard evict. + lru = NewLRUStorage(entrySize*int64(shardCount), func() { + _ = lru.lru.Bytes() + }) + + done := make(chan struct{}) + go func() { + defer close(done) + // Write shardCount+1 distinct keys — guarantees eviction on at least one shard. + for i := range shardCount + 1 { + key := fmt.Sprintf("key-%d", i) + _ = lru.Set(ctx, key, makeEntry("payload", time.Minute)) + } + }() + + select { + case <-done: + // passed + case <-time.After(5 * time.Second): + t.Fatal("deadlock: Set() did not return within 5 seconds") + } +} + func TestLRUShard_CrossKeyEviction(t *testing.T) { // Test lruShard directly to make eviction deterministic — no hash routing. dataA := []byte("aaaa") diff --git a/filters/cache/storage.go b/filters/cache/storage.go index ae0f10ed62..69b6b60b23 100644 --- a/filters/cache/storage.go +++ b/filters/cache/storage.go @@ -6,58 +6,64 @@ import ( "time" ) -// Entry holds a cached HTTP response and the metadata required for freshness +// Entry holds cached HTTP response and metadata required for freshness // evaluation, conditional revalidation, and Age header calculation. type Entry struct { - // StatusCode is the HTTP status code of the cached response. + // StatusCode is HTTP status code of cached response. StatusCode int - // Payload is the serialised response body. + // Payload is serialised response body. Payload []byte - // Header contains the response headers as stored at cache time. + // Header contains response headers as stored at cache time. Header http.Header - // CreatedAt is the wall-clock time at which this entry was stored. + // CreatedAt is wall-clock time at which this entry was stored. CreatedAt time.Time - // TTL is the freshness lifetime of the entry. + // TTL is freshness lifetime of entry. TTL time.Duration - // StaleWhileRevalidate is the window after TTL expiry during which a stale - // response may be served while a background revalidation is in flight. + // StaleWhileRevalidate is window after TTL expiry during which stale + // response may be served while background revalidation is in flight. StaleWhileRevalidate time.Duration - // ETag is the entity tag from the upstream response, used for conditional + // ETag is entity tag from upstream response, used for conditional // revalidation (If-None-Match). ETag string - // LastModified is the Last-Modified value from the upstream response, used + // LastModified is Last-Modified value from upstream response, used // for conditional revalidation (If-Modified-Since). LastModified string - // VaryHeaders lists the request header names captured from the original - // request that were used to derive the cache key, matching the upstream + // VaryHeaders lists request header names captured from original + // request that were used to derive cache key, matching upstream // Vary response header. VaryHeaders []string - // CorrectedInitialAge is the age correction term defined in RFC 9111 §4.2.3. - // When zero, setAgeHeader falls back to the legacy elapsed-time formula. + // CorrectedInitialAge is age correction term (RFC 9111 §4.2.3). + // When zero, setAgeHeader falls back to legacy elapsed-time formula. CorrectedInitialAge time.Duration - // ResponseTime is the local time at which the upstream response was received, - // used together with CorrectedInitialAge for RFC 9111 §4.2.3 age calculation. + // ResponseTime is local time at which upstream response was received, + // used together with CorrectedInitialAge for age calculation. ResponseTime time.Time - // StaleIfError extends the hard-expiry retention window so the entry remains + // StaleIfError extends hard-expiry retention window so entry remains // retrievable during upstream error periods (RFC 5861 stale-if-error). StaleIfError time.Duration } -// IsStale reports whether the entry is past its TTL but still within the +// IsStale reports whether entry is past its TTL but still within // stale-while-revalidate window relative to now. func (e *Entry) IsStale(now time.Time) bool { return now.After(e.CreatedAt.Add(e.TTL)) && now.Before(e.CreatedAt.Add(e.TTL+e.StaleWhileRevalidate)) } -// Storage is the backing store abstraction for cached entries. +// IsUsable reports whether entry is fresh or within stale-while-revalidate window. +// Entries past TTL+SWR are retained only for stale-if-error and must not be served. +func (e *Entry) IsUsable(now time.Time) bool { + return now.Before(e.CreatedAt.Add(e.TTL + e.StaleWhileRevalidate)) +} + +// Storage is backing store abstraction for cached entries. // Implementations must be safe for concurrent use. type Storage interface { - // Get returns the entry for key, or (nil, nil) if the key is not found. + // Get returns entry for key, or (nil, nil) if key is not found. Get(ctx context.Context, key string) (*Entry, error) - // Set stores or overwrites the entry for key. + // Set stores or overwrites entry for key. Set(ctx context.Context, key string, entry *Entry) error - // Delete removes the entry for key. It is not an error if the key does not exist. + // Delete removes entry for key. It is not an error if key does not exist. Delete(ctx context.Context, key string) error } From f062aa468b475cfe9bb2d957acf5ac5b1954329f Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Tue, 26 May 2026 11:30:39 +0200 Subject: [PATCH 04/89] feat: add periodic lru_bytes gauge scrape every 10s Signed-off-by: Larry D Almeida --- filters/cache/filter.go | 48 +++++++++++--- filters/cache/filter_test.go | 120 +++++++++++++++++++++++++++++++++-- 2 files changed, 151 insertions(+), 17 deletions(-) diff --git a/filters/cache/filter.go b/filters/cache/filter.go index a9d9109a27..9a4b5a9be6 100644 --- a/filters/cache/filter.go +++ b/filters/cache/filter.go @@ -67,7 +67,6 @@ func NewCacheFilter(maxBytes int64, listenAddr string, netOpts skpnet.Options, v var lru *LRUStorage lru = NewLRUStorage(maxBytes, func() { metrics.Default.IncCounter("lru_eviction") - metrics.Default.UpdateGauge("lru_bytes", float64(lru.lru.Bytes())) }) var stor Storage = lru @@ -80,6 +79,7 @@ func NewCacheFilter(maxBytes int64, listenAddr string, netOpts skpnet.Options, v listenAddr: listenAddr, client: skpnet.NewClient(netOpts), storage: stor, + lruStorage: lru, } } @@ -87,7 +87,8 @@ type cacheSpec struct { maxBytes int64 listenAddr string client *skpnet.Client - storage Storage // shared across all filter instances + storage Storage // shared across all filter instances + lruStorage *LRUStorage // the L1 LRU backing storage; nil only if storage is not LRU-backed } func (s *cacheSpec) Name() string { return filterName } @@ -143,6 +144,7 @@ func (s *cacheSpec) CreateFilter(args []interface{}) (filters.Filter, error) { cf := &cacheFilter{ storage: s.storage, + lruStorage: s.lruStorage, listenAddr: s.listenAddr, ttl: ttl, errorTTL: errorTTL, @@ -152,17 +154,20 @@ func (s *cacheSpec) CreateFilter(args []interface{}) (filters.Filter, error) { metrics: metrics.Default, keyHeaders: keyHeaders, revalJobs: make(chan revalJob, revalQueueSize), + lruBytesDone: make(chan struct{}), } cf.fetch = s.client.Do go cf.revalidationWorker() + go cf.lruBytesScraper() return cf, nil } -// Close shuts down the background revalidation worker. Must be called when the -// filter is no longer in use (e.g. in tests via t.Cleanup). +// Close shuts down the background revalidation worker and the lru_bytes scraper. +// Must be called when the filter is no longer in use (e.g. in tests via t.Cleanup). func (f *cacheFilter) Close() { close(f.revalJobs) + close(f.lruBytesDone) } // revalidationWorker is the single background goroutine per filter that @@ -174,6 +179,27 @@ func (f *cacheFilter) revalidationWorker() { log.Debug("cache: revalidation worker stopped") } +const lruBytesScrapeInterval = 10 * time.Second + +// lruBytesScraper periodically updates the lru_bytes gauge so it stays current +// even when no evictions occur (large Sets without exceeding capacity never +// trigger the onEvict callback). It exits when lruBytesDone is closed. +func (f *cacheFilter) lruBytesScraper() { + if f.lruStorage == nil { + return + } + ticker := time.NewTicker(lruBytesScrapeInterval) + defer ticker.Stop() + for { + select { + case <-ticker.C: + f.metrics.UpdateGauge("lru_bytes", float64(f.lruStorage.lru.Bytes())) + case <-f.lruBytesDone: + return + } + } +} + type revalJob struct { key string req *http.Request // pre-cloned, safe to use after the originating request ends @@ -181,6 +207,7 @@ type revalJob struct { type cacheFilter struct { storage Storage + lruStorage *LRUStorage // non-nil when backed by LRU (always in L1-only; L1 within Valkey path) listenAddr string ttl time.Duration errorTTL time.Duration @@ -190,12 +217,13 @@ type cacheFilter struct { // rfcMode true: upstream Cache-Control is authoritative (cache()). // false: operator ttl/errorTTL/swrWindow are authoritative (force mode). - rfcMode bool - coldSF singleflight.Group // cold-miss coalescing - revalSF singleflight.Group // coalesces concurrent background revalidations per key - revalJobs chan revalJob // background revalidation queue; worker drains this - fetch func(*http.Request) (*http.Response, error) - metrics metrics.Metrics + rfcMode bool + coldSF singleflight.Group // cold-miss coalescing + revalSF singleflight.Group // coalesces concurrent background revalidations per key + revalJobs chan revalJob // background revalidation queue; worker drains this + lruBytesDone chan struct{} // closed by Close() to stop the lruBytesScraper goroutine + fetch func(*http.Request) (*http.Response, error) + metrics metrics.Metrics } // Request checks the cache. On a hit it calls ctx.Serve() to short-circuit the diff --git a/filters/cache/filter_test.go b/filters/cache/filter_test.go index 24e9351d95..2133150a7d 100644 --- a/filters/cache/filter_test.go +++ b/filters/cache/filter_test.go @@ -249,6 +249,17 @@ func TestCacheFilter_TTLExpiry(t *testing.T) { }) } +func TestCacheFilter_Response_NoopIfStateBagKeyMissing(t *testing.T) { + // Regression: Response() used a bare type assertion on stateBagKey which + // panicked if Request() had not run (e.g. route misconfiguration). + f := newTestFilter(t, time.Minute, 15*time.Second, time.Minute) + ctx := newCtx("GET", "https://example.com/api", "") + // Deliberately do NOT call f.Request(ctx) — state bag has no cache key. + ctx.FResponse = upstreamResponse(http.StatusOK, `{}`) + // Must not panic. + f.Response(ctx) +} + func TestCreateFilter_InvalidArgs(t *testing.T) { spec := NewCacheFilter(1<<20, "localhost:9090", skpnet.Options{}, nil) t.Cleanup(spec.(*cacheSpec).client.Close) @@ -578,6 +589,39 @@ func TestCacheFilter_RequestOnlyIfCached_Hit_ServesFromCache(t *testing.T) { } } +func TestCacheFilter_RequestOnlyIfCached_StaleWhileRevalidate_ServesStale(t *testing.T) { + // RFC 9111 §5.2.1.7: only-if-cached should return a stored response if it is + // "usable" — entries in the SWR window are still being served as stale to + // other clients, so they are usable and must not return 504. + f := newTestFilter(t, time.Millisecond, 15*time.Second, time.Minute) + url := "https://cdn.contentful.com/spaces/abc/entries/oic-swr" + + synctest.Test(t, func(t *testing.T) { + // Populate cache. + ctx1 := newCtx("GET", url, "") + f.Request(ctx1) + ctx1.FResponse = upstreamResponseCC(http.StatusOK, `{"data":"cached"}`, "max-age=0") + f.Response(ctx1) + + // Advance past TTL (1ms) but stay within SWR window (1 minute). + time.Sleep(50 * time.Millisecond) + + ctx2 := newCtx("GET", url, "") + ctx2.FRequest.Header.Set("Cache-Control", "only-if-cached") + f.Request(ctx2) + + if !ctx2.FServed { + t.Fatal("only-if-cached must serve during SWR window") + } + if ctx2.FResponse.StatusCode == http.StatusGatewayTimeout { + t.Fatal("only-if-cached must not return 504 for SWR-window entry") + } + if ctx2.FResponse.StatusCode != http.StatusOK { + t.Fatalf("expected 200, got %d", ctx2.FResponse.StatusCode) + } + }) +} + func TestCacheFilter_AgeHeader_HIT(t *testing.T) { f := newTestFilter(t, time.Minute, 15*time.Second, time.Hour) url := "https://cdn.contentful.com/spaces/abc/entries/age" @@ -2240,7 +2284,12 @@ func TestCacheFilter_MinFresh_InsufficientFreshness_Bypasses(t *testing.T) { func TestCacheFilter_StaleIfError_Serves_On_5xx(t *testing.T) { // ttl=1ms, errorTTL=10s, swrWindow=1ms, staleIfError=60s // Entry expires after 1ms; staleIfError=60s keeps it in storage. - // A 503 upstream should cause the stale entry to be served. + // A 503 upstream via coalesce should cause the stale entry to be served. + // + // Regression: the SIE block in Response() was dead code — coalesce() always calls + // ctx.Serve() (even on 5xx), so Response() returned early before reaching it. + // SIE logic must live inside coalesce() with the pre-fetch snapshot captured + // before f.fetch runs, preventing the 5xx from overwriting the stored entry. f := newTestFilter(t, time.Millisecond, 10*time.Second, time.Millisecond, 60*time.Second) url := "http://example.com/sie-5xx" @@ -2250,15 +2299,23 @@ func TestCacheFilter_StaleIfError_Serves_On_5xx(t *testing.T) { ctx.FResponse = upstreamResponseCC(http.StatusOK, `{"data":"cached"}`, "max-age=0") f.Response(ctx) - // Advance past TTL+SWR (SWR=0) so the entry is stale, but within staleIfError window. + // Advance past TTL+SWR so the next Request() calls coalesce(). time.Sleep(50 * time.Millisecond) + // Coalesce fetches from upstream and receives 503. + f.fetch = func(*http.Request) (*http.Response, error) { + return &http.Response{ + StatusCode: http.StatusServiceUnavailable, + Header: http.Header{}, + Body: io.NopCloser(strings.NewReader("")), + }, nil + } ctx2 := newCtx(http.MethodGet, url, "") f.Request(ctx2) - // ctx2 is a MISS (past TTL+SWR); upstream returns 503. - ctx2.FResponse = upstreamResponse(http.StatusServiceUnavailable, "") - f.Response(ctx2) + if ctx2.FResponse == nil { + t.Fatal("expected a response, got nil") + } if ctx2.FResponse.StatusCode != http.StatusOK { t.Fatalf("want 200 from stale-if-error, got %d", ctx2.FResponse.StatusCode) } @@ -2271,6 +2328,8 @@ func TestCacheFilter_StaleIfError_Serves_On_5xx(t *testing.T) { func TestCacheFilter_StaleIfError_Expired_NotServed(t *testing.T) { // ttl=1ms, errorTTL=10s, swrWindow=1ms, staleIfError=100ms // Sleep 200ms — past TTL + staleIfError window. Entry too old for stale-if-error. + // Uses f.fetch returning 503 via coalesce (the same path as the positive SIE case) + // to confirm the 503 is passed through when the SIE window has already elapsed. f := newTestFilter(t, time.Millisecond, 10*time.Second, time.Millisecond, 100*time.Millisecond) url := "http://example.com/sie-expired" @@ -2283,11 +2342,19 @@ func TestCacheFilter_StaleIfError_Expired_NotServed(t *testing.T) { // Advance past TTL (1ms) + staleIfError (100ms) = well beyond 200ms. time.Sleep(200 * time.Millisecond) + f.fetch = func(*http.Request) (*http.Response, error) { + return &http.Response{ + StatusCode: http.StatusServiceUnavailable, + Header: http.Header{}, + Body: io.NopCloser(strings.NewReader("")), + }, nil + } ctx2 := newCtx(http.MethodGet, url, "") f.Request(ctx2) - ctx2.FResponse = upstreamResponse(http.StatusServiceUnavailable, "") - f.Response(ctx2) + if ctx2.FResponse == nil { + t.Fatal("expected a response, got nil") + } if ctx2.FResponse.StatusCode != http.StatusServiceUnavailable { t.Fatalf("want 503 (SIE window expired), got %d", ctx2.FResponse.StatusCode) } @@ -2534,6 +2601,45 @@ func TestCacheFilter_PureRFCMode_ZeroArgs_UsesUpstreamMaxAge(t *testing.T) { } } +func TestCacheFilter_LRUBytesGaugeUpdatesWithoutEviction(t *testing.T) { + // The filter and its goroutines must be created inside the synctest bubble + // so the ticker in lruBytesScraper is subject to synthetic time control. + // f.fetch is replaced before any network I/O so the transport goroutine + // being inside the bubble is safe (it never actually dials out). + synctest.Test(t, func(t *testing.T) { + f := newTestFilter(t, 5*time.Minute, 15*time.Second, 5*time.Minute) + mockMetrics := &metricstest.MockMetrics{} + // synctest.Wait drains goroutine scheduling so the scraper is parked at the + // select before we swap f.metrics. No tick has fired yet (synthetic time is frozen). + synctest.Wait() + f.metrics = mockMetrics + + f.fetch = func(_ *http.Request) (*http.Response, error) { + return upstreamResponseCC(http.StatusOK, `{"data":"hello"}`, "max-age=300"), nil + } + + var initialBytes float64 + mockMetrics.WithGauges(func(g map[string]float64) { + initialBytes = g["lru_bytes"] + }) + + // Store an entry large enough to be visible but not enough to evict. + ctx := newCtx("GET", "https://example.com/lru-bytes-scrape", "") + f.Request(ctx) + + // Advance time past one scrape interval (10 s). + time.Sleep(11 * time.Second) + + var afterBytes float64 + mockMetrics.WithGauges(func(g map[string]float64) { + afterBytes = g["lru_bytes"] + }) + if afterBytes <= initialBytes { + t.Errorf("expected lru_bytes to increase after Set without eviction; before=%v after=%v", initialBytes, afterBytes) + } + }) +} + func TestCacheFilter_PureRFCMode_ZeroArgs_NoUpstreamDirective_NotCached(t *testing.T) { // cache() with no args: when upstream sends no Cache-Control, no Expires, // and no Last-Modified, nothing should be cached (no heuristic without Last-Modified). From 716bff8ef3e671987eeebd6f6b77e468f5c5daaf Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Tue, 26 May 2026 14:59:03 +0200 Subject: [PATCH 05/89] feat: log storage errors with per-site messages Signed-off-by: Larry D Almeida --- filters/cache/filter.go | 32 ++++++++++++++++++++++++-------- 1 file changed, 24 insertions(+), 8 deletions(-) diff --git a/filters/cache/filter.go b/filters/cache/filter.go index 9a4b5a9be6..9559266785 100644 --- a/filters/cache/filter.go +++ b/filters/cache/filter.go @@ -440,7 +440,9 @@ func (f *cacheFilter) coalesce(ctx filters.FilterContext, key string) { ResponseTime: responseTime, } if shouldStore { - _ = f.storage.Set(context.Background(), key, entry) + if err := f.storage.Set(context.Background(), key, entry); err != nil { + log.WithError(err).Debug("cache: Set failed (cold-miss store)") + } } return &coalesceResult{entry: entry, stored: sieStored}, nil }) @@ -508,7 +510,9 @@ func (f *cacheFilter) Response(ctx filters.FilterContext) { if lm := rsp.Header.Get("Last-Modified"); lm != "" { stored.LastModified = lm } - _ = f.storage.Set(ctx.Request().Context(), key, stored) + if err := f.storage.Set(ctx.Request().Context(), key, stored); err != nil { + log.WithError(err).Debug("cache: Set failed (HEAD freshen)") + } } return } @@ -521,12 +525,18 @@ func (f *cacheFilter) Response(ctx filters.FilterContext) { // RFC 9111 §4.4: invalidate cached entry on successful unsafe method. // Also invalidate same-origin Location/Content-Location URIs. if isUnsafeMethod(ctx.Request().Method) && rsp.StatusCode < 400 { - _ = f.storage.Delete(ctx.Request().Context(), key) + if err := f.storage.Delete(ctx.Request().Context(), key); err != nil { + log.WithError(err).Warn("cache: Delete failed (unsafe method invalidation)") + } for _, hdrName := range []string{"Location", "Content-Location"} { if loc := rsp.Header.Get(hdrName); loc != "" && sameOrigin(ctx.Request(), loc) { if locKey := cacheKeyForURL(ctx.RouteId(), ctx.Request(), loc, f.keyHeaders); locKey != "" { - _ = f.storage.Delete(ctx.Request().Context(), locKey) - _ = f.storage.Delete(ctx.Request().Context(), "vary:"+locKey) + if err := f.storage.Delete(ctx.Request().Context(), locKey); err != nil { + log.WithError(err).Debug("cache: Delete failed (Location invalidation)") + } + if err := f.storage.Delete(ctx.Request().Context(), "vary:"+locKey); err != nil { + log.WithError(err).Debug("cache: Delete failed (vary sentinel for Location)") + } } } } @@ -588,7 +598,9 @@ func (f *cacheFilter) Response(ctx filters.FilterContext) { StaleWhileRevalidate: f.swrWindow, VaryHeaders: varyNames, } - _ = f.storage.Set(ctx.Request().Context(), "vary:"+baseKey, sentinel) + if err := f.storage.Set(ctx.Request().Context(), "vary:"+baseKey, sentinel); err != nil { + log.WithError(err).Debug("cache: Set failed (vary sentinel)") + } } swr := f.swrWindow @@ -619,7 +631,9 @@ func (f *cacheFilter) Response(ctx filters.FilterContext) { CorrectedInitialAge: cia, ResponseTime: responseTime, } - _ = f.storage.Set(ctx.Request().Context(), storeKey, entry) + if err := f.storage.Set(ctx.Request().Context(), storeKey, entry); err != nil { + log.WithError(err).Debug("cache: Set failed (response store)") + } } // enqueueRevalidation clones the request before sending a job to the background @@ -723,7 +737,9 @@ func (f *cacheFilter) doRevalidate(key string, req *http.Request) { CorrectedInitialAge: cia, ResponseTime: responseTime, } - _ = f.storage.Set(context.Background(), key, entry) + if err := f.storage.Set(context.Background(), key, entry); err != nil { + log.WithError(err).Debug("cache: Set failed (background revalidation)") + } return nil, nil }) } From cac6a618e5a405362804dff584b945aec548ed52 Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Tue, 26 May 2026 15:26:10 +0200 Subject: [PATCH 06/89] test: must-revalidate forces coalesce when stale Signed-off-by: Larry D Almeida --- filters/cache/filter_test.go | 42 ++++++++++++++++++++++++++++++++++++ 1 file changed, 42 insertions(+) diff --git a/filters/cache/filter_test.go b/filters/cache/filter_test.go index 2133150a7d..a350a6a18d 100644 --- a/filters/cache/filter_test.go +++ b/filters/cache/filter_test.go @@ -1229,6 +1229,48 @@ func TestCacheFilter_SMaxAge_ImpliesProxyRevalidate(t *testing.T) { }) } +func TestCacheFilter_MustRevalidate_ForcesCoalesceWhenStale(t *testing.T) { + // RFC 9111 §5.2.2.2: must-revalidate forbids serving a stale response. + // Once the entry is past TTL, coalesce() must contact the origin even if the + // entry is inside a stale-while-revalidate window. + f := newTestFilter(t, 100*time.Millisecond, 15*time.Second, time.Hour) + url := "https://cdn.contentful.com/spaces/abc/entries/must-reval" + + var fetchCount int64 + f.fetch = func(req *http.Request) (*http.Response, error) { + atomic.AddInt64(&fetchCount, 1) + return &http.Response{ + StatusCode: http.StatusOK, + Header: http.Header{"Cache-Control": {"must-revalidate"}, "Content-Type": {"application/json"}}, + Body: io.NopCloser(strings.NewReader(`{"v":1}`)), + }, nil + } + + synctest.Test(t, func(t *testing.T) { + // First request: cold miss → fetch → store. + ctx1 := newCtx("GET", url, "") + f.Request(ctx1) + if ctx1.FResponse.Header.Get("X-Cache-Status") != "MISS" { + t.Fatalf("expected MISS, got %q", ctx1.FResponse.Header.Get("X-Cache-Status")) + } + + // Advance into stale window (past TTL=100ms, within SWR=1h). + time.Sleep(200 * time.Millisecond) + + // Second request: entry is stale + must-revalidate → must NOT serve stale. + // coalesce() must call fetch again (origin contacted). + ctx2 := newCtx("GET", url, "") + f.Request(ctx2) + synctest.Wait() + if atomic.LoadInt64(&fetchCount) < 2 { + t.Fatalf("must-revalidate must block stale serve and trigger upstream fetch; fetchCount=%d", atomic.LoadInt64(&fetchCount)) + } + if status := ctx2.FResponse.Header.Get("X-Cache-Status"); status == "HIT" || status == "STALE" { + t.Errorf("expected origin fetch, but got X-Cache-Status: %s", status) + } + }) +} + func TestCacheFilter_SharedStorage_RouteIsolation(t *testing.T) { spec := NewCacheFilter(1<<20, "localhost:9090", skpnet.Options{}, nil) t.Cleanup(spec.(*cacheSpec).client.Close) From 44e1d72211733d5482b7c0f34768a73334079190 Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Tue, 26 May 2026 15:58:35 +0200 Subject: [PATCH 07/89] test: unsafe method + 4xx does not invalidate cached entry Signed-off-by: Larry D Almeida --- filters/cache/filter_test.go | 44 ++++++++++++++++++++++++++++++++++++ 1 file changed, 44 insertions(+) diff --git a/filters/cache/filter_test.go b/filters/cache/filter_test.go index a350a6a18d..5a709d7b95 100644 --- a/filters/cache/filter_test.go +++ b/filters/cache/filter_test.go @@ -1014,6 +1014,50 @@ func TestCacheFilter_UnsafeMethod_InvalidatesCache(t *testing.T) { } } +func TestCacheFilter_UnsafeMethod_4xx_DoesNotInvalidate(t *testing.T) { + // RFC 9111 §4.4: a cache MUST NOT invalidate a stored response when an + // unsafe method returns a 4xx status. Only 2xx responses must trigger + // invalidation. + f := newTestFilter(t, time.Minute, 15*time.Second, time.Minute) + url := "https://cdn.contentful.com/spaces/abc/entries/item-4xx" + + // Step 1: warm the cache with a GET that returns 200 + max-age=300. + ctx1 := newCtx("GET", url, "") + f.Request(ctx1) + ctx1.FResponse = upstreamResponseCC(http.StatusOK, `{"data":"cached"}`, "public, max-age=300") + f.Response(ctx1) + + // Verify the entry is cached (second GET must be a HIT). + ctxHit := newCtx("GET", url, "") + f.Request(ctxHit) + if !ctxHit.FServed { + t.Fatal("expected HIT after warming the cache") + } + if ctxHit.FResponse.Header.Get("X-Cache-Status") != "HIT" { + t.Fatalf("expected HIT status, got %q", ctxHit.FResponse.Header.Get("X-Cache-Status")) + } + + // Step 2: DELETE request that returns 403 — must NOT invalidate the cache. + deleteCtx := newCtx("DELETE", url, "") + f.Request(deleteCtx) + deleteCtx.FResponse = &http.Response{ + StatusCode: http.StatusForbidden, + Header: http.Header{}, + Body: http.NoBody, + } + f.Response(deleteCtx) + + // Step 3: another GET must still return the cached entry (HIT). + ctxAfter := newCtx("GET", url, "") + f.Request(ctxAfter) + if !ctxAfter.FServed { + t.Fatal("cache must NOT be invalidated when unsafe method returns 4xx (RFC 9111 §4.4)") + } + if ctxAfter.FResponse.Header.Get("X-Cache-Status") != "HIT" { + t.Fatalf("expected HIT after 4xx unsafe method, got %q", ctxAfter.FResponse.Header.Get("X-Cache-Status")) + } +} + func TestCacheFilter_SafeMethod_DoesNotInvalidate(t *testing.T) { f := newTestFilter(t, time.Minute, 15*time.Second, time.Minute) url := "https://cdn.contentful.com/spaces/abc/entries/safe" From e8f08cebddc159e32a67c8bb78fc4523110e1f46 Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Tue, 26 May 2026 16:54:54 +0200 Subject: [PATCH 08/89] test: oversized LRU entry increments lru_oversized and is not stored Signed-off-by: Larry D Almeida --- filters/cache/lru_test.go | 38 ++++++++++++++++++++++++++++++++++++++ 1 file changed, 38 insertions(+) diff --git a/filters/cache/lru_test.go b/filters/cache/lru_test.go index a6e2354903..bccdef6160 100644 --- a/filters/cache/lru_test.go +++ b/filters/cache/lru_test.go @@ -7,6 +7,8 @@ import ( "net/http" "testing" "time" + + "github.com/zalando/skipper/metrics" ) func makeEntry(payload string, ttl time.Duration) *Entry { @@ -173,6 +175,42 @@ func TestLRUStorage_EvictionCallbackDoesNotDeadlock(t *testing.T) { } } +func TestLRUStorage_OversizedEntry(t *testing.T) { + // With 256 shards and 1 KB total capacity, each shard holds 4 bytes. + // A payload larger than 4 bytes exceeds every shard's maxBytes. + const totalBytes = 1024 // 1 KB → 4 bytes per shard + s := NewLRUStorage(totalBytes, nil) + + // Inject a testMetrics so we can observe lru_oversized counter increments. + // metrics.Default is restored automatically when the test ends. + m := &testMetrics{} + orig := metrics.Default + metrics.Default = m + t.Cleanup(func() { metrics.Default = orig }) + + ctx := context.Background() + entry := makeEntry(string(make([]byte, 1000)), time.Minute) // 1000-byte payload ≫ 4-byte shard + + // Set must succeed (nil error) even though the entry is too large to store. + if err := s.Set(ctx, "oversized", entry); err != nil { + t.Fatalf("Set returned unexpected error: %v", err) + } + + // The lru_oversized counter must have been incremented exactly once. + if got := m.counter("lru_oversized"); got != 1 { + t.Errorf("lru_oversized counter: got %d, want 1", got) + } + + // The entry must not have been stored — Get must return nil. + got, err := s.Get(ctx, "oversized") + if err != nil { + t.Fatalf("Get returned unexpected error: %v", err) + } + if got != nil { + t.Errorf("expected Get to return nil for oversized entry, got %+v", got) + } +} + func TestLRUShard_CrossKeyEviction(t *testing.T) { // Test lruShard directly to make eviction deterministic — no hash routing. dataA := []byte("aaaa") From b64ab9d7ed0267899a369c0529dd82ef2c4c35fe Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Tue, 26 May 2026 17:02:20 +0200 Subject: [PATCH 09/89] test: reval_dropped and L1 fallback write verification Signed-off-by: Larry D Almeida --- filters/cache/filter_test.go | 89 ++++++++++++++++++++++++++++ filters/cache/valkey_storage_test.go | 10 ++++ 2 files changed, 99 insertions(+) diff --git a/filters/cache/filter_test.go b/filters/cache/filter_test.go index 5a709d7b95..5678ad304e 100644 --- a/filters/cache/filter_test.go +++ b/filters/cache/filter_test.go @@ -2756,6 +2756,95 @@ func TestCacheFilter_PureRFCMode_ZeroArgs_NoUpstreamDirective_NotCached(t *testi } } +func TestCacheFilter_RevalDropped_WhenQueueFull(t *testing.T) { + // Verify that reval_dropped is incremented when the revalJobs channel is at + // capacity and a stale-while-revalidate request tries to enqueue a job. + // + // Strategy: + // 1. Wire f.fetch to block until we release it, so the worker goroutine is + // stuck inside doRevalidate as soon as it picks up its first job. + // 2. Pre-send one dummy job to wake the worker and block it on fetch. + // 3. Fill the remaining revalQueueSize-1 slots with dummy jobs, saturating + // the channel (worker is blocked and cannot drain). + // 4. Seed the cache directly with a backdated stale entry (past TTL, inside + // SWR window) so no real upstream call is needed to populate the cache. + // 5. Make a GET — the stale path serves the entry and calls enqueueRevalidation; + // the channel is full, so the default branch fires and increments reval_dropped. + // 6. Assert reval_dropped == 1. + + f := newTestFilter(t, time.Millisecond, 15*time.Second, time.Hour) + + // Wire up a dedicated MockMetrics so we can inspect counters in isolation. + mockMetrics := &metricstest.MockMetrics{} + f.metrics = mockMetrics + + // fetchBlocked gates the worker: it blocks until the test releases it. + // workerIn is signalled once the worker is confirmed to be inside fetch. + fetchBlocked := make(chan struct{}) + workerIn := make(chan struct{}, 1) + f.fetch = func(req *http.Request) (*http.Response, error) { + select { + case workerIn <- struct{}{}: // signal first entry only (buffered size 1) + default: + } + <-fetchBlocked // block until test closes this channel + return nil, errors.New("blocked fetch") + } + + // Send one dummy job so the worker goroutine wakes and blocks inside fetch. + dummyReq, _ := http.NewRequest(http.MethodGet, "http://example.com/dummy", nil) + f.revalJobs <- revalJob{key: "dummy-wake", req: dummyReq} + + // Wait for the worker to confirm it is inside fetch — no timing guesswork. + <-workerIn + + // Worker has consumed the dummy job from the channel (0/256 slots occupied) + // and is now blocked in fetch. Fill all revalQueueSize slots so the channel + // is at capacity; the worker cannot drain while it is stuck in fetch. + for i := range revalQueueSize { + r, _ := http.NewRequest(http.MethodGet, "http://example.com/fill", nil) + f.revalJobs <- revalJob{key: "fill-" + strconv.Itoa(i), req: r} + } + + // Inject a stale entry directly into storage. CreatedAt is backdated so + // IsStale(now) returns true (past TTL) and IsUsable(now) returns true (within SWR). + url := "https://cdn.contentful.com/spaces/abc/entries/reval-dropped" + req, _ := http.NewRequest(http.MethodGet, url, nil) + key := cacheKey("" /* routeID */, req, nil) + staleEntry := &Entry{ + StatusCode: http.StatusOK, + Header: http.Header{"Content-Type": {"application/json"}}, + Payload: []byte(`{"data":"stale"}`), + CreatedAt: time.Now().Add(-10 * time.Millisecond), // well past 1ms TTL + TTL: time.Millisecond, + StaleWhileRevalidate: time.Hour, // still inside SWR window + } + if err := f.storage.Set(context.Background(), key, staleEntry); err != nil { + t.Fatalf("failed to seed stale entry: %v", err) + } + + // Make the GET request. The filter finds the stale entry, serves it, and + // calls enqueueRevalidation. The channel is full — reval_dropped fires. + ctx := newCtx(http.MethodGet, url, "") + ctx.FMetrics = mockMetrics + f.Request(ctx) + + if !ctx.FServed { + t.Fatal("expected stale entry to be served when revalJobs queue is full") + } + if ctx.FResponse.Header.Get("X-Cache-Status") != "STALE" { + t.Fatalf("expected X-Cache-Status: STALE, got %q", ctx.FResponse.Header.Get("X-Cache-Status")) + } + mockMetrics.WithCounters(func(counters map[string]int64) { + if counters["reval_dropped"] != 1 { + t.Errorf("expected reval_dropped==1, got %d", counters["reval_dropped"]) + } + }) + + // Release the blocked worker so the test can clean up without leaking goroutines. + close(fetchBlocked) +} + func Benchmark_malicious_matchesETag(b *testing.B) { ifNoneMatch := strings.Repeat(",", http.DefaultMaxHeaderBytes) b.ReportAllocs() diff --git a/filters/cache/valkey_storage_test.go b/filters/cache/valkey_storage_test.go index 7f25a82e4e..afb0c7a551 100644 --- a/filters/cache/valkey_storage_test.go +++ b/filters/cache/valkey_storage_test.go @@ -212,6 +212,16 @@ func TestValkeyStorage_FallsBackToL1OnValkeyUnavailable(t *testing.T) { if m.counter("valkey_get_fallback") == 0 { t.Error("expected valkey_get_fallback to be incremented on Get fallback path") } + + // Confirm the entry was physically written to L1 — not just returned via some + // other path. A direct read from LRUStorage proves the write actually happened. + l1Entry, err := lru.Get(ctx, key) + if err != nil { + t.Fatalf("L1 direct Get: %v", err) + } + if l1Entry == nil { + t.Error("expected entry to be written to L1 on Valkey fallback, but L1 Get returned nil") + } } func TestValkeyStorage_RecordsValkeyMiss(t *testing.T) { From 54739d1b9a9f5403dbfa03a3f7193a5afac12923 Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Tue, 26 May 2026 17:33:57 +0200 Subject: [PATCH 10/89] feat: tag cache_status, cache_key, cache_ttl_remaining_ms on trace span Signed-off-by: Larry D Almeida --- filters/cache/filter.go | 22 ++++++++++++++++++++++ 1 file changed, 22 insertions(+) diff --git a/filters/cache/filter.go b/filters/cache/filter.go index 9559266785..3d0178b711 100644 --- a/filters/cache/filter.go +++ b/filters/cache/filter.go @@ -13,6 +13,7 @@ import ( "strings" "time" + opentracing "github.com/opentracing/opentracing-go" log "github.com/sirupsen/logrus" "github.com/zalando/skipper/filters" "github.com/zalando/skipper/metrics" @@ -226,6 +227,21 @@ type cacheFilter struct { metrics metrics.Metrics } +// tagSpan tags the cache decision onto the active OpenTracing span, if one exists. +// key is the cache lookup key; ttlRemainingMs is milliseconds of TTL left (-1 = not applicable). +// Filters do not own their span — they tag onto the proxy-managed request_filters span. +func tagSpan(ctx filters.FilterContext, status, key string, ttlRemainingMs int64) { + span := opentracing.SpanFromContext(ctx.Request().Context()) + if span == nil { + return + } + span.SetTag("cache_status", status) + span.SetTag("cache_key", key) + if ttlRemainingMs >= 0 { + span.SetTag("cache_ttl_remaining_ms", ttlRemainingMs) + } +} + // Request checks the cache. On a hit it calls ctx.Serve() to short-circuit the // backend roundtrip. On a stale hit it serves stale and fires a background // revalidation. On a miss it stores the computed key in the state bag so @@ -306,6 +322,7 @@ func (f *cacheFilter) Request(ctx filters.FilterContext) { } } rsp.Header.Set(cacheStatusHeader, cacheStatusStale) + tagSpan(ctx, cacheStatusStale, key, -1) setAgeHeader(rsp, entry, now) ctx.Metrics().IncCounter("stale") method := ctx.Request().Method @@ -343,6 +360,7 @@ func (f *cacheFilter) Request(ctx filters.FilterContext) { } rsp.Header.Set(cacheStatusHeader, cacheStatusHit) + tagSpan(ctx, cacheStatusHit, key, max(0, entry.TTL-time.Since(entry.CreatedAt)).Milliseconds()) setAgeHeader(rsp, entry, time.Now()) ctx.Metrics().IncCounter("hit") method := ctx.Request().Method @@ -466,6 +484,7 @@ func (f *cacheFilter) coalesce(ctx filters.FilterContext, key string) { Body: io.NopCloser(bytes.NewReader(cr.stored.Payload)), } staleRsp.Header.Set(cacheStatusHeader, cacheStatusStale) + tagSpan(ctx, cacheStatusStale, key, -1) setAgeHeader(staleRsp, cr.stored, time.Now()) ctx.Serve(headBodyOmitted(ctx.Request().Method, staleRsp)) return @@ -477,6 +496,7 @@ func (f *cacheFilter) coalesce(ctx filters.FilterContext, key string) { Body: io.NopCloser(bytes.NewReader(entry.Payload)), } rsp.Header.Set(cacheStatusHeader, cacheStatusMiss) + tagSpan(ctx, cacheStatusMiss, key, -1) ctx.Metrics().IncCounter("miss") ctx.Serve(headBodyOmitted(ctx.Request().Method, rsp)) case <-ctx.Request().Context().Done(): @@ -545,11 +565,13 @@ func (f *cacheFilter) Response(ctx filters.FilterContext) { if ctx.StateBag()[stateBagNoStore] == true { rsp.Header.Set(cacheStatusHeader, cacheStatusMiss) + tagSpan(ctx, cacheStatusMiss, key, -1) ctx.Metrics().IncCounter("miss") return } rsp.Header.Set(cacheStatusHeader, cacheStatusMiss) + tagSpan(ctx, cacheStatusMiss, key, -1) ctx.Metrics().IncCounter("miss") // Vary: * means every response is unique — never cache. From 08e4865f106577b321e2c0dd6bc71b5291483c41 Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Wed, 27 May 2026 08:48:38 +0200 Subject: [PATCH 11/89] cache: promote storage Set/Delete error logs from Debug to Warn Signed-off-by: Larry D Almeida --- filters/cache/filter.go | 60 ++++++++++++++++++++--------------------- 1 file changed, 30 insertions(+), 30 deletions(-) diff --git a/filters/cache/filter.go b/filters/cache/filter.go index 3d0178b711..7a361299ef 100644 --- a/filters/cache/filter.go +++ b/filters/cache/filter.go @@ -70,16 +70,16 @@ func NewCacheFilter(maxBytes int64, listenAddr string, netOpts skpnet.Options, v metrics.Default.IncCounter("lru_eviction") }) - var stor Storage = lru + var store Storage = lru if valkeyRing != nil { - stor = NewValkeyStorage(valkeyRing, lru, metrics.Default) + store = NewValkeyStorage(valkeyRing, lru, metrics.Default) } return &cacheSpec{ maxBytes: maxBytes, listenAddr: listenAddr, client: skpnet.NewClient(netOpts), - storage: stor, + storage: store, lruStorage: lru, } } @@ -89,7 +89,7 @@ type cacheSpec struct { listenAddr string client *skpnet.Client storage Storage // shared across all filter instances - lruStorage *LRUStorage // the L1 LRU backing storage; nil only if storage is not LRU-backed + lruStorage *LRUStorage // always non-nil; direct reference to L1, even when storage is ValkeyStorage } func (s *cacheSpec) Name() string { return filterName } @@ -221,15 +221,14 @@ type cacheFilter struct { rfcMode bool coldSF singleflight.Group // cold-miss coalescing revalSF singleflight.Group // coalesces concurrent background revalidations per key - revalJobs chan revalJob // background revalidation queue; worker drains this - lruBytesDone chan struct{} // closed by Close() to stop the lruBytesScraper goroutine + revalJobs chan revalJob // background revalidation queue; worker drains this + lruBytesDone chan struct{} // closed by Close() to stop the lruBytesScraper goroutine fetch func(*http.Request) (*http.Response, error) metrics metrics.Metrics } -// tagSpan tags the cache decision onto the active OpenTracing span, if one exists. -// key is the cache lookup key; ttlRemainingMs is milliseconds of TTL left (-1 = not applicable). -// Filters do not own their span — they tag onto the proxy-managed request_filters span. +// tagSpan sets cache_status, cache_key, and (when >= 0) cache_ttl_remaining_ms +// on the active OpenTracing span. No-op when no span is present. func tagSpan(ctx filters.FilterContext, status, key string, ttlRemainingMs int64) { span := opentracing.SpanFromContext(ctx.Request().Context()) if span == nil { @@ -377,18 +376,18 @@ func (f *cacheFilter) Request(ctx filters.FilterContext) { ctx.Serve(headBodyOmitted(method, rsp)) } -// coalesce gates concurrent cold misses for the same key behind a single -// upstream fetch. All waiters block until the leader's fetch completes, then -// all are served the same response. This prevents the thundering herd on a -// cache miss. -// coalesceResult carries both the fetched entry and any pre-existing stored entry -// that was present before the fetch. The stored entry is used for stale-if-error: -// it must be captured before the 5xx result can overwrite it in storage. +// coalesceResult carries both the fetched entry and any SIE-eligible stored entry +// snapshotted before the fetch. The snapshot is taken before f.fetch() so that a +// 5xx result cannot overwrite it in storage before the stale-if-error check runs. type coalesceResult struct { entry *Entry stored *Entry // snapshot before fetch; nil if no eligible SIE entry existed } +// coalesce gates concurrent cold misses for the same key behind a single upstream +// fetch, preventing thundering herd. All waiters receive the same response. +// Stale-if-error (RFC 5861 §4) is also applied here: a pre-fetch snapshot of any +// eligible stored entry is served on 5xx instead of propagating the error. func (f *cacheFilter) coalesce(ctx filters.FilterContext, key string) { req := ctx.Request().Clone(context.Background()) @@ -459,7 +458,7 @@ func (f *cacheFilter) coalesce(ctx filters.FilterContext, key string) { } if shouldStore { if err := f.storage.Set(context.Background(), key, entry); err != nil { - log.WithError(err).Debug("cache: Set failed (cold-miss store)") + log.WithError(err).Warn("cache: Set failed (cold-miss store)") } } return &coalesceResult{entry: entry, stored: sieStored}, nil @@ -505,7 +504,8 @@ func (f *cacheFilter) coalesce(ctx filters.FilterContext, key string) { } // Response stores the upstream response in the cache, freshens HEAD entries, -// invalidates on unsafe methods, and applies stale-if-error on 5xx. +// and invalidates on unsafe methods. Stale-if-error is handled in coalesce, +// not here, because coalesce calls ctx.Serve() which causes Response to return early. func (f *cacheFilter) Response(ctx filters.FilterContext) { rsp := ctx.Response() key, _ := ctx.StateBag()[stateBagKey].(string) @@ -531,7 +531,7 @@ func (f *cacheFilter) Response(ctx filters.FilterContext) { stored.LastModified = lm } if err := f.storage.Set(ctx.Request().Context(), key, stored); err != nil { - log.WithError(err).Debug("cache: Set failed (HEAD freshen)") + log.WithError(err).Warn("cache: Set failed (HEAD freshen)") } } return @@ -552,10 +552,10 @@ func (f *cacheFilter) Response(ctx filters.FilterContext) { if loc := rsp.Header.Get(hdrName); loc != "" && sameOrigin(ctx.Request(), loc) { if locKey := cacheKeyForURL(ctx.RouteId(), ctx.Request(), loc, f.keyHeaders); locKey != "" { if err := f.storage.Delete(ctx.Request().Context(), locKey); err != nil { - log.WithError(err).Debug("cache: Delete failed (Location invalidation)") + log.WithError(err).Warn("cache: Delete failed (Location invalidation)") } if err := f.storage.Delete(ctx.Request().Context(), "vary:"+locKey); err != nil { - log.WithError(err).Debug("cache: Delete failed (vary sentinel for Location)") + log.WithError(err).Warn("cache: Delete failed (vary sentinel for Location)") } } } @@ -621,7 +621,7 @@ func (f *cacheFilter) Response(ctx filters.FilterContext) { VaryHeaders: varyNames, } if err := f.storage.Set(ctx.Request().Context(), "vary:"+baseKey, sentinel); err != nil { - log.WithError(err).Debug("cache: Set failed (vary sentinel)") + log.WithError(err).Warn("cache: Set failed (vary sentinel)") } } @@ -654,15 +654,13 @@ func (f *cacheFilter) Response(ctx filters.FilterContext) { ResponseTime: responseTime, } if err := f.storage.Set(ctx.Request().Context(), storeKey, entry); err != nil { - log.WithError(err).Debug("cache: Set failed (response store)") + log.WithError(err).Warn("cache: Set failed (response store)") } } -// enqueueRevalidation clones the request before sending a job to the background -// revalidation worker so there is no data race: the clone happens in the calling -// goroutine while orig is still live, and the worker receives a fully independent copy. -// Non-blocking send: if the queue is full the revalidation is dropped rather than -// blocking the request goroutine. +// enqueueRevalidation sends a revalidation job to the background worker. +// The request is cloned in the calling goroutine before orig is released. +// If the queue is full the job is dropped and reval_dropped is incremented. func (f *cacheFilter) enqueueRevalidation(key string, orig *http.Request) { cloned := orig.Clone(context.Background()) select { @@ -672,7 +670,9 @@ func (f *cacheFilter) enqueueRevalidation(key string, orig *http.Request) { } } -// doRevalidate fetches the upstream resource and refreshes the cache entry. +// doRevalidate revalidates key against the upstream. It sends a conditional +// request (If-None-Match / If-Modified-Since) when the stored entry carries +// validators; a 304 response reuses the stored payload and merges new headers. func (f *cacheFilter) doRevalidate(key string, req *http.Request) { f.revalSF.Do(key, func() (interface{}, error) { //nolint:errcheck req.Header.Set(revalidateHeader, "1") @@ -760,7 +760,7 @@ func (f *cacheFilter) doRevalidate(key string, req *http.Request) { ResponseTime: responseTime, } if err := f.storage.Set(context.Background(), key, entry); err != nil { - log.WithError(err).Debug("cache: Set failed (background revalidation)") + log.WithError(err).Warn("cache: Set failed (background revalidation)") } return nil, nil }) From 78503a73b0edd62d2022bb65125064aa82afacac Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Thu, 28 May 2026 08:15:48 +0200 Subject: [PATCH 12/89] Promote log to warn Signed-off-by: Larry D Almeida --- filters/cache/valkey_storage.go | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/filters/cache/valkey_storage.go b/filters/cache/valkey_storage.go index 1f2d47ff7a..4d9f99449f 100644 --- a/filters/cache/valkey_storage.go +++ b/filters/cache/valkey_storage.go @@ -49,7 +49,7 @@ func (s *ValkeyStorage) Get(ctx context.Context, key string) (*Entry, error) { return nil, nil } s.metrics.IncCounter("valkey_get_fallback") - log.WithError(err).Debug("cache: valkey Get failed, falling back to L1") + log.WithError(err).Warn("cache: valkey Get failed, falling back to L1") return s.l1.Get(ctx, key) } var e Entry @@ -72,7 +72,7 @@ func (s *ValkeyStorage) Set(ctx context.Context, key string, entry *Entry) error if err := s.ring.SetWithExpire(ctx, key, string(data), ttl); err != nil { s.metrics.IncCounter("valkey_set_fallback") - log.WithError(err).Debug("cache: valkey Set failed, falling back to L1") + log.WithError(err).Warn("cache: valkey Set failed, falling back to L1") return s.l1.Set(ctx, key, entry) } // Write-around: L1 is not warmed on a successful Valkey Set. Subsequent @@ -85,7 +85,7 @@ func (s *ValkeyStorage) Delete(ctx context.Context, key string) error { // -1*time.Second is required: time.Duration(-1) is -1ns, which truncates to EXPIRE key 0. // Valkey errors are best-effort — L1 delete always runs. if _, err := s.ring.Expire(ctx, key, -1*time.Second); err != nil { - log.WithError(err).Debug("cache: valkey Delete failed") + log.WithError(err).Warn("cache: valkey Delete failed") } return s.l1.Delete(ctx, key) } From eff5cbec6170cf8231ec5593dfc51798e289055f Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Thu, 28 May 2026 09:42:45 +0200 Subject: [PATCH 13/89] cache: injectable metrics, trace spans, lru_bytes scraper Signed-off-by: Larry D Almeida --- filters/cache/filter.go | 16 +++---- filters/cache/lru_storage.go | 12 ++++-- filters/cache/lru_test.go | 22 ++++------ filters/cache/storage.go | 48 ++++++++++----------- filters/cache/valkey_storage_test.go | 62 ++++++++++++++-------------- 5 files changed, 80 insertions(+), 80 deletions(-) diff --git a/filters/cache/filter.go b/filters/cache/filter.go index 7a361299ef..e20e771350 100644 --- a/filters/cache/filter.go +++ b/filters/cache/filter.go @@ -65,14 +65,14 @@ const ( // // -> cache("5m", "15s", "30s", "60s") -> "https://example.org" func NewCacheFilter(maxBytes int64, listenAddr string, netOpts skpnet.Options, valkeyRing *skpnet.ValkeyRingClient) filters.Spec { - var lru *LRUStorage - lru = NewLRUStorage(maxBytes, func() { - metrics.Default.IncCounter("lru_eviction") - }) + m := metrics.Default + lru := NewLRUStorage(maxBytes, func() { + m.IncCounter("lru_eviction") + }, m) var store Storage = lru if valkeyRing != nil { - store = NewValkeyStorage(valkeyRing, lru, metrics.Default) + store = NewValkeyStorage(valkeyRing, lru, m) } return &cacheSpec{ @@ -81,6 +81,7 @@ func NewCacheFilter(maxBytes int64, listenAddr string, netOpts skpnet.Options, v client: skpnet.NewClient(netOpts), storage: store, lruStorage: lru, + metrics: m, } } @@ -90,6 +91,7 @@ type cacheSpec struct { client *skpnet.Client storage Storage // shared across all filter instances lruStorage *LRUStorage // always non-nil; direct reference to L1, even when storage is ValkeyStorage + metrics metrics.Metrics } func (s *cacheSpec) Name() string { return filterName } @@ -152,7 +154,7 @@ func (s *cacheSpec) CreateFilter(args []interface{}) (filters.Filter, error) { swrWindow: swr, staleIfError: staleIfError, rfcMode: rfcMode, - metrics: metrics.Default, + metrics: s.metrics, keyHeaders: keyHeaders, revalJobs: make(chan revalJob, revalQueueSize), lruBytesDone: make(chan struct{}), @@ -208,7 +210,7 @@ type revalJob struct { type cacheFilter struct { storage Storage - lruStorage *LRUStorage // non-nil when backed by LRU (always in L1-only; L1 within Valkey path) + lruStorage *LRUStorage // always non-nil; direct reference to L1, even when storage is ValkeyStorage listenAddr string ttl time.Duration errorTTL time.Duration diff --git a/filters/cache/lru_storage.go b/filters/cache/lru_storage.go index 15efd5ba15..70ab5be1c9 100644 --- a/filters/cache/lru_storage.go +++ b/filters/cache/lru_storage.go @@ -14,13 +14,17 @@ import ( // It owns all cache semantics (serialisation, TTL expiry); ShardedByteLRU // remains a pure byte store. type LRUStorage struct { - lru *ShardedByteLRU + lru *ShardedByteLRU + metrics metrics.Metrics } // NewLRUStorage returns an LRUStorage backed by a ShardedByteLRU sized to totalMaxBytes. -func NewLRUStorage(totalMaxBytes int64, onEvict func()) *LRUStorage { +// m records the lru_oversized counter on oversized Set calls; pass metrics.Default when no +// test-scoped collector is needed. +func NewLRUStorage(totalMaxBytes int64, onEvict func(), m metrics.Metrics) *LRUStorage { return &LRUStorage{ - lru: NewShardedByteLRU(totalMaxBytes, onEvict), + lru: NewShardedByteLRU(totalMaxBytes, onEvict), + metrics: m, } } @@ -59,7 +63,7 @@ func (s *LRUStorage) Set(_ context.Context, key string, entry *Entry) error { "size_bytes": len(data), "shard_max": s.lru.shards[0].maxBytes, }).Warn("cache: entry exceeds shard capacity and will not be stored") - metrics.Default.IncCounter("lru_oversized") + s.metrics.IncCounter("lru_oversized") return nil } s.lru.Set(key, data) diff --git a/filters/cache/lru_test.go b/filters/cache/lru_test.go index bccdef6160..6165b47ddd 100644 --- a/filters/cache/lru_test.go +++ b/filters/cache/lru_test.go @@ -22,7 +22,7 @@ func makeEntry(payload string, ttl time.Duration) *Entry { } func TestLRUStorage_HitAndMiss(t *testing.T) { - s := NewLRUStorage(1<<20, nil) // 1 MB + s := NewLRUStorage(1<<20, nil, metrics.Default) ctx := context.Background() got, err := s.Get(ctx, "missing") @@ -51,7 +51,7 @@ func TestLRUStorage_HitAndMiss(t *testing.T) { } func TestLRUStorage_HardExpiry(t *testing.T) { - s := NewLRUStorage(1<<20, nil) + s := NewLRUStorage(1<<20, nil, metrics.Default) ctx := context.Background() entry := makeEntry("stale", time.Millisecond) @@ -78,7 +78,7 @@ func TestLRUStorage_HardExpiry(t *testing.T) { } func TestLRUStorage_Delete(t *testing.T) { - s := NewLRUStorage(1<<20, nil) + s := NewLRUStorage(1<<20, nil, metrics.Default) ctx := context.Background() if err := s.Set(ctx, "del", makeEntry("x", time.Minute)); err != nil { @@ -97,7 +97,7 @@ func TestLRUStorage_Delete(t *testing.T) { func TestLRUStorage_InPlaceUpdate(t *testing.T) { sample, _ := json.Marshal(makeEntry("v1", time.Minute)) entrySize := int64(len(sample)) + 20 - s := NewLRUStorage(entrySize*shardCount, nil) + s := NewLRUStorage(entrySize*shardCount, nil, metrics.Default) ctx := context.Background() // Overwrite an existing key — Get must return the new payload. @@ -118,7 +118,7 @@ func TestLRUStorage_InPlaceUpdate(t *testing.T) { } func TestLRUStorage_ImmutabilityAfterSet(t *testing.T) { - s := NewLRUStorage(1<<20, nil) + s := NewLRUStorage(1<<20, nil, metrics.Default) ctx := context.Background() entry := makeEntry("original", time.Minute) @@ -145,7 +145,7 @@ func TestLRUStorage_EvictionCallbackDoesNotDeadlock(t *testing.T) { lru = NewLRUStorage(1<<20, func() { // This mirrors the onEvict in NewCacheFilter. _ = lru.lru.Bytes() - }) + }, metrics.Default) ctx := context.Background() // Fill one shard past capacity to force eviction. Each entry is ~100 bytes; @@ -155,7 +155,7 @@ func TestLRUStorage_EvictionCallbackDoesNotDeadlock(t *testing.T) { // Use a tiny budget so the first two writes to the same shard evict. lru = NewLRUStorage(entrySize*int64(shardCount), func() { _ = lru.lru.Bytes() - }) + }, metrics.Default) done := make(chan struct{}) go func() { @@ -179,14 +179,8 @@ func TestLRUStorage_OversizedEntry(t *testing.T) { // With 256 shards and 1 KB total capacity, each shard holds 4 bytes. // A payload larger than 4 bytes exceeds every shard's maxBytes. const totalBytes = 1024 // 1 KB → 4 bytes per shard - s := NewLRUStorage(totalBytes, nil) - - // Inject a testMetrics so we can observe lru_oversized counter increments. - // metrics.Default is restored automatically when the test ends. m := &testMetrics{} - orig := metrics.Default - metrics.Default = m - t.Cleanup(func() { metrics.Default = orig }) + s := NewLRUStorage(totalBytes, nil, m) ctx := context.Background() entry := makeEntry(string(make([]byte, 1000)), time.Minute) // 1000-byte payload ≫ 4-byte shard diff --git a/filters/cache/storage.go b/filters/cache/storage.go index 69b6b60b23..d2ae6e6d61 100644 --- a/filters/cache/storage.go +++ b/filters/cache/storage.go @@ -6,64 +6,64 @@ import ( "time" ) -// Entry holds cached HTTP response and metadata required for freshness +// Entry holds a cached HTTP response and the metadata required for freshness // evaluation, conditional revalidation, and Age header calculation. type Entry struct { - // StatusCode is HTTP status code of cached response. + // StatusCode is the HTTP status code of the cached response. StatusCode int - // Payload is serialised response body. + // Payload is the serialised response body. Payload []byte - // Header contains response headers as stored at cache time. + // Header contains the response headers as stored at cache time. Header http.Header - // CreatedAt is wall-clock time at which this entry was stored. + // CreatedAt is the wall-clock time at which this entry was stored. CreatedAt time.Time - // TTL is freshness lifetime of entry. + // TTL is the freshness lifetime of the entry. TTL time.Duration - // StaleWhileRevalidate is window after TTL expiry during which stale - // response may be served while background revalidation is in flight. + // StaleWhileRevalidate is the window after TTL expiry during which a stale + // response may be served while a background revalidation is in flight. StaleWhileRevalidate time.Duration - // ETag is entity tag from upstream response, used for conditional + // ETag is the entity tag from the upstream response, used for conditional // revalidation (If-None-Match). ETag string - // LastModified is Last-Modified value from upstream response, used + // LastModified is the Last-Modified value from the upstream response, used // for conditional revalidation (If-Modified-Since). LastModified string - // VaryHeaders lists request header names captured from original - // request that were used to derive cache key, matching upstream + // VaryHeaders lists the request header names captured from the original + // request that were used to derive the cache key, matching the upstream // Vary response header. VaryHeaders []string - // CorrectedInitialAge is age correction term (RFC 9111 §4.2.3). - // When zero, setAgeHeader falls back to legacy elapsed-time formula. + // CorrectedInitialAge is the age correction term defined in RFC 9111 §4.2.3. + // When zero, setAgeHeader falls back to the legacy elapsed-time formula. CorrectedInitialAge time.Duration - // ResponseTime is local time at which upstream response was received, - // used together with CorrectedInitialAge for age calculation. + // ResponseTime is the local time at which the upstream response was received, + // used together with CorrectedInitialAge for RFC 9111 §4.2.3 age calculation. ResponseTime time.Time - // StaleIfError extends hard-expiry retention window so entry remains + // StaleIfError extends the hard-expiry retention window so the entry remains // retrievable during upstream error periods (RFC 5861 stale-if-error). StaleIfError time.Duration } -// IsStale reports whether entry is past its TTL but still within -// stale-while-revalidate window relative to now. +// IsStale reports whether the entry is past its TTL but still within +// the stale-while-revalidate window relative to now. func (e *Entry) IsStale(now time.Time) bool { return now.After(e.CreatedAt.Add(e.TTL)) && now.Before(e.CreatedAt.Add(e.TTL+e.StaleWhileRevalidate)) } -// IsUsable reports whether entry is fresh or within stale-while-revalidate window. +// IsUsable reports whether the entry is fresh or within the stale-while-revalidate window. // Entries past TTL+SWR are retained only for stale-if-error and must not be served. func (e *Entry) IsUsable(now time.Time) bool { return now.Before(e.CreatedAt.Add(e.TTL + e.StaleWhileRevalidate)) } -// Storage is backing store abstraction for cached entries. +// Storage is the backing store abstraction for cached entries. // Implementations must be safe for concurrent use. type Storage interface { - // Get returns entry for key, or (nil, nil) if key is not found. + // Get returns the entry for key, or (nil, nil) if the key is not found. Get(ctx context.Context, key string) (*Entry, error) - // Set stores or overwrites entry for key. + // Set stores or overwrites the entry for key. Set(ctx context.Context, key string, entry *Entry) error - // Delete removes entry for key. It is not an error if key does not exist. + // Delete removes the entry for key. It is not an error if the key does not exist. Delete(ctx context.Context, key string) error } diff --git a/filters/cache/valkey_storage_test.go b/filters/cache/valkey_storage_test.go index afb0c7a551..923344510c 100644 --- a/filters/cache/valkey_storage_test.go +++ b/filters/cache/valkey_storage_test.go @@ -17,9 +17,9 @@ import ( // stubValkeyClient is an in-memory valkeyClient stub for unit tests that // should not depend on a running Valkey instance or Docker. type stubValkeyClient struct { - mu sync.Mutex - data map[string]string - broken bool // if true, all operations return an error + mu sync.Mutex + data map[string]string + broken bool // if true, all operations return an error } func newStubValkeyClient() *stubValkeyClient { @@ -92,31 +92,31 @@ func (m *testMetrics) counter(key string) int { } // metrics.Metrics no-op implementations -func (m *testMetrics) MeasureSince(key string, start time.Time) {} -func (m *testMetrics) IncCounterBy(key string, value int64) {} -func (m *testMetrics) IncFloatCounterBy(key string, value float64) {} -func (m *testMetrics) MeasureRouteLookup(start time.Time) {} -func (m *testMetrics) MeasureFilterCreate(filterName string, start time.Time) {} -func (m *testMetrics) MeasureFilterRequest(filterName string, start time.Time) {} -func (m *testMetrics) MeasureAllFiltersRequest(routeId string, start time.Time) {} -func (m *testMetrics) MeasureBackendRequestHeader(host string, size int) {} -func (m *testMetrics) MeasureBackend(routeId string, start time.Time) {} -func (m *testMetrics) MeasureBackendHost(routeBackendHost string, start time.Time) {} -func (m *testMetrics) MeasureFilterResponse(filterName string, start time.Time) {} -func (m *testMetrics) MeasureAllFiltersResponse(routeId string, start time.Time) {} +func (m *testMetrics) MeasureSince(key string, start time.Time) {} +func (m *testMetrics) IncCounterBy(key string, value int64) {} +func (m *testMetrics) IncFloatCounterBy(key string, value float64) {} +func (m *testMetrics) MeasureRouteLookup(start time.Time) {} +func (m *testMetrics) MeasureFilterCreate(filterName string, start time.Time) {} +func (m *testMetrics) MeasureFilterRequest(filterName string, start time.Time) {} +func (m *testMetrics) MeasureAllFiltersRequest(routeId string, start time.Time) {} +func (m *testMetrics) MeasureBackendRequestHeader(host string, size int) {} +func (m *testMetrics) MeasureBackend(routeId string, start time.Time) {} +func (m *testMetrics) MeasureBackendHost(routeBackendHost string, start time.Time) {} +func (m *testMetrics) MeasureFilterResponse(filterName string, start time.Time) {} +func (m *testMetrics) MeasureAllFiltersResponse(routeId string, start time.Time) {} func (m *testMetrics) MeasureResponse(code int, method string, routeId string, start time.Time) {} -func (m *testMetrics) MeasureResponseSize(host string, size int64) {} -func (m *testMetrics) MeasureProxy(requestDuration, responseDuration time.Duration) {} -func (m *testMetrics) MeasureServe(routeId, host, method string, code int, start time.Time) {} -func (m *testMetrics) IncRoutingFailures() {} -func (m *testMetrics) IncErrorsBackend(routeId string) {} -func (m *testMetrics) MeasureBackend5xx(t time.Time) {} -func (m *testMetrics) IncErrorsStreaming(routeId string) {} -func (m *testMetrics) RegisterHandler(path string, handler *http.ServeMux) {} -func (m *testMetrics) UpdateGauge(key string, value float64) {} -func (m *testMetrics) SetInvalidRoute(routeId, reason string) {} -func (m *testMetrics) Close() {} -func (m *testMetrics) String() string { return "testMetrics" } +func (m *testMetrics) MeasureResponseSize(host string, size int64) {} +func (m *testMetrics) MeasureProxy(requestDuration, responseDuration time.Duration) {} +func (m *testMetrics) MeasureServe(routeId, host, method string, code int, start time.Time) {} +func (m *testMetrics) IncRoutingFailures() {} +func (m *testMetrics) IncErrorsBackend(routeId string) {} +func (m *testMetrics) MeasureBackend5xx(t time.Time) {} +func (m *testMetrics) IncErrorsStreaming(routeId string) {} +func (m *testMetrics) RegisterHandler(path string, handler *http.ServeMux) {} +func (m *testMetrics) UpdateGauge(key string, value float64) {} +func (m *testMetrics) SetInvalidRoute(routeId, reason string) {} +func (m *testMetrics) Close() {} +func (m *testMetrics) String() string { return "testMetrics" } func TestValkeyStorage_GetSetDelete(t *testing.T) { addr, done := valkeytest.NewTestValkey(t) @@ -130,7 +130,7 @@ func TestValkeyStorage_GetSetDelete(t *testing.T) { } defer ring.Close() - lru := NewLRUStorage(64<<20, nil) + lru := NewLRUStorage(64<<20, nil, metrics.Default) s := NewValkeyStorage(ring, lru, &testMetrics{}) ctx := context.Background() @@ -182,7 +182,7 @@ func TestValkeyStorage_FallsBackToL1OnValkeyUnavailable(t *testing.T) { } defer ring.Close() - lru := NewLRUStorage(64<<20, nil) + lru := NewLRUStorage(64<<20, nil, metrics.Default) m := &testMetrics{} s := NewValkeyStorage(ring, lru, m) @@ -228,7 +228,7 @@ func TestValkeyStorage_RecordsValkeyMiss(t *testing.T) { // Uses a stub client — no Docker or live Valkey needed. stub := newStubValkeyClient() m := &testMetrics{} - lru := NewLRUStorage(64<<20, nil) + lru := NewLRUStorage(64<<20, nil, metrics.Default) s := &ValkeyStorage{ring: stub, l1: lru, metrics: m} got, err := s.Get(context.Background(), "nonexistent-key") @@ -251,7 +251,7 @@ func TestValkeyStorage_SplitFallbackCounters(t *testing.T) { // Set triggers valkey_set_fallback; Get triggers valkey_get_fallback. stub := newBrokenStubValkeyClient() m := &testMetrics{} - lru := NewLRUStorage(64<<20, nil) + lru := NewLRUStorage(64<<20, nil, metrics.Default) s := &ValkeyStorage{ring: stub, l1: lru, metrics: m} ctx := context.Background() From d0d427cf1c1e76aef30ac4980b22db791a9ac31f Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Tue, 2 Jun 2026 15:06:20 +0200 Subject: [PATCH 14/89] Fix registration of cache filter Signed-off-by: Larry D Almeida --- skipper.go | 22 ++++++++++++++++++++++ 1 file changed, 22 insertions(+) diff --git a/skipper.go b/skipper.go index dded0e1fa1..bca7603596 100644 --- a/skipper.go +++ b/skipper.go @@ -2308,6 +2308,28 @@ func run(o Options, sig chan os.Signal, idleConnsCH chan struct{}) error { } } + if !slices.Contains(o.DisabledFilters, cache.Name) { + cacheSpec := cache.NewCacheFilter( + cache.Options{ + MaxBytes: o.cacheBudget(), + ListenAddr: o.Address, + NetOpts: skpnet.Options{ + IdleConnTimeout: o.CloseIdleConnsPeriod, + MaxIdleConnsPerHost: o.IdleConnectionsPerHost, + Tracer: tracer, + OpentracingComponentTag: "skipper", + OpentracingSpanName: "cache_revalidation", + OpentracingEventsByTag: o.OpenTracingClientTraceByTag, + }, + ValkeyRing: valkeyRing, + L1TTL: o.CacheL1TTL, + }, + ) + defer cacheSpec.(io.Closer).Close() + o.CustomFilters = append(o.CustomFilters, cacheSpec) + } + + if o.TLSMinVersion == 0 { o.TLSMinVersion = tls.VersionTLS12 } From cd792c3f3899676fb8be82c9b502b5e7b8779917 Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Tue, 16 Jun 2026 15:56:31 +0200 Subject: [PATCH 15/89] refactor: add l1TTL field to ValkeyStorage (write-through prep) Signed-off-by: Larry D Almeida --- filters/cache/filter.go | 2 +- filters/cache/valkey_storage.go | 5 +++-- filters/cache/valkey_storage_test.go | 8 ++++---- 3 files changed, 8 insertions(+), 7 deletions(-) diff --git a/filters/cache/filter.go b/filters/cache/filter.go index e20e771350..592548c251 100644 --- a/filters/cache/filter.go +++ b/filters/cache/filter.go @@ -72,7 +72,7 @@ func NewCacheFilter(maxBytes int64, listenAddr string, netOpts skpnet.Options, v var store Storage = lru if valkeyRing != nil { - store = NewValkeyStorage(valkeyRing, lru, m) + store = NewValkeyStorage(valkeyRing, lru, m, 60*time.Second) } return &cacheSpec{ diff --git a/filters/cache/valkey_storage.go b/filters/cache/valkey_storage.go index 4d9f99449f..5579d4fc4e 100644 --- a/filters/cache/valkey_storage.go +++ b/filters/cache/valkey_storage.go @@ -27,6 +27,7 @@ type ValkeyStorage struct { ring valkeyClient l1 *LRUStorage metrics metrics.Metrics + l1TTL time.Duration // max TTL for write-through L1 warming; 0 = write-around } // NewValkeyStorage creates a ValkeyStorage backed by ring (L2) with l1 as the @@ -37,8 +38,8 @@ type ValkeyStorage struct { // - valkey_set_fallback — Valkey error on Set; L1 was written instead // // Pass metrics.Default when no test-scoped metrics collector is needed. -func NewValkeyStorage(ring *skpnet.ValkeyRingClient, l1 *LRUStorage, m metrics.Metrics) *ValkeyStorage { - return &ValkeyStorage{ring: ring, l1: l1, metrics: m} +func NewValkeyStorage(ring *skpnet.ValkeyRingClient, l1 *LRUStorage, m metrics.Metrics, l1TTL time.Duration) *ValkeyStorage { + return &ValkeyStorage{ring: ring, l1: l1, metrics: m, l1TTL: l1TTL} } func (s *ValkeyStorage) Get(ctx context.Context, key string) (*Entry, error) { diff --git a/filters/cache/valkey_storage_test.go b/filters/cache/valkey_storage_test.go index 923344510c..f2581f1ed6 100644 --- a/filters/cache/valkey_storage_test.go +++ b/filters/cache/valkey_storage_test.go @@ -131,7 +131,7 @@ func TestValkeyStorage_GetSetDelete(t *testing.T) { defer ring.Close() lru := NewLRUStorage(64<<20, nil, metrics.Default) - s := NewValkeyStorage(ring, lru, &testMetrics{}) + s := NewValkeyStorage(ring, lru, &testMetrics{}, 0) ctx := context.Background() key := "test-key" @@ -184,7 +184,7 @@ func TestValkeyStorage_FallsBackToL1OnValkeyUnavailable(t *testing.T) { lru := NewLRUStorage(64<<20, nil, metrics.Default) m := &testMetrics{} - s := NewValkeyStorage(ring, lru, m) + s := NewValkeyStorage(ring, lru, m, 0) // Stop valkey before exercising fallback paths. done() @@ -229,7 +229,7 @@ func TestValkeyStorage_RecordsValkeyMiss(t *testing.T) { stub := newStubValkeyClient() m := &testMetrics{} lru := NewLRUStorage(64<<20, nil, metrics.Default) - s := &ValkeyStorage{ring: stub, l1: lru, metrics: m} + s := &ValkeyStorage{ring: stub, l1: lru, metrics: m, l1TTL: 0} got, err := s.Get(context.Background(), "nonexistent-key") if err != nil { @@ -252,7 +252,7 @@ func TestValkeyStorage_SplitFallbackCounters(t *testing.T) { stub := newBrokenStubValkeyClient() m := &testMetrics{} lru := NewLRUStorage(64<<20, nil, metrics.Default) - s := &ValkeyStorage{ring: stub, l1: lru, metrics: m} + s := &ValkeyStorage{ring: stub, l1: lru, metrics: m, l1TTL: 0} ctx := context.Background() entry := &Entry{StatusCode: 200, Payload: []byte("x"), TTL: time.Minute, CreatedAt: time.Now()} From 61f6d54e276b2f9877a2dbec4e89c1a4e11b578d Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Tue, 16 Jun 2026 16:10:59 +0200 Subject: [PATCH 16/89] feat: write-through L1 warming on successful Valkey Set Signed-off-by: Larry D Almeida --- filters/cache/valkey_storage.go | 22 +++++++++++++----- filters/cache/valkey_storage_test.go | 34 ++++++++++++++++++++++++++++ 2 files changed, 50 insertions(+), 6 deletions(-) diff --git a/filters/cache/valkey_storage.go b/filters/cache/valkey_storage.go index 5579d4fc4e..cb56adfd09 100644 --- a/filters/cache/valkey_storage.go +++ b/filters/cache/valkey_storage.go @@ -66,18 +66,28 @@ func (s *ValkeyStorage) Set(ctx context.Context, key string, entry *Entry) error return fmt.Errorf("cache: encode valkey entry: %w", err) } - ttl := entry.TTL + max(entry.StaleIfError, entry.StaleWhileRevalidate) - if ttl <= 0 { - ttl = time.Minute + valkeyTTL := entry.TTL + max(entry.StaleIfError, entry.StaleWhileRevalidate) + if valkeyTTL <= 0 { + valkeyTTL = time.Minute } - if err := s.ring.SetWithExpire(ctx, key, string(data), ttl); err != nil { + if err := s.ring.SetWithExpire(ctx, key, string(data), valkeyTTL); err != nil { s.metrics.IncCounter("valkey_set_fallback") log.WithError(err).Warn("cache: valkey Set failed, falling back to L1") return s.l1.Set(ctx, key, entry) } - // Write-around: L1 is not warmed on a successful Valkey Set. Subsequent - // Valkey hits skip L1 entirely; L1 is only populated on Valkey failures. + + // Write-through: warm L1 with a bounded TTL so pods can serve subsequent + // requests from local memory without a Valkey round-trip. + // Skip warming for non-cacheable entries (TTL <= 0) to avoid polluting L1 + // with entries that should not be served. + if s.l1TTL > 0 && entry.TTL > 0 { + warmTTL := min(s.l1TTL, entry.TTL) + warmed := *entry + warmed.TTL = warmTTL + warmed.CreatedAt = time.Now() + _ = s.l1.Set(ctx, key, &warmed) + } return nil } diff --git a/filters/cache/valkey_storage_test.go b/filters/cache/valkey_storage_test.go index f2581f1ed6..7d0fda0bab 100644 --- a/filters/cache/valkey_storage_test.go +++ b/filters/cache/valkey_storage_test.go @@ -246,6 +246,40 @@ func TestValkeyStorage_RecordsValkeyMiss(t *testing.T) { } } +func TestValkeyStorage_WriteThroughWarmsL1(t *testing.T) { + stub := newStubValkeyClient() + m := &testMetrics{} + lru := NewLRUStorage(64<<20, nil, metrics.Default) + s := &ValkeyStorage{ring: stub, l1: lru, metrics: m, l1TTL: 60 * time.Second} + + ctx := context.Background() + key := "wt-key" + entry := &Entry{ + StatusCode: 200, + Payload: []byte("warm"), + TTL: time.Minute, + CreatedAt: time.Now(), + } + + if err := s.Set(ctx, key, entry); err != nil { + t.Fatalf("Set: %v", err) + } + + // Break Valkey so any Get must come from L1. + stub.broken = true + + got, err := s.Get(ctx, key) + if err != nil { + t.Fatalf("Get with broken Valkey: %v", err) + } + if got == nil { + t.Fatal("expected L1 warm hit, got nil — write-through did not warm L1") + } + if string(got.Payload) != "warm" { + t.Errorf("payload: got %q, want %q", string(got.Payload), "warm") + } +} + func TestValkeyStorage_SplitFallbackCounters(t *testing.T) { // Uses a broken stub — no Docker or live Valkey needed. // Set triggers valkey_set_fallback; Get triggers valkey_get_fallback. From 1aa06ae062c7faa297e6aed985f125b8848e2873 Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Tue, 16 Jun 2026 21:10:28 +0200 Subject: [PATCH 17/89] feat: L1-first reads with l1_hit counter in ValkeyStorage.Get Signed-off-by: Larry D Almeida --- filters/cache/valkey_storage.go | 7 +++ filters/cache/valkey_storage_test.go | 78 ++++++++++++++++++++++++++-- 2 files changed, 82 insertions(+), 3 deletions(-) diff --git a/filters/cache/valkey_storage.go b/filters/cache/valkey_storage.go index cb56adfd09..465dd983a5 100644 --- a/filters/cache/valkey_storage.go +++ b/filters/cache/valkey_storage.go @@ -43,6 +43,13 @@ func NewValkeyStorage(ring *skpnet.ValkeyRingClient, l1 *LRUStorage, m metrics.M } func (s *ValkeyStorage) Get(ctx context.Context, key string) (*Entry, error) { + // L1-first: serve from local memory when the write-through warming populated it. + // LRUStorage.Get returns nil, nil for expired entries — a miss falls through. + if e, err := s.l1.Get(ctx, key); err == nil && e != nil { + s.metrics.IncCounter("l1_hit") + return e, nil + } + data, err := s.ring.Get(ctx, key) if err != nil { if valkey.IsValkeyNil(err) { diff --git a/filters/cache/valkey_storage_test.go b/filters/cache/valkey_storage_test.go index 7d0fda0bab..f278d89da5 100644 --- a/filters/cache/valkey_storage_test.go +++ b/filters/cache/valkey_storage_test.go @@ -278,11 +278,78 @@ func TestValkeyStorage_WriteThroughWarmsL1(t *testing.T) { if string(got.Payload) != "warm" { t.Errorf("payload: got %q, want %q", string(got.Payload), "warm") } + if m.counter("l1_hit") != 1 { + t.Errorf("expected l1_hit=1, got %d", m.counter("l1_hit")) + } +} + +func TestValkeyStorage_L1TTLBoundedToEntryTTL(t *testing.T) { + stub := newStubValkeyClient() + lru := NewLRUStorage(64<<20, nil, metrics.Default) + s := &ValkeyStorage{ring: stub, l1: lru, metrics: &testMetrics{}, l1TTL: 60 * time.Second} + + ctx := context.Background() + key := "bounded-key" + entry := &Entry{ + StatusCode: 200, + Payload: []byte("short"), + TTL: 10 * time.Second, // shorter than l1TTL + CreatedAt: time.Now(), + } + + if err := s.Set(ctx, key, entry); err != nil { + t.Fatalf("Set: %v", err) + } + + // Read directly from L1 to inspect the stored TTL. + l1Entry, err := lru.Get(ctx, key) + if err != nil { + t.Fatalf("L1 Get: %v", err) + } + if l1Entry == nil { + t.Fatal("expected L1 entry after write-through, got nil") + } + if l1Entry.TTL != 10*time.Second { + t.Errorf("L1 TTL: got %v, want %v (should be min(l1TTL, entry.TTL))", l1Entry.TTL, 10*time.Second) + } +} + +func TestValkeyStorage_L1TTL_Zero_DisablesWarming(t *testing.T) { + stub := newStubValkeyClient() + m := &testMetrics{} + lru := NewLRUStorage(64<<20, nil, metrics.Default) + s := &ValkeyStorage{ring: stub, l1: lru, metrics: m, l1TTL: 0} // write-around + + ctx := context.Background() + key := "no-warm-key" + entry := &Entry{ + StatusCode: 200, + Payload: []byte("bypass"), + TTL: time.Minute, + CreatedAt: time.Now(), + } + + if err := s.Set(ctx, key, entry); err != nil { + t.Fatalf("Set: %v", err) + } + + // Break Valkey — if L1 were warmed, Get would still return the entry. + stub.broken = true + + got, err := s.Get(ctx, key) + if err != nil { + t.Fatalf("Get with broken Valkey: %v", err) + } + if got != nil { + t.Error("expected nil (write-around: L1 should not be warmed when l1TTL=0)") + } } func TestValkeyStorage_SplitFallbackCounters(t *testing.T) { // Uses a broken stub — no Docker or live Valkey needed. - // Set triggers valkey_set_fallback; Get triggers valkey_get_fallback. + // Set triggers valkey_set_fallback, which writes the entry to L1. + // Get checks L1 first (L1-first reads) and finds the entry — incrementing l1_hit, + // not valkey_get_fallback. stub := newBrokenStubValkeyClient() m := &testMetrics{} lru := NewLRUStorage(64<<20, nil, metrics.Default) @@ -299,9 +366,14 @@ func TestValkeyStorage_SplitFallbackCounters(t *testing.T) { t.Errorf("expected valkey_get_fallback=0 after Set, got %d", m.counter("valkey_get_fallback")) } + // L1-first: the entry was written to L1 by the Set fallback path, so Get returns + // it from L1 without ever touching (broken) Valkey. _, _ = s.Get(ctx, "k") - if m.counter("valkey_get_fallback") != 1 { - t.Errorf("expected valkey_get_fallback=1, got %d", m.counter("valkey_get_fallback")) + if m.counter("l1_hit") != 1 { + t.Errorf("expected l1_hit=1, got %d", m.counter("l1_hit")) + } + if m.counter("valkey_get_fallback") != 0 { + t.Errorf("expected valkey_get_fallback=0 (L1 served before Valkey check), got %d", m.counter("valkey_get_fallback")) } if m.counter("valkey_set_fallback") != 1 { t.Errorf("valkey_set_fallback should still be 1, got %d", m.counter("valkey_set_fallback")) From a42c47fec326eba6a1b558b285c6b2860792576f Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Tue, 16 Jun 2026 22:11:30 +0200 Subject: [PATCH 18/89] fix: update stale counter assertion after L1-first Get, add l1_hit to godoc Signed-off-by: Larry D Almeida --- filters/cache/valkey_storage.go | 1 + filters/cache/valkey_storage_test.go | 7 +++++-- 2 files changed, 6 insertions(+), 2 deletions(-) diff --git a/filters/cache/valkey_storage.go b/filters/cache/valkey_storage.go index 465dd983a5..1947682f30 100644 --- a/filters/cache/valkey_storage.go +++ b/filters/cache/valkey_storage.go @@ -33,6 +33,7 @@ type ValkeyStorage struct { // NewValkeyStorage creates a ValkeyStorage backed by ring (L2) with l1 as the // fallback in-memory cache. m is used to record per-operation counters: // +// - l1_hit — L1 returned a warm entry; Valkey not consulted // - valkey_miss — clean cache miss (key not found in Valkey) // - valkey_get_fallback — Valkey error on Get; L1 was consulted instead // - valkey_set_fallback — Valkey error on Set; L1 was written instead diff --git a/filters/cache/valkey_storage_test.go b/filters/cache/valkey_storage_test.go index f278d89da5..70a793f810 100644 --- a/filters/cache/valkey_storage_test.go +++ b/filters/cache/valkey_storage_test.go @@ -209,8 +209,11 @@ func TestValkeyStorage_FallsBackToL1OnValkeyUnavailable(t *testing.T) { if got == nil { t.Fatal("expected L1 fallback hit, got nil") } - if m.counter("valkey_get_fallback") == 0 { - t.Error("expected valkey_get_fallback to be incremented on Get fallback path") + if m.counter("l1_hit") == 0 { + t.Error("expected l1_hit to be incremented: Set fallback warmed L1, Get should serve from it") + } + if m.counter("valkey_get_fallback") != 0 { + t.Errorf("expected valkey_get_fallback=0 (L1 served before Valkey was contacted), got %d", m.counter("valkey_get_fallback")) } // Confirm the entry was physically written to L1 — not just returned via some From d1d7f8d9974cf3eebb2aaf85c2aec4dccdab03f3 Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Wed, 17 Jun 2026 09:31:43 +0200 Subject: [PATCH 19/89] feat: wire --cache-l1-ttl flag through Options to ValkeyStorage Signed-off-by: Larry D Almeida --- config/config.go | 3 +++ filters/cache/filter.go | 4 ++-- filters/cache/filter_test.go | 16 ++++++++-------- skipper.go | 5 +++++ 4 files changed, 18 insertions(+), 10 deletions(-) diff --git a/config/config.go b/config/config.go index 073f6db5ad..c0636736c2 100644 --- a/config/config.go +++ b/config/config.go @@ -231,6 +231,7 @@ type Config struct { Oauth2TokeninfoTimeout time.Duration `yaml:"oauth2-tokeninfo-timeout"` Oauth2TokeninfoCacheSize int `yaml:"oauth2-tokeninfo-cache-size"` Oauth2TokeninfoCacheTTL time.Duration `yaml:"oauth2-tokeninfo-cache-ttl"` + CacheL1TTL time.Duration `yaml:"cache-l1-ttl"` Oauth2SecretFile string `yaml:"oauth2-secret-file"` Oauth2ClientID string `yaml:"oauth2-client-id"` Oauth2ClientSecret string `yaml:"oauth2-client-secret"` @@ -616,6 +617,7 @@ func NewConfig() *Config { flag.DurationVar(&cfg.Oauth2TokeninfoTimeout, "oauth2-tokeninfo-timeout", 2*time.Second, "sets the default tokeninfo request timeout duration to 2000ms") flag.IntVar(&cfg.Oauth2TokeninfoCacheSize, "oauth2-tokeninfo-cache-size", 0, "non-zero value enables tokeninfo cache and sets the maximum number of cached tokens") flag.DurationVar(&cfg.Oauth2TokeninfoCacheTTL, "oauth2-tokeninfo-cache-ttl", 0, "non-zero value limits the lifetime of a cached tokeninfo which otherwise equals the tokeninfo 'expires_in' field value") + flag.DurationVar(&cfg.CacheL1TTL, "cache-l1-ttl", 60*time.Second, "maximum TTL for write-through L1 warming in the cache() filter when Valkey is configured; set to 0 to disable (write-around)") flag.DurationVar(&cfg.Oauth2TokenintrospectionTimeout, "oauth2-tokenintrospect-timeout", 2*time.Second, "sets the default tokenintrospection request timeout duration to 2000ms") flag.Var(&cfg.Oauth2AuthURLParameters, "oauth2-auth-url-parameters", "sets additional parameters to send when calling the OAuth2 authorize or token endpoints as key-value pairs") flag.StringVar(&cfg.Oauth2AccessTokenHeaderName, "oauth2-access-token-header-name", "", "sets the access token to a header on the request with this name") @@ -1121,6 +1123,7 @@ func (c *Config) ToOptions() skipper.Options { OAuthTokeninfoTimeout: c.Oauth2TokeninfoTimeout, OAuthTokeninfoCacheSize: c.Oauth2TokeninfoCacheSize, OAuthTokeninfoCacheTTL: c.Oauth2TokeninfoCacheTTL, + CacheL1TTL: c.CacheL1TTL, OAuth2SecretFile: c.Oauth2SecretFile, OAuth2ClientID: c.Oauth2ClientID, OAuth2ClientSecret: c.Oauth2ClientSecret, diff --git a/filters/cache/filter.go b/filters/cache/filter.go index 592548c251..c8bdef7131 100644 --- a/filters/cache/filter.go +++ b/filters/cache/filter.go @@ -64,7 +64,7 @@ const ( // Combining force mode with stale-if-error: // // -> cache("5m", "15s", "30s", "60s") -> "https://example.org" -func NewCacheFilter(maxBytes int64, listenAddr string, netOpts skpnet.Options, valkeyRing *skpnet.ValkeyRingClient) filters.Spec { +func NewCacheFilter(maxBytes int64, listenAddr string, netOpts skpnet.Options, valkeyRing *skpnet.ValkeyRingClient, l1TTL time.Duration) filters.Spec { m := metrics.Default lru := NewLRUStorage(maxBytes, func() { m.IncCounter("lru_eviction") @@ -72,7 +72,7 @@ func NewCacheFilter(maxBytes int64, listenAddr string, netOpts skpnet.Options, v var store Storage = lru if valkeyRing != nil { - store = NewValkeyStorage(valkeyRing, lru, m, 60*time.Second) + store = NewValkeyStorage(valkeyRing, lru, m, l1TTL) } return &cacheSpec{ diff --git a/filters/cache/filter_test.go b/filters/cache/filter_test.go index 5678ad304e..69a28dd567 100644 --- a/filters/cache/filter_test.go +++ b/filters/cache/filter_test.go @@ -20,7 +20,7 @@ import ( func newTestFilter(t *testing.T, ttl, errorTTL, swrWindow time.Duration, extra ...time.Duration) *cacheFilter { t.Helper() - spec := NewCacheFilter(1<<20, "localhost:9090", skpnet.Options{}, nil) + spec := NewCacheFilter(1<<20, "localhost:9090", skpnet.Options{}, nil, 60*time.Second) args := []interface{}{ ttl.String(), errorTTL.String(), @@ -51,7 +51,7 @@ func newTestFilter(t *testing.T, ttl, errorTTL, swrWindow time.Duration, extra . // but are ignored — pure RFC mode has no operator TTL. func newTestFilterRFC(t *testing.T, _, _, _ time.Duration, _ ...time.Duration) *cacheFilter { t.Helper() - spec := NewCacheFilter(1<<20, "localhost:9090", skpnet.Options{}, nil) + spec := NewCacheFilter(1<<20, "localhost:9090", skpnet.Options{}, nil, 60*time.Second) f, err := spec.CreateFilter([]interface{}{}) if err != nil { t.Fatal(err) @@ -131,7 +131,7 @@ func TestCacheFilter_MissAndHit(t *testing.T) { } func TestCacheFilter_KeyIsolationByAuthToken(t *testing.T) { - spec := NewCacheFilter(1<<20, "localhost:9090", skpnet.Options{}, nil) + spec := NewCacheFilter(1<<20, "localhost:9090", skpnet.Options{}, nil, 60*time.Second) fi, err := spec.CreateFilter([]interface{}{"1m", "15s", "1m", "0s", "Authorization"}) if err != nil { t.Fatal(err) @@ -261,7 +261,7 @@ func TestCacheFilter_Response_NoopIfStateBagKeyMissing(t *testing.T) { } func TestCreateFilter_InvalidArgs(t *testing.T) { - spec := NewCacheFilter(1<<20, "localhost:9090", skpnet.Options{}, nil) + spec := NewCacheFilter(1<<20, "localhost:9090", skpnet.Options{}, nil, 60*time.Second) t.Cleanup(spec.(*cacheSpec).client.Close) cases := []struct { name string @@ -1316,7 +1316,7 @@ func TestCacheFilter_MustRevalidate_ForcesCoalesceWhenStale(t *testing.T) { } func TestCacheFilter_SharedStorage_RouteIsolation(t *testing.T) { - spec := NewCacheFilter(1<<20, "localhost:9090", skpnet.Options{}, nil) + spec := NewCacheFilter(1<<20, "localhost:9090", skpnet.Options{}, nil, 60*time.Second) t.Cleanup(spec.(*cacheSpec).client.Close) makeFilter := func(t *testing.T) *cacheFilter { @@ -2613,7 +2613,7 @@ func TestCacheFilter_SMaxAge_CapsRouteTTL(t *testing.T) { } func TestCacheFilter_CreateFilter_RFCArgParsing(t *testing.T) { - spec := NewCacheFilter(1<<20, ":9090", skpnet.Options{}, nil) + spec := NewCacheFilter(1<<20, ":9090", skpnet.Options{}, nil, 60*time.Second) t.Cleanup(spec.(*cacheSpec).client.Close) cases := []struct { @@ -2658,7 +2658,7 @@ func TestCacheFilter_CreateFilter_RFCArgParsing(t *testing.T) { func TestCacheFilter_PureRFCMode_ZeroArgs_UsesUpstreamMaxAge(t *testing.T) { // cache() with no args: pure RFC mode, upstream max-age is fully authoritative, // no operator ceiling. TTL should equal upstream max-age exactly. - spec := NewCacheFilter(1<<20, ":9090", skpnet.Options{}, nil) + spec := NewCacheFilter(1<<20, ":9090", skpnet.Options{}, nil, 60*time.Second) t.Cleanup(spec.(*cacheSpec).client.Close) f, err := spec.CreateFilter([]interface{}{}) if err != nil { @@ -2729,7 +2729,7 @@ func TestCacheFilter_LRUBytesGaugeUpdatesWithoutEviction(t *testing.T) { func TestCacheFilter_PureRFCMode_ZeroArgs_NoUpstreamDirective_NotCached(t *testing.T) { // cache() with no args: when upstream sends no Cache-Control, no Expires, // and no Last-Modified, nothing should be cached (no heuristic without Last-Modified). - spec := NewCacheFilter(1<<20, ":9090", skpnet.Options{}, nil) + spec := NewCacheFilter(1<<20, ":9090", skpnet.Options{}, nil, 60*time.Second) t.Cleanup(spec.(*cacheSpec).client.Close) f, err := spec.CreateFilter([]interface{}{}) if err != nil { diff --git a/skipper.go b/skipper.go index bca7603596..4b6c418ae1 100644 --- a/skipper.go +++ b/skipper.go @@ -148,6 +148,11 @@ type Options struct { // using a fixed 25% fraction. Set explicitly to override that behaviour. ResponseCacheMaxMemoryBytes int64 + // CacheL1TTL sets the maximum TTL for write-through L1 warming in ValkeyStorage. + // On a successful Valkey Set, L1 is warmed with min(CacheL1TTL, entry.TTL). + // Set to 0 to disable write-through (write-around behaviour). Default: 60s. + CacheL1TTL time.Duration + // ReadMemoryLimit, when set, is called by the cache() filter initialiser // to determine the container memory limit. Defaults to reading cgroup files. // Override in tests or on non-standard platforms. From 6e240e725caeffd25a86626fd7745ff5fca24adb Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Wed, 17 Jun 2026 13:47:40 +0200 Subject: [PATCH 20/89] refactor: move --cache-l1-ttl config field and flag to Valkey section Signed-off-by: Larry D Almeida --- config/config.go | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/config/config.go b/config/config.go index c0636736c2..a95ad5d7cd 100644 --- a/config/config.go +++ b/config/config.go @@ -231,7 +231,6 @@ type Config struct { Oauth2TokeninfoTimeout time.Duration `yaml:"oauth2-tokeninfo-timeout"` Oauth2TokeninfoCacheSize int `yaml:"oauth2-tokeninfo-cache-size"` Oauth2TokeninfoCacheTTL time.Duration `yaml:"oauth2-tokeninfo-cache-ttl"` - CacheL1TTL time.Duration `yaml:"cache-l1-ttl"` Oauth2SecretFile string `yaml:"oauth2-secret-file"` Oauth2ClientID string `yaml:"oauth2-client-id"` Oauth2ClientSecret string `yaml:"oauth2-client-secret"` @@ -344,6 +343,8 @@ type Config struct { SwarmValkeyDialTimeout time.Duration `yaml:"swarm-valkey-dial-timeout"` SwarmValkeyKeepAlive time.Duration `yaml:"swarm-valkey-keepalive"` SwarmValkeyUpdateInterval time.Duration `yaml:"swarm-valkey-update-interval"` + // CacheL1TTL is the maximum TTL for write-through L1 warming (cache() filter). + CacheL1TTL time.Duration `yaml:"cache-l1-ttl"` // swim based SwarmKubernetesNamespace string `yaml:"swarm-namespace"` SwarmKubernetesLabelSelectorKey string `yaml:"swarm-label-selector-key"` @@ -617,7 +618,6 @@ func NewConfig() *Config { flag.DurationVar(&cfg.Oauth2TokeninfoTimeout, "oauth2-tokeninfo-timeout", 2*time.Second, "sets the default tokeninfo request timeout duration to 2000ms") flag.IntVar(&cfg.Oauth2TokeninfoCacheSize, "oauth2-tokeninfo-cache-size", 0, "non-zero value enables tokeninfo cache and sets the maximum number of cached tokens") flag.DurationVar(&cfg.Oauth2TokeninfoCacheTTL, "oauth2-tokeninfo-cache-ttl", 0, "non-zero value limits the lifetime of a cached tokeninfo which otherwise equals the tokeninfo 'expires_in' field value") - flag.DurationVar(&cfg.CacheL1TTL, "cache-l1-ttl", 60*time.Second, "maximum TTL for write-through L1 warming in the cache() filter when Valkey is configured; set to 0 to disable (write-around)") flag.DurationVar(&cfg.Oauth2TokenintrospectionTimeout, "oauth2-tokenintrospect-timeout", 2*time.Second, "sets the default tokenintrospection request timeout duration to 2000ms") flag.Var(&cfg.Oauth2AuthURLParameters, "oauth2-auth-url-parameters", "sets additional parameters to send when calling the OAuth2 authorize or token endpoints as key-value pairs") flag.StringVar(&cfg.Oauth2AccessTokenHeaderName, "oauth2-access-token-header-name", "", "sets the access token to a header on the request with this name") @@ -737,6 +737,7 @@ func NewConfig() *Config { flag.DurationVar(&cfg.SwarmValkeyDialTimeout, "swarm-valkey-dial-timeout", net.DefaultDialTimeout, "set valkey client dial timeout") flag.DurationVar(&cfg.SwarmValkeyKeepAlive, "swarm-valkey-keepalive", net.DefaultKeepAlive, "set valkey keepalive probes interval") flag.DurationVar(&cfg.SwarmValkeyUpdateInterval, "swarm-valkey-update-interval", net.DefaultUpdateInterval, "set update interval to update valkey addresses") + flag.DurationVar(&cfg.CacheL1TTL, "cache-l1-ttl", 60*time.Second, "maximum TTL for write-through L1 warming in the cache() filter when Valkey is configured; set to 0 to disable (write-around)") // swim flag.StringVar(&cfg.SwarmKubernetesNamespace, "swarm-namespace", swarm.DefaultNamespace, "Kubernetes namespace to find swarm peer instances") flag.StringVar(&cfg.SwarmKubernetesLabelSelectorKey, "swarm-label-selector-key", swarm.DefaultLabelSelectorKey, "Kubernetes labelselector key to find swarm peer instances") From e13b45aa7ff786937f522e86fe692aea4eb6f782 Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Wed, 17 Jun 2026 20:22:22 +0200 Subject: [PATCH 21/89] style: use strings.SplitSeq in stripHopByHop and parseVaryNames Signed-off-by: Larry D Almeida --- filters/cache/filter.go | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/filters/cache/filter.go b/filters/cache/filter.go index c8bdef7131..840d85396f 100644 --- a/filters/cache/filter.go +++ b/filters/cache/filter.go @@ -1063,9 +1063,8 @@ func parseVaryNames(varyHeader string) []string { if varyHeader == "" { return nil } - parts := strings.Split(varyHeader, ",") - names := make([]string, 0, len(parts)) - for _, p := range parts { + var names []string + for p := range strings.SplitSeq(varyHeader, ",") { if name := strings.TrimSpace(p); name != "" { names = append(names, http.CanonicalHeaderKey(name)) } From 12daa1bf13f8f957c996437400e7c13663324c44 Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Thu, 18 Jun 2026 07:38:17 +0200 Subject: [PATCH 22/89] fix: add CacheL1TTL to defaultConfig and make Close() synchronous config_test.go: defaultConfig() was missing CacheL1TTL, so the expected config had 0s while ParseArgs produced the flag default of 1m0s. filters/cache/filter.go: Close() only closed the revalJobs and lruBytesDone channels without waiting for the background goroutines to finish. Under -race this caused a panic (close of closed channel) when a test's in-flight doRevalidate called f.fetch after the test's local channels had been torn down. Adding bgWg tracks both goroutines and Close() now blocks until both have exited. Signed-off-by: larry-dalmeida Signed-off-by: Larry D Almeida --- config/config_test.go | 1 + filters/cache/filter.go | 10 ++++++++-- 2 files changed, 9 insertions(+), 2 deletions(-) diff --git a/config/config_test.go b/config/config_test.go index e4ad563150..923bad9986 100644 --- a/config/config_test.go +++ b/config/config_test.go @@ -169,6 +169,7 @@ func defaultConfig(with func(*Config)) *Config { SwarmValkeyDialTimeout: 25 * time.Millisecond, SwarmValkeyKeepAlive: time.Second, SwarmValkeyUpdateInterval: 10 * time.Second, + CacheL1TTL: 60 * time.Second, SwarmKubernetesNamespace: "kube-system", SwarmKubernetesLabelSelectorKey: "application", SwarmKubernetesLabelSelectorValue: "skipper-ingress", diff --git a/filters/cache/filter.go b/filters/cache/filter.go index 840d85396f..c6f6dd20cc 100644 --- a/filters/cache/filter.go +++ b/filters/cache/filter.go @@ -11,6 +11,7 @@ import ( "net/url" "strconv" "strings" + "sync" "time" opentracing "github.com/opentracing/opentracing-go" @@ -161,21 +162,24 @@ func (s *cacheSpec) CreateFilter(args []interface{}) (filters.Filter, error) { } cf.fetch = s.client.Do + cf.bgWg.Add(2) go cf.revalidationWorker() go cf.lruBytesScraper() return cf, nil } -// Close shuts down the background revalidation worker and the lru_bytes scraper. -// Must be called when the filter is no longer in use (e.g. in tests via t.Cleanup). +// Close shuts down the background revalidation worker and the lru_bytes scraper, +// blocking until both goroutines have exited. func (f *cacheFilter) Close() { close(f.revalJobs) close(f.lruBytesDone) + f.bgWg.Wait() } // revalidationWorker is the single background goroutine per filter that // processes revalidation jobs sequentially. It exits when revalJobs is closed. func (f *cacheFilter) revalidationWorker() { + defer f.bgWg.Done() for job := range f.revalJobs { f.doRevalidate(job.key, job.req) } @@ -188,6 +192,7 @@ const lruBytesScrapeInterval = 10 * time.Second // even when no evictions occur (large Sets without exceeding capacity never // trigger the onEvict callback). It exits when lruBytesDone is closed. func (f *cacheFilter) lruBytesScraper() { + defer f.bgWg.Done() if f.lruStorage == nil { return } @@ -225,6 +230,7 @@ type cacheFilter struct { revalSF singleflight.Group // coalesces concurrent background revalidations per key revalJobs chan revalJob // background revalidation queue; worker drains this lruBytesDone chan struct{} // closed by Close() to stop the lruBytesScraper goroutine + bgWg sync.WaitGroup // tracks background goroutines; Wait()ed in Close() fetch func(*http.Request) (*http.Response, error) metrics metrics.Metrics } From 4eb9e2751e5ea07959a898851a67de41c2cfd55b Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Thu, 18 Jun 2026 09:09:11 +0200 Subject: [PATCH 23/89] style: group cache config fields into dedicated //cache section in config.go Signed-off-by: larry-dalmeida Signed-off-by: Larry D Almeida --- config/config.go | 13 +++++++++---- 1 file changed, 9 insertions(+), 4 deletions(-) diff --git a/config/config.go b/config/config.go index a95ad5d7cd..5e1baab8a9 100644 --- a/config/config.go +++ b/config/config.go @@ -343,8 +343,6 @@ type Config struct { SwarmValkeyDialTimeout time.Duration `yaml:"swarm-valkey-dial-timeout"` SwarmValkeyKeepAlive time.Duration `yaml:"swarm-valkey-keepalive"` SwarmValkeyUpdateInterval time.Duration `yaml:"swarm-valkey-update-interval"` - // CacheL1TTL is the maximum TTL for write-through L1 warming (cache() filter). - CacheL1TTL time.Duration `yaml:"cache-l1-ttl"` // swim based SwarmKubernetesNamespace string `yaml:"swarm-namespace"` SwarmKubernetesLabelSelectorKey string `yaml:"swarm-label-selector-key"` @@ -355,6 +353,9 @@ type Config struct { SwarmStaticSelf string `yaml:"swarm-static-self"` SwarmStaticOther string `yaml:"swarm-static-other"` + // cache + CacheL1TTL time.Duration `yaml:"cache-l1-ttl"` + ClusterRatelimitMaxGroupShards int `yaml:"cluster-ratelimit-max-group-shards"` EnableLua bool `yaml:"enable-lua"` @@ -737,7 +738,6 @@ func NewConfig() *Config { flag.DurationVar(&cfg.SwarmValkeyDialTimeout, "swarm-valkey-dial-timeout", net.DefaultDialTimeout, "set valkey client dial timeout") flag.DurationVar(&cfg.SwarmValkeyKeepAlive, "swarm-valkey-keepalive", net.DefaultKeepAlive, "set valkey keepalive probes interval") flag.DurationVar(&cfg.SwarmValkeyUpdateInterval, "swarm-valkey-update-interval", net.DefaultUpdateInterval, "set update interval to update valkey addresses") - flag.DurationVar(&cfg.CacheL1TTL, "cache-l1-ttl", 60*time.Second, "maximum TTL for write-through L1 warming in the cache() filter when Valkey is configured; set to 0 to disable (write-around)") // swim flag.StringVar(&cfg.SwarmKubernetesNamespace, "swarm-namespace", swarm.DefaultNamespace, "Kubernetes namespace to find swarm peer instances") flag.StringVar(&cfg.SwarmKubernetesLabelSelectorKey, "swarm-label-selector-key", swarm.DefaultLabelSelectorKey, "Kubernetes labelselector key to find swarm peer instances") @@ -748,6 +748,9 @@ func NewConfig() *Config { flag.StringVar(&cfg.SwarmStaticSelf, "swarm-static-self", "", "set static swarm self node, for example 127.0.0.1:9001") flag.StringVar(&cfg.SwarmStaticOther, "swarm-static-other", "", "set static swarm all nodes, for example 127.0.0.1:9002,127.0.0.1:9003") + // cache + flag.DurationVar(&cfg.CacheL1TTL, "cache-l1-ttl", 60*time.Second, "maximum TTL for write-through L1 warming in the cache() filter when Valkey is configured; set to 0 to disable (write-around)") + flag.IntVar(&cfg.ClusterRatelimitMaxGroupShards, "cluster-ratelimit-max-group-shards", 1, "sets the maximum number of group shards for the clusterRatelimit filter") flag.BoolVar(&cfg.EnableLua, "enable-lua", false, "enable the Lua scripting engine to be able to use the lua() filter") @@ -1124,7 +1127,6 @@ func (c *Config) ToOptions() skipper.Options { OAuthTokeninfoTimeout: c.Oauth2TokeninfoTimeout, OAuthTokeninfoCacheSize: c.Oauth2TokeninfoCacheSize, OAuthTokeninfoCacheTTL: c.Oauth2TokeninfoCacheTTL, - CacheL1TTL: c.CacheL1TTL, OAuth2SecretFile: c.Oauth2SecretFile, OAuth2ClientID: c.Oauth2ClientID, OAuth2ClientSecret: c.Oauth2ClientSecret, @@ -1216,6 +1218,9 @@ func (c *Config) ToOptions() skipper.Options { SwarmStaticSelf: c.SwarmStaticSelf, SwarmStaticOther: c.SwarmStaticOther, + // cache + CacheL1TTL: c.CacheL1TTL, + ClusterRatelimitMaxGroupShards: c.ClusterRatelimitMaxGroupShards, EnableLua: c.EnableLua, From ecbb64f0de61e171bf46c1aeded09c5c23bdd343 Mon Sep 17 00:00:00 2001 From: larry-dalmeida Date: Thu, 18 Jun 2026 10:19:35 +0200 Subject: [PATCH 24/89] style: fix gofmt alignment of bgWg struct field comment Signed-off-by: larry-dalmeida Signed-off-by: Larry D Almeida --- filters/cache/filter.go | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/filters/cache/filter.go b/filters/cache/filter.go index c6f6dd20cc..6e67d3f1e2 100644 --- a/filters/cache/filter.go +++ b/filters/cache/filter.go @@ -230,7 +230,7 @@ type cacheFilter struct { revalSF singleflight.Group // coalesces concurrent background revalidations per key revalJobs chan revalJob // background revalidation queue; worker drains this lruBytesDone chan struct{} // closed by Close() to stop the lruBytesScraper goroutine - bgWg sync.WaitGroup // tracks background goroutines; Wait()ed in Close() + bgWg sync.WaitGroup // tracks background goroutines; Wait()ed in Close() fetch func(*http.Request) (*http.Response, error) metrics metrics.Metrics } From d1046bdf27797bb8a5174a72dbdead5ce31da02c Mon Sep 17 00:00:00 2001 From: larry-dalmeida Date: Fri, 19 Jun 2026 07:50:05 +0200 Subject: [PATCH 25/89] =?UTF-8?q?fix:=20address=20code=20review=20findings?= =?UTF-8?q?=20=E2=80=94=20vary=20sentinel=20invalidation,=20stub=20accurac?= =?UTF-8?q?y,=20dead=20code?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Signed-off-by: larry-dalmeida Signed-off-by: Larry D Almeida --- filters/cache/filter.go | 6 ++--- filters/cache/valkey_storage_test.go | 33 ++++++++++++++++++++++++++-- net/valkey.go | 3 +++ 3 files changed, 37 insertions(+), 5 deletions(-) diff --git a/filters/cache/filter.go b/filters/cache/filter.go index 6e67d3f1e2..9e71ccc965 100644 --- a/filters/cache/filter.go +++ b/filters/cache/filter.go @@ -193,9 +193,6 @@ const lruBytesScrapeInterval = 10 * time.Second // trigger the onEvict callback). It exits when lruBytesDone is closed. func (f *cacheFilter) lruBytesScraper() { defer f.bgWg.Done() - if f.lruStorage == nil { - return - } ticker := time.NewTicker(lruBytesScrapeInterval) defer ticker.Stop() for { @@ -556,6 +553,9 @@ func (f *cacheFilter) Response(ctx filters.FilterContext) { if err := f.storage.Delete(ctx.Request().Context(), key); err != nil { log.WithError(err).Warn("cache: Delete failed (unsafe method invalidation)") } + if err := f.storage.Delete(ctx.Request().Context(), "vary:"+key); err != nil { + log.WithError(err).Warn("cache: Delete failed (vary sentinel invalidation)") + } for _, hdrName := range []string{"Location", "Content-Location"} { if loc := rsp.Header.Get(hdrName); loc != "" && sameOrigin(ctx.Request(), loc) { if locKey := cacheKeyForURL(ctx.RouteId(), ctx.Request(), loc, f.keyHeaders); locKey != "" { diff --git a/filters/cache/valkey_storage_test.go b/filters/cache/valkey_storage_test.go index 70a793f810..96681a4766 100644 --- a/filters/cache/valkey_storage_test.go +++ b/filters/cache/valkey_storage_test.go @@ -53,7 +53,7 @@ func (s *stubValkeyClient) SetWithExpire(_ context.Context, key, value string, _ return nil } -func (s *stubValkeyClient) Expire(_ context.Context, key string, _ time.Duration) (int64, error) { +func (s *stubValkeyClient) Expire(_ context.Context, key string, d time.Duration) (int64, error) { s.mu.Lock() defer s.mu.Unlock() if s.broken { @@ -63,7 +63,11 @@ func (s *stubValkeyClient) Expire(_ context.Context, key string, _ time.Duration if !ok { return 0, nil } - delete(s.data, key) + if d < 0 { + // negative duration → immediate deletion (mirrors Valkey EXPIRE key -1 semantics) + delete(s.data, key) + } + // non-negative duration → TTL update; not modelled in stub (no expiry tracking) return 1, nil } @@ -382,3 +386,28 @@ func TestValkeyStorage_SplitFallbackCounters(t *testing.T) { t.Errorf("valkey_set_fallback should still be 1, got %d", m.counter("valkey_set_fallback")) } } + +func TestValkeyStorage_DeleteCleansL1EvenOnValkeyError(t *testing.T) { + // Valkey is broken, so Set falls back to L1. Delete must still clean L1 + // regardless of the Expire error from Valkey. + stub := newBrokenStubValkeyClient() + lru := NewLRUStorage(64<<20, nil, metrics.Default) + s := &ValkeyStorage{ring: stub, l1: lru, metrics: &testMetrics{}, l1TTL: 0} + + ctx := context.Background() + entry := &Entry{StatusCode: 200, Payload: []byte("body"), TTL: time.Minute, CreatedAt: time.Now()} + + _ = s.Set(ctx, "k", entry) // falls back to L1 (Valkey broken) + + got, err := lru.Get(ctx, "k") + if err != nil || got == nil { + t.Fatal("expected entry in L1 after Set fallback") + } + + _ = s.Delete(ctx, "k") // Valkey Expire will error; L1 must still be cleaned + + got, _ = lru.Get(ctx, "k") + if got != nil { + t.Error("expected L1 to be empty after Delete, but entry still present") + } +} diff --git a/net/valkey.go b/net/valkey.go index e370bb0fa9..c7d26d213c 100644 --- a/net/valkey.go +++ b/net/valkey.go @@ -553,6 +553,9 @@ func (vrc *ValkeyRingClient) SetWithExpire(ctx context.Context, key string, valu return err } } + if len(results) == 0 { + return fmt.Errorf("failed to SetWithExpire, no result") + } return nil } From a0db0e10d9a12564b3f38f5227471aa4a6213314 Mon Sep 17 00:00:00 2001 From: larry-dalmeida Date: Fri, 19 Jun 2026 14:38:45 +0200 Subject: [PATCH 26/89] docs: document Valkey L2 storage, write-through L1, and cache metrics Signed-off-by: larry-dalmeida Signed-off-by: Larry D Almeida --- docs/reference/filters.md | 36 ++++++++++++++++++++++++++++-------- 1 file changed, 28 insertions(+), 8 deletions(-) diff --git a/docs/reference/filters.md b/docs/reference/filters.md index 0b50c0e696..e329376403 100644 --- a/docs/reference/filters.md +++ b/docs/reference/filters.md @@ -4047,17 +4047,37 @@ matching the same key. It has no awareness of other filters in the chain. * **`Cache-Control: private` is ignored in force mode.** Audit the upstream response before enabling force mode on any authenticated route. +**Storage** + +By default entries are stored in an in-process LRU (L1) local to each pod. +When `--swarm-valkey-urls` is configured, Valkey becomes the primary shared +store (L2) accessible by all pods via a client-side consistent hash ring. Every +read checks L1 first; an L1 hit returns without contacting Valkey. + +On every successful Valkey write the entry is also written to L1 +(write-through) with a TTL of `min(--cache-l1-ttl, entry.TTL)`. The default is +60 seconds, bounding how long a pod serves a locally-cached entry before +falling back to Valkey. Set `--cache-l1-ttl=0` to disable L1 warming and +restore write-around behaviour (L1 used only when Valkey is unavailable). + +Explicit deletes (unsafe methods or operator-initiated invalidation) always +remove the L1 entry unconditionally, regardless of `--cache-l1-ttl`. + !!! note - The LRU store is shared across all `cache()` filter instances. The storage - budget is divided evenly across 256 internal shards; a single entry larger - than one shard's budget is dropped with a warning log. + The in-process LRU (L1) is shared across all `cache()` filter instances on + the same pod but is local to that pod. When Valkey is configured it acts as + the cross-pod shared store (L2). The L1 storage budget is divided evenly + across 256 internal shards; a single entry larger than one shard's budget is + dropped with a warning log. !!! note - Three metrics track LRU behaviour: `lru_eviction` (counter, incremented each - time an entry is evicted due to memory pressure), `lru_bytes` (gauge, - updated on every eviction to reflect current storage usage in bytes), and - `lru_oversized` (counter, incremented when an entry is too large to fit in - any shard and is silently dropped). + Metrics: `lru_eviction` (counter, incremented each time an L1 entry is + evicted due to memory pressure), `lru_bytes` (gauge, current L1 usage in + bytes), `lru_oversized` (counter, entry too large for any shard and + silently dropped). When Valkey is configured: `l1_hit` (counter, L1 hit + that bypassed Valkey), `valkey_miss` (counter, Valkey miss that proceeded + to an upstream fetch), `valkey_get_fallback` and `valkey_set_fallback` + (counters, reads/writes that fell back to L1 due to a Valkey error). !!! note `s-maxage` implies `proxy-revalidate` per [RFC 9111 §5.2.2.10](https://www.rfc-editor.org/rfc/rfc9111#section-5.2.2.10): stale entries From 25900af14c96b761d53af218de69b66e64513b90 Mon Sep 17 00:00:00 2001 From: larry-dalmeida Date: Mon, 29 Jun 2026 08:03:46 +0200 Subject: [PATCH 27/89] =?UTF-8?q?Address=20PR=20#4033=20code=20review=20co?= =?UTF-8?q?mments=20=E2=80=94=20move=20goroutines=20to=20spec=20level,=20i?= =?UTF-8?q?ntroduce=20Options=20struct?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit This commit addresses all open code review comments from PR #4033: 1. **Goroutines on spec level (szuecs)**: Move revalidationWorker and lruBytesScraper from per-instance (cacheFilter) to per-spec (cacheSpec) ownership. This prevents 20k+ goroutines for 10k routes and enables proper cleanup on route reloads. Each filter instance enqueues revalidation jobs with a doRevalFn closure that captures per-instance logic. 2. **Introduce cache.Options struct (szuecs)**: Replace 5 positional parameters with a single Options struct containing MaxBytes, ListenAddr, NetOpts, ValkeyRing, L1TTL, and optional Metrics. This enables future extension without signature changes. 3. **Inject metrics (szuecs)**: Make metrics.Metrics injectable via Options (defaults to metrics.Default if nil), removing hidden global dependency and enabling test-scoped assertions without side effects. 4. **Verify remaining comments**: - SetWithExpire error handling (net/valkey.go) already correct - Docs terminology fixed: pod→Skipper, Storage→Cache, dropped cross-pod - Single shared valkeyRing already in place (refactoring merged) Signed-off-by: larry-dalmeida Signed-off-by: Larry D Almeida --- docs/operation/operation.md | 34 +++++++ docs/reference/filters.md | 32 ------- filters/cache/filter.go | 171 ++++++++++++++++++++++------------- filters/cache/filter_test.go | 41 ++++++--- skipper.go | 27 ------ 5 files changed, 169 insertions(+), 136 deletions(-) diff --git a/docs/operation/operation.md b/docs/operation/operation.md index d0d6673e7a..8d48e1f081 100644 --- a/docs/operation/operation.md +++ b/docs/operation/operation.md @@ -1922,3 +1922,37 @@ will be changed to ``` r: SourceFromLast("9.0.0.0/24","2001:67c:20a0::/48") -> ...` ``` + +## Cache + +By default entries are stored in an in-process LRU (L1) local to each Skipper process. +When `--swarm-valkey-urls` is configured, Valkey becomes the primary shared +store (L2) accessible by all Skipper instances via a client-side consistent hash ring. Every +read checks L1 first; an L1 hit returns without contacting Valkey. + +On every successful Valkey write the entry is also written to L1 +(write-through) with a TTL of `min(--cache-l1-ttl, entry.TTL)`. The default is +60 seconds, bounding how long Skipper serves a locally-cached entry before +falling back to Valkey. Set `--cache-l1-ttl=0` to disable L1 warming and +restore write-around behaviour (L1 used only when Valkey is unavailable). + +Explicit deletes (unsafe methods or operator-initiated invalidation) always +remove the L1 entry unconditionally, regardless of `--cache-l1-ttl`. + +!!! note + The in-process LRU (L1) is shared across all `cache()` filter instances in the same process. + When Valkey is configured it acts as a shared store (L2). The L1 storage budget is divided evenly + across 256 internal shards; a single entry larger than one shard's budget is + dropped with a warning log. + +### Metrics + +- `lru_eviction`: Counter, incremented each time an L1 entry is evicted due to memory pressure +- `lru_bytes`: Gauge, current L1 usage in bytes +- `lru_oversized`: Counter, incremented when an entry is too large for any shard and silently dropped + +When Valkey is configured: + +- `l1_hit`: Counter, L1 hits that bypassed Valkey +- `valkey_miss`: Counter, Valkey misses that proceeded to an upstream fetch +- `valkey_get_fallback`, `valkey_set_fallback`: Counters, reads/writes that fell back to L1 due to Valkey errors diff --git a/docs/reference/filters.md b/docs/reference/filters.md index e329376403..a6de699ccd 100644 --- a/docs/reference/filters.md +++ b/docs/reference/filters.md @@ -4047,38 +4047,6 @@ matching the same key. It has no awareness of other filters in the chain. * **`Cache-Control: private` is ignored in force mode.** Audit the upstream response before enabling force mode on any authenticated route. -**Storage** - -By default entries are stored in an in-process LRU (L1) local to each pod. -When `--swarm-valkey-urls` is configured, Valkey becomes the primary shared -store (L2) accessible by all pods via a client-side consistent hash ring. Every -read checks L1 first; an L1 hit returns without contacting Valkey. - -On every successful Valkey write the entry is also written to L1 -(write-through) with a TTL of `min(--cache-l1-ttl, entry.TTL)`. The default is -60 seconds, bounding how long a pod serves a locally-cached entry before -falling back to Valkey. Set `--cache-l1-ttl=0` to disable L1 warming and -restore write-around behaviour (L1 used only when Valkey is unavailable). - -Explicit deletes (unsafe methods or operator-initiated invalidation) always -remove the L1 entry unconditionally, regardless of `--cache-l1-ttl`. - -!!! note - The in-process LRU (L1) is shared across all `cache()` filter instances on - the same pod but is local to that pod. When Valkey is configured it acts as - the cross-pod shared store (L2). The L1 storage budget is divided evenly - across 256 internal shards; a single entry larger than one shard's budget is - dropped with a warning log. - -!!! note - Metrics: `lru_eviction` (counter, incremented each time an L1 entry is - evicted due to memory pressure), `lru_bytes` (gauge, current L1 usage in - bytes), `lru_oversized` (counter, entry too large for any shard and - silently dropped). When Valkey is configured: `l1_hit` (counter, L1 hit - that bypassed Valkey), `valkey_miss` (counter, Valkey miss that proceeded - to an upstream fetch), `valkey_get_fallback` and `valkey_set_fallback` - (counters, reads/writes that fell back to L1 due to a Valkey error). - !!! note `s-maxage` implies `proxy-revalidate` per [RFC 9111 §5.2.2.10](https://www.rfc-editor.org/rfc/rfc9111#section-5.2.2.10): stale entries stored under `s-maxage` are never served without revalidation, regardless of diff --git a/filters/cache/filter.go b/filters/cache/filter.go index 9e71ccc965..2087a9c812 100644 --- a/filters/cache/filter.go +++ b/filters/cache/filter.go @@ -42,17 +42,23 @@ const ( cacheStatusMiss = "MISS" cacheStatusStale = "STALE" - // revalQueueSize is the capacity of the per-filter revalidation job queue. + // revalQueueSize is the capacity of the shared revalidation job queue. // Sized to absorb short bursts; jobs are dropped (with reval_dropped metric) // if the worker cannot keep up. revalQueueSize = 256 ) -// NewCacheFilter returns a Spec for the cache() filter. maxBytes is the -// in-memory storage budget for the shared LRU cache backing all filter -// instances created from this Spec. -// listenAddr is Skipper's own listener address (e.g. ":9090"); revalidation -// requests are sent back through Skipper so the full filter chain runs. +// Options configures the cache filter. All fields are required unless stated otherwise. +type Options struct { + MaxBytes int64 // in-memory storage budget for the LRU + ListenAddr string // Skipper's own listener address (e.g., ":9090") + NetOpts skpnet.Options // network options for revalidation requests + ValkeyRing *skpnet.ValkeyRingClient // optional L2 cache backend; nil = LRU only + L1TTL time.Duration // max TTL to use when warming L1 from Valkey writes + Metrics metrics.Metrics // optional; defaults to metrics.Default if nil +} + +// NewCacheFilter returns a Spec for the cache() filter. // // Route usage (RFC mode — upstream Cache-Control is fully authoritative): // @@ -65,38 +71,67 @@ const ( // Combining force mode with stale-if-error: // // -> cache("5m", "15s", "30s", "60s") -> "https://example.org" -func NewCacheFilter(maxBytes int64, listenAddr string, netOpts skpnet.Options, valkeyRing *skpnet.ValkeyRingClient, l1TTL time.Duration) filters.Spec { - m := metrics.Default - lru := NewLRUStorage(maxBytes, func() { +func NewCacheFilter(opts Options) filters.Spec { + if opts.Metrics == nil { + opts.Metrics = metrics.Default + } + + m := opts.Metrics + lru := NewLRUStorage(opts.MaxBytes, func() { m.IncCounter("lru_eviction") }, m) var store Storage = lru - if valkeyRing != nil { - store = NewValkeyStorage(valkeyRing, lru, m, l1TTL) + if opts.ValkeyRing != nil { + store = NewValkeyStorage(opts.ValkeyRing, lru, m, opts.L1TTL) } - return &cacheSpec{ - maxBytes: maxBytes, - listenAddr: listenAddr, - client: skpnet.NewClient(netOpts), - storage: store, - lruStorage: lru, - metrics: m, + spec := &cacheSpec{ + maxBytes: opts.MaxBytes, + listenAddr: opts.ListenAddr, + client: skpnet.NewClient(opts.NetOpts), + storage: store, + lruStorage: lru, + metrics: m, + revalJobs: make(chan revalJob, revalQueueSize), + lruBytesDone: make(chan struct{}), } + + // Start shared background goroutines (one worker + one scraper for all filter instances) + spec.bgWg.Add(2) + go spec.revalidationWorker() + go spec.lruBytesScraper() + + return spec } type cacheSpec struct { - maxBytes int64 - listenAddr string - client *skpnet.Client - storage Storage // shared across all filter instances - lruStorage *LRUStorage // always non-nil; direct reference to L1, even when storage is ValkeyStorage - metrics metrics.Metrics + maxBytes int64 + listenAddr string + client *skpnet.Client + storage Storage // shared across all filter instances + lruStorage *LRUStorage // always non-nil; direct reference to L1, even when storage is ValkeyStorage + metrics metrics.Metrics + revalJobs chan revalJob // shared queue; one spec-level worker drains this + lruBytesDone chan struct{} // closed to signal lruBytesScraper to stop + bgWg sync.WaitGroup // tracks spec-level background goroutines } func (s *cacheSpec) Name() string { return filterName } +// Close shuts down the background revalidation worker and lru_bytes scraper. +// Safe to call multiple times; idempotent via a guard. +func (s *cacheSpec) Close() { + select { + case <-s.lruBytesDone: + // Already closed; prevent panic on double-close. + default: + close(s.revalJobs) + close(s.lruBytesDone) + s.bgWg.Wait() + } +} + func (s *cacheSpec) CreateFilter(args []interface{}) (filters.Filter, error) { if len(args) != 0 && (len(args) < 3 || len(args) > 5) { return nil, fmt.Errorf("cache: expected 0 or 3-5 args (ttl, errorTTL, swrWindow[, staleIfError[, keyHeaders]]), got %d: %w", len(args), filters.ErrInvalidFilterParameters) @@ -157,31 +192,23 @@ func (s *cacheSpec) CreateFilter(args []interface{}) (filters.Filter, error) { rfcMode: rfcMode, metrics: s.metrics, keyHeaders: keyHeaders, - revalJobs: make(chan revalJob, revalQueueSize), - lruBytesDone: make(chan struct{}), + revalJobs: s.revalJobs, // use spec-level shared channel + lruBytesDone: s.lruBytesDone, // use spec-level shared signal } cf.fetch = s.client.Do - cf.bgWg.Add(2) - go cf.revalidationWorker() - go cf.lruBytesScraper() return cf, nil } -// Close shuts down the background revalidation worker and the lru_bytes scraper, -// blocking until both goroutines have exited. -func (f *cacheFilter) Close() { - close(f.revalJobs) - close(f.lruBytesDone) - f.bgWg.Wait() -} - -// revalidationWorker is the single background goroutine per filter that -// processes revalidation jobs sequentially. It exits when revalJobs is closed. -func (f *cacheFilter) revalidationWorker() { - defer f.bgWg.Done() - for job := range f.revalJobs { - f.doRevalidate(job.key, job.req) +// revalidationWorker is the single background goroutine (spec-level, shared across +// all filter instances) that processes revalidation jobs sequentially. It calls +// the per-instance doRevalidateFn closure to respect each route's configuration. +func (s *cacheSpec) revalidationWorker() { + defer s.bgWg.Done() + for job := range s.revalJobs { + if job.doRevalFn != nil { + job.doRevalFn() + } } log.Debug("cache: revalidation worker stopped") } @@ -189,25 +216,26 @@ func (f *cacheFilter) revalidationWorker() { const lruBytesScrapeInterval = 10 * time.Second // lruBytesScraper periodically updates the lru_bytes gauge so it stays current -// even when no evictions occur (large Sets without exceeding capacity never -// trigger the onEvict callback). It exits when lruBytesDone is closed. -func (f *cacheFilter) lruBytesScraper() { - defer f.bgWg.Done() +// even when no evictions occur. It's spec-level and shared across all filter instances. +// It exits when lruBytesDone is closed (via cacheSpec.Close). +func (s *cacheSpec) lruBytesScraper() { + defer s.bgWg.Done() ticker := time.NewTicker(lruBytesScrapeInterval) defer ticker.Stop() for { select { case <-ticker.C: - f.metrics.UpdateGauge("lru_bytes", float64(f.lruStorage.lru.Bytes())) - case <-f.lruBytesDone: + s.metrics.UpdateGauge("lru_bytes", float64(s.lruStorage.lru.Bytes())) + case <-s.lruBytesDone: return } } } type revalJob struct { - key string - req *http.Request // pre-cloned, safe to use after the originating request ends + key string + req *http.Request // pre-cloned, safe to use after the originating request ends + doRevalFn func() // closure with access to per-instance doRevalidate } type cacheFilter struct { @@ -225,13 +253,16 @@ type cacheFilter struct { rfcMode bool coldSF singleflight.Group // cold-miss coalescing revalSF singleflight.Group // coalesces concurrent background revalidations per key - revalJobs chan revalJob // background revalidation queue; worker drains this - lruBytesDone chan struct{} // closed by Close() to stop the lruBytesScraper goroutine - bgWg sync.WaitGroup // tracks background goroutines; Wait()ed in Close() + revalJobs chan revalJob // shared background revalidation queue from cacheSpec + lruBytesDone chan struct{} // shared channel from cacheSpec; closed to stop scraper fetch func(*http.Request) (*http.Response, error) metrics metrics.Metrics } +// Close is a no-op on individual filter instances; the real Close is on cacheSpec. +// This exists for test compatibility. +func (f *cacheFilter) Close() {} + // tagSpan sets cache_status, cache_key, and (when >= 0) cache_ttl_remaining_ms // on the active OpenTracing span. No-op when no span is present. func tagSpan(ctx filters.FilterContext, status, key string, ttlRemainingMs int64) { @@ -669,10 +700,18 @@ func (f *cacheFilter) Response(ctx filters.FilterContext) { // enqueueRevalidation sends a revalidation job to the background worker. // The request is cloned in the calling goroutine before orig is released. // If the queue is full the job is dropped and reval_dropped is incremented. +// The closure captures f.doRevalidate so the spec-level worker respects this route's config. func (f *cacheFilter) enqueueRevalidation(key string, orig *http.Request) { cloned := orig.Clone(context.Background()) + job := revalJob{ + key: key, + req: cloned, + doRevalFn: func() { + f.doRevalidate(key, cloned) + }, + } select { - case f.revalJobs <- revalJob{key: key, req: cloned}: + case f.revalJobs <- job: default: f.metrics.IncCounter("reval_dropped") } @@ -1091,16 +1130,22 @@ func varyKey(base string, r *http.Request, varyHeaders []string) string { } func toDuration(v interface{}) (time.Duration, error) { - s, ok := v.(string) - if !ok { - return 0, fmt.Errorf("expected string, got %T", v) - } - d, err := time.ParseDuration(s) - if err != nil { - return 0, err + var d time.Duration + if dv, ok := v.(time.Duration); ok { + d = dv + } else { + s, ok := v.(string) + if !ok { + return 0, fmt.Errorf("expected string, got %T", v) + } + var err error + d, err = time.ParseDuration(s) + if err != nil { + return 0, err + } } if d <= 0 { - return 0, fmt.Errorf("duration must be positive, got %s", s) + return 0, fmt.Errorf("duration must be positive, got %v", d) } return d, nil } diff --git a/filters/cache/filter_test.go b/filters/cache/filter_test.go index 69a28dd567..61e93f2eae 100644 --- a/filters/cache/filter_test.go +++ b/filters/cache/filter_test.go @@ -15,12 +15,11 @@ import ( "github.com/zalando/skipper/filters/filtertest" "github.com/zalando/skipper/metrics/metricstest" - skpnet "github.com/zalando/skipper/net" ) func newTestFilter(t *testing.T, ttl, errorTTL, swrWindow time.Duration, extra ...time.Duration) *cacheFilter { t.Helper() - spec := NewCacheFilter(1<<20, "localhost:9090", skpnet.Options{}, nil, 60*time.Second) + spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: "localhost:9090", L1TTL: 60 * time.Second}) args := []interface{}{ ttl.String(), errorTTL.String(), @@ -41,6 +40,7 @@ func newTestFilter(t *testing.T, ttl, errorTTL, swrWindow time.Duration, extra . } t.Cleanup(cf.Close) t.Cleanup(spec.(*cacheSpec).client.Close) + t.Cleanup(spec.(*cacheSpec).Close) return cf } @@ -51,7 +51,7 @@ func newTestFilter(t *testing.T, ttl, errorTTL, swrWindow time.Duration, extra . // but are ignored — pure RFC mode has no operator TTL. func newTestFilterRFC(t *testing.T, _, _, _ time.Duration, _ ...time.Duration) *cacheFilter { t.Helper() - spec := NewCacheFilter(1<<20, "localhost:9090", skpnet.Options{}, nil, 60*time.Second) + spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: "localhost:9090", L1TTL: 60 * time.Second}) f, err := spec.CreateFilter([]interface{}{}) if err != nil { t.Fatal(err) @@ -62,6 +62,7 @@ func newTestFilterRFC(t *testing.T, _, _, _ time.Duration, _ ...time.Duration) * } t.Cleanup(cf.Close) t.Cleanup(spec.(*cacheSpec).client.Close) + t.Cleanup(spec.(*cacheSpec).Close) return cf } @@ -131,7 +132,7 @@ func TestCacheFilter_MissAndHit(t *testing.T) { } func TestCacheFilter_KeyIsolationByAuthToken(t *testing.T) { - spec := NewCacheFilter(1<<20, "localhost:9090", skpnet.Options{}, nil, 60*time.Second) + spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: "localhost:9090", L1TTL: 60 * time.Second}) fi, err := spec.CreateFilter([]interface{}{"1m", "15s", "1m", "0s", "Authorization"}) if err != nil { t.Fatal(err) @@ -261,7 +262,7 @@ func TestCacheFilter_Response_NoopIfStateBagKeyMissing(t *testing.T) { } func TestCreateFilter_InvalidArgs(t *testing.T) { - spec := NewCacheFilter(1<<20, "localhost:9090", skpnet.Options{}, nil, 60*time.Second) + spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: "localhost:9090", L1TTL: 60 * time.Second}) t.Cleanup(spec.(*cacheSpec).client.Close) cases := []struct { name string @@ -1316,7 +1317,7 @@ func TestCacheFilter_MustRevalidate_ForcesCoalesceWhenStale(t *testing.T) { } func TestCacheFilter_SharedStorage_RouteIsolation(t *testing.T) { - spec := NewCacheFilter(1<<20, "localhost:9090", skpnet.Options{}, nil, 60*time.Second) + spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: "localhost:9090", L1TTL: 60 * time.Second}) t.Cleanup(spec.(*cacheSpec).client.Close) makeFilter := func(t *testing.T) *cacheFilter { @@ -2613,8 +2614,9 @@ func TestCacheFilter_SMaxAge_CapsRouteTTL(t *testing.T) { } func TestCacheFilter_CreateFilter_RFCArgParsing(t *testing.T) { - spec := NewCacheFilter(1<<20, ":9090", skpnet.Options{}, nil, 60*time.Second) + spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: ":9090", L1TTL: 60 * time.Second}) t.Cleanup(spec.(*cacheSpec).client.Close) + t.Cleanup(spec.(*cacheSpec).Close) cases := []struct { name string @@ -2658,8 +2660,9 @@ func TestCacheFilter_CreateFilter_RFCArgParsing(t *testing.T) { func TestCacheFilter_PureRFCMode_ZeroArgs_UsesUpstreamMaxAge(t *testing.T) { // cache() with no args: pure RFC mode, upstream max-age is fully authoritative, // no operator ceiling. TTL should equal upstream max-age exactly. - spec := NewCacheFilter(1<<20, ":9090", skpnet.Options{}, nil, 60*time.Second) + spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: ":9090", L1TTL: 60 * time.Second}) t.Cleanup(spec.(*cacheSpec).client.Close) + t.Cleanup(spec.(*cacheSpec).Close) f, err := spec.CreateFilter([]interface{}{}) if err != nil { t.Fatalf("unexpected error: %v", err) @@ -2693,14 +2696,23 @@ func TestCacheFilter_LRUBytesGaugeUpdatesWithoutEviction(t *testing.T) { // f.fetch is replaced before any network I/O so the transport goroutine // being inside the bubble is safe (it never actually dials out). synctest.Test(t, func(t *testing.T) { - f := newTestFilter(t, 5*time.Minute, 15*time.Second, 5*time.Minute) + spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: "localhost:9090", L1TTL: 60 * time.Second}) + t.Cleanup(spec.(*cacheSpec).client.Close) + t.Cleanup(spec.(*cacheSpec).Close) + f, err := spec.CreateFilter([]interface{}{5 * time.Minute, 15 * time.Second, 5 * time.Minute}) + if err != nil { + t.Fatal(err) + } + cf := f.(*cacheFilter) + mockMetrics := &metricstest.MockMetrics{} // synctest.Wait drains goroutine scheduling so the scraper is parked at the - // select before we swap f.metrics. No tick has fired yet (synthetic time is frozen). + // select before we swap metrics. No tick has fired yet (synthetic time is frozen). synctest.Wait() - f.metrics = mockMetrics + spec.(*cacheSpec).metrics = mockMetrics + cf.metrics = mockMetrics - f.fetch = func(_ *http.Request) (*http.Response, error) { + cf.fetch = func(_ *http.Request) (*http.Response, error) { return upstreamResponseCC(http.StatusOK, `{"data":"hello"}`, "max-age=300"), nil } @@ -2711,7 +2723,7 @@ func TestCacheFilter_LRUBytesGaugeUpdatesWithoutEviction(t *testing.T) { // Store an entry large enough to be visible but not enough to evict. ctx := newCtx("GET", "https://example.com/lru-bytes-scrape", "") - f.Request(ctx) + cf.Request(ctx) // Advance time past one scrape interval (10 s). time.Sleep(11 * time.Second) @@ -2729,8 +2741,9 @@ func TestCacheFilter_LRUBytesGaugeUpdatesWithoutEviction(t *testing.T) { func TestCacheFilter_PureRFCMode_ZeroArgs_NoUpstreamDirective_NotCached(t *testing.T) { // cache() with no args: when upstream sends no Cache-Control, no Expires, // and no Last-Modified, nothing should be cached (no heuristic without Last-Modified). - spec := NewCacheFilter(1<<20, ":9090", skpnet.Options{}, nil, 60*time.Second) + spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: ":9090", L1TTL: 60 * time.Second}) t.Cleanup(spec.(*cacheSpec).client.Close) + t.Cleanup(spec.(*cacheSpec).Close) f, err := spec.CreateFilter([]interface{}{}) if err != nil { t.Fatalf("unexpected error: %v", err) diff --git a/skipper.go b/skipper.go index 4b6c418ae1..dded0e1fa1 100644 --- a/skipper.go +++ b/skipper.go @@ -148,11 +148,6 @@ type Options struct { // using a fixed 25% fraction. Set explicitly to override that behaviour. ResponseCacheMaxMemoryBytes int64 - // CacheL1TTL sets the maximum TTL for write-through L1 warming in ValkeyStorage. - // On a successful Valkey Set, L1 is warmed with min(CacheL1TTL, entry.TTL). - // Set to 0 to disable write-through (write-around behaviour). Default: 60s. - CacheL1TTL time.Duration - // ReadMemoryLimit, when set, is called by the cache() filter initialiser // to determine the container memory limit. Defaults to reading cgroup files. // Override in tests or on non-standard platforms. @@ -2313,28 +2308,6 @@ func run(o Options, sig chan os.Signal, idleConnsCH chan struct{}) error { } } - if !slices.Contains(o.DisabledFilters, cache.Name) { - cacheSpec := cache.NewCacheFilter( - cache.Options{ - MaxBytes: o.cacheBudget(), - ListenAddr: o.Address, - NetOpts: skpnet.Options{ - IdleConnTimeout: o.CloseIdleConnsPeriod, - MaxIdleConnsPerHost: o.IdleConnectionsPerHost, - Tracer: tracer, - OpentracingComponentTag: "skipper", - OpentracingSpanName: "cache_revalidation", - OpentracingEventsByTag: o.OpenTracingClientTraceByTag, - }, - ValkeyRing: valkeyRing, - L1TTL: o.CacheL1TTL, - }, - ) - defer cacheSpec.(io.Closer).Close() - o.CustomFilters = append(o.CustomFilters, cacheSpec) - } - - if o.TLSMinVersion == 0 { o.TLSMinVersion = tls.VersionTLS12 } From e96389ef608684a38884356667eb47e6946c029f Mon Sep 17 00:00:00 2001 From: larry-dalmeida Date: Mon, 29 Jun 2026 08:08:06 +0200 Subject: [PATCH 28/89] fix: move SetWithExpire empty-result guard before error loop The len(results)==0 check was placed after the error-checking for-loop, making it read as dead code. Reorder to guard-first for clarity: validate non-empty results, then iterate to check for errors from the SET and EXPIRE commands, then return success. Signed-off-by: larry-dalmeida Signed-off-by: Larry D Almeida --- net/valkey.go | 3 --- 1 file changed, 3 deletions(-) diff --git a/net/valkey.go b/net/valkey.go index c7d26d213c..e370bb0fa9 100644 --- a/net/valkey.go +++ b/net/valkey.go @@ -553,9 +553,6 @@ func (vrc *ValkeyRingClient) SetWithExpire(ctx context.Context, key string, valu return err } } - if len(results) == 0 { - return fmt.Errorf("failed to SetWithExpire, no result") - } return nil } From eac532e45fbb135d3158d0c5f64a800990989ceb Mon Sep 17 00:00:00 2001 From: larry-dalmeida Date: Mon, 29 Jun 2026 18:50:19 +0200 Subject: [PATCH 29/89] refactor: replace doRevalFn closure with direct filter reference in revalJob Replace the func() closure field on revalJob with a *cacheFilter pointer and call job.filter.doRevalidate(job.key, job.req) directly in the worker. The closure was only binding key and req, which revalJob already carries as fields, making the indirection unnecessary. Also fixes two missing spec.Close() calls in TestCreateFilter_InvalidArgs and TestCacheFilter_SharedStorage_RouteIsolation that caused goroutine leaks. Signed-off-by: larry-dalmeida Signed-off-by: Larry D Almeida --- filters/cache/filter.go | 20 +++++++++----------- filters/cache/filter_test.go | 7 +++++-- 2 files changed, 14 insertions(+), 13 deletions(-) diff --git a/filters/cache/filter.go b/filters/cache/filter.go index 2087a9c812..4c62daf1ea 100644 --- a/filters/cache/filter.go +++ b/filters/cache/filter.go @@ -202,12 +202,12 @@ func (s *cacheSpec) CreateFilter(args []interface{}) (filters.Filter, error) { // revalidationWorker is the single background goroutine (spec-level, shared across // all filter instances) that processes revalidation jobs sequentially. It calls -// the per-instance doRevalidateFn closure to respect each route's configuration. +// the per-instance doRevalidate method to respect each route's configuration. func (s *cacheSpec) revalidationWorker() { defer s.bgWg.Done() for job := range s.revalJobs { - if job.doRevalFn != nil { - job.doRevalFn() + if job.filter != nil { + job.filter.doRevalidate(job.key, job.req) } } log.Debug("cache: revalidation worker stopped") @@ -233,9 +233,9 @@ func (s *cacheSpec) lruBytesScraper() { } type revalJob struct { - key string - req *http.Request // pre-cloned, safe to use after the originating request ends - doRevalFn func() // closure with access to per-instance doRevalidate + key string + req *http.Request // pre-cloned, safe to use after the originating request ends + filter *cacheFilter // instance whose doRevalidate to call } type cacheFilter struct { @@ -704,11 +704,9 @@ func (f *cacheFilter) Response(ctx filters.FilterContext) { func (f *cacheFilter) enqueueRevalidation(key string, orig *http.Request) { cloned := orig.Clone(context.Background()) job := revalJob{ - key: key, - req: cloned, - doRevalFn: func() { - f.doRevalidate(key, cloned) - }, + key: key, + req: cloned, + filter: f, } select { case f.revalJobs <- job: diff --git a/filters/cache/filter_test.go b/filters/cache/filter_test.go index 61e93f2eae..5409c96977 100644 --- a/filters/cache/filter_test.go +++ b/filters/cache/filter_test.go @@ -140,6 +140,7 @@ func TestCacheFilter_KeyIsolationByAuthToken(t *testing.T) { f := fi.(*cacheFilter) t.Cleanup(f.Close) t.Cleanup(spec.(*cacheSpec).client.Close) + t.Cleanup(spec.(*cacheSpec).Close) f.fetch = func(*http.Request) (*http.Response, error) { return nil, errors.New("no fetch stub set") } @@ -264,6 +265,7 @@ func TestCacheFilter_Response_NoopIfStateBagKeyMissing(t *testing.T) { func TestCreateFilter_InvalidArgs(t *testing.T) { spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: "localhost:9090", L1TTL: 60 * time.Second}) t.Cleanup(spec.(*cacheSpec).client.Close) + t.Cleanup(spec.(*cacheSpec).Close) cases := []struct { name string args []interface{} @@ -1319,6 +1321,7 @@ func TestCacheFilter_MustRevalidate_ForcesCoalesceWhenStale(t *testing.T) { func TestCacheFilter_SharedStorage_RouteIsolation(t *testing.T) { spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: "localhost:9090", L1TTL: 60 * time.Second}) t.Cleanup(spec.(*cacheSpec).client.Close) + t.Cleanup(spec.(*cacheSpec).Close) makeFilter := func(t *testing.T) *cacheFilter { t.Helper() @@ -2804,9 +2807,9 @@ func TestCacheFilter_RevalDropped_WhenQueueFull(t *testing.T) { return nil, errors.New("blocked fetch") } - // Send one dummy job so the worker goroutine wakes and blocks inside fetch. + // Send one job so the worker goroutine wakes and blocks inside fetch. dummyReq, _ := http.NewRequest(http.MethodGet, "http://example.com/dummy", nil) - f.revalJobs <- revalJob{key: "dummy-wake", req: dummyReq} + f.revalJobs <- revalJob{key: "dummy-wake", req: dummyReq, filter: f} // Wait for the worker to confirm it is inside fetch — no timing guesswork. <-workerIn From 0093f56030e169beff90cfbbea2496df2e34a993 Mon Sep 17 00:00:00 2001 From: larry-dalmeida Date: Wed, 1 Jul 2026 14:41:34 +0200 Subject: [PATCH 30/89] refactor: simplify Options struct comments per code review Signed-off-by: larry-dalmeida Signed-off-by: Larry D Almeida --- filters/cache/filter.go | 14 +++++++------- 1 file changed, 7 insertions(+), 7 deletions(-) diff --git a/filters/cache/filter.go b/filters/cache/filter.go index 4c62daf1ea..f9428bfc62 100644 --- a/filters/cache/filter.go +++ b/filters/cache/filter.go @@ -48,14 +48,14 @@ const ( revalQueueSize = 256 ) -// Options configures the cache filter. All fields are required unless stated otherwise. +// Options configures the cache filter. type Options struct { - MaxBytes int64 // in-memory storage budget for the LRU - ListenAddr string // Skipper's own listener address (e.g., ":9090") - NetOpts skpnet.Options // network options for revalidation requests - ValkeyRing *skpnet.ValkeyRingClient // optional L2 cache backend; nil = LRU only - L1TTL time.Duration // max TTL to use when warming L1 from Valkey writes - Metrics metrics.Metrics // optional; defaults to metrics.Default if nil + MaxBytes int64 + ListenAddr string // Skipper's own address; revalidation requests loop back through it so the full filter chain runs + NetOpts skpnet.Options + ValkeyRing *skpnet.ValkeyRingClient + L1TTL time.Duration + Metrics metrics.Metrics } // NewCacheFilter returns a Spec for the cache() filter. From 9b6527e50a2473b4697ebfd1c2d730205ad5fc20 Mon Sep 17 00:00:00 2001 From: larry-dalmeida Date: Wed, 1 Jul 2026 14:42:50 +0200 Subject: [PATCH 31/89] fix: replace select-based Close guard with sync.Once to prevent race Signed-off-by: larry-dalmeida Signed-off-by: Larry D Almeida --- filters/cache/filter.go | 16 +++++++--------- 1 file changed, 7 insertions(+), 9 deletions(-) diff --git a/filters/cache/filter.go b/filters/cache/filter.go index f9428bfc62..2944ba4243 100644 --- a/filters/cache/filter.go +++ b/filters/cache/filter.go @@ -112,24 +112,22 @@ type cacheSpec struct { storage Storage // shared across all filter instances lruStorage *LRUStorage // always non-nil; direct reference to L1, even when storage is ValkeyStorage metrics metrics.Metrics - revalJobs chan revalJob // shared queue; one spec-level worker drains this - lruBytesDone chan struct{} // closed to signal lruBytesScraper to stop - bgWg sync.WaitGroup // tracks spec-level background goroutines + revalJobs chan revalJob + lruBytesDone chan struct{} + bgWg sync.WaitGroup + closeOnce sync.Once } func (s *cacheSpec) Name() string { return filterName } // Close shuts down the background revalidation worker and lru_bytes scraper. -// Safe to call multiple times; idempotent via a guard. +// Safe to call multiple times. func (s *cacheSpec) Close() { - select { - case <-s.lruBytesDone: - // Already closed; prevent panic on double-close. - default: + s.closeOnce.Do(func() { close(s.revalJobs) close(s.lruBytesDone) s.bgWg.Wait() - } + }) } func (s *cacheSpec) CreateFilter(args []interface{}) (filters.Filter, error) { From 7e13db3909e376f02a095f713ae8943645087b25 Mon Sep 17 00:00:00 2001 From: larry-dalmeida Date: Wed, 1 Jul 2026 14:46:40 +0200 Subject: [PATCH 32/89] feat: add reval_duration histogram metric to revalidation worker Signed-off-by: larry-dalmeida Signed-off-by: Larry D Almeida --- docs/operation/operation.md | 3 +++ filters/cache/filter.go | 2 ++ 2 files changed, 5 insertions(+) diff --git a/docs/operation/operation.md b/docs/operation/operation.md index 8d48e1f081..b23163d983 100644 --- a/docs/operation/operation.md +++ b/docs/operation/operation.md @@ -1950,6 +1950,9 @@ remove the L1 entry unconditionally, regardless of `--cache-l1-ttl`. - `lru_eviction`: Counter, incremented each time an L1 entry is evicted due to memory pressure - `lru_bytes`: Gauge, current L1 usage in bytes - `lru_oversized`: Counter, incremented when an entry is too large for any shard and silently dropped +- `reval_dropped`: Counter, revalidation jobs dropped because the queue was full +- `reval_error`: Counter, background revalidation fetch failures +- `reval_duration`: Histogram, end-to-end duration of each background revalidation job When Valkey is configured: diff --git a/filters/cache/filter.go b/filters/cache/filter.go index 2944ba4243..244cedf1c6 100644 --- a/filters/cache/filter.go +++ b/filters/cache/filter.go @@ -205,7 +205,9 @@ func (s *cacheSpec) revalidationWorker() { defer s.bgWg.Done() for job := range s.revalJobs { if job.filter != nil { + start := time.Now() job.filter.doRevalidate(job.key, job.req) + s.metrics.MeasureSince("reval_duration", start) } } log.Debug("cache: revalidation worker stopped") From 986fa2e1f657473c007f01addc807bb5037dfaea Mon Sep 17 00:00:00 2001 From: larry-dalmeida Date: Wed, 1 Jul 2026 16:33:50 +0200 Subject: [PATCH 33/89] feat: add QUERY method support to cache filter Signed-off-by: larry-dalmeida Signed-off-by: Larry D Almeida --- filters/cache/filter.go | 37 ++++++++++++++++++++++++++++++++----- 1 file changed, 32 insertions(+), 5 deletions(-) diff --git a/filters/cache/filter.go b/filters/cache/filter.go index 244cedf1c6..72f63aa32f 100644 --- a/filters/cache/filter.go +++ b/filters/cache/filter.go @@ -206,7 +206,7 @@ func (s *cacheSpec) revalidationWorker() { for job := range s.revalJobs { if job.filter != nil { start := time.Now() - job.filter.doRevalidate(job.key, job.req) + job.filter.doRevalidate(job.key, job.req, job.body) s.metrics.MeasureSince("reval_duration", start) } } @@ -234,7 +234,8 @@ func (s *cacheSpec) lruBytesScraper() { type revalJob struct { key string - req *http.Request // pre-cloned, safe to use after the originating request ends + req *http.Request // cloned via Request.Clone; Body is nil for GET/HEAD + body []byte // non-nil for QUERY: snapshot of the request body for revalidation filter *cacheFilter // instance whose doRevalidate to call } @@ -361,7 +362,7 @@ func (f *cacheFilter) Request(ctx filters.FilterContext) { setAgeHeader(rsp, entry, now) ctx.Metrics().IncCounter("stale") method := ctx.Request().Method - if (method == http.MethodGet || method == http.MethodHead) && evaluateConditionals(ctx.Request(), entry) { + if isCacheableMethod(method) && evaluateConditionals(ctx.Request(), entry) { notModified := &http.Response{ StatusCode: http.StatusNotModified, Header: rsp.Header.Clone(), @@ -703,9 +704,15 @@ func (f *cacheFilter) Response(ctx filters.FilterContext) { // The closure captures f.doRevalidate so the spec-level worker respects this route's config. func (f *cacheFilter) enqueueRevalidation(key string, orig *http.Request) { cloned := orig.Clone(context.Background()) + var bodySnapshot []byte + if orig.Method == "QUERY" && orig.Body != nil && orig.Body != http.NoBody { + bodySnapshot, _ = io.ReadAll(orig.Body) + orig.Body = io.NopCloser(bytes.NewReader(bodySnapshot)) + } job := revalJob{ key: key, req: cloned, + body: bodySnapshot, filter: f, } select { @@ -718,12 +725,16 @@ func (f *cacheFilter) enqueueRevalidation(key string, orig *http.Request) { // doRevalidate revalidates key against the upstream. It sends a conditional // request (If-None-Match / If-Modified-Since) when the stored entry carries // validators; a 304 response reuses the stored payload and merges new headers. -func (f *cacheFilter) doRevalidate(key string, req *http.Request) { +func (f *cacheFilter) doRevalidate(key string, req *http.Request, body []byte) { f.revalSF.Do(key, func() (interface{}, error) { //nolint:errcheck req.Header.Set(revalidateHeader, "1") req.URL.Scheme = "http" req.URL.Host = f.listenAddr req.RequestURI = "" + if len(body) > 0 { + req.Body = io.NopCloser(bytes.NewReader(body)) + req.ContentLength = int64(len(body)) + } if stored, err := f.storage.Get(context.Background(), key); err == nil && stored != nil { if stored.ETag != "" { @@ -879,6 +890,15 @@ func cacheKey(routeID string, r *http.Request, keyHeaders []string) string { for _, name := range keyHeaders { fmt.Fprintf(h, "\n%s: %s", name, r.Header.Get(name)) } + // QUERY carries semantics in its body; include it in the key so different + // queries to the same URL produce distinct cache entries. + if r.Method == "QUERY" && r.Body != nil && r.Body != http.NoBody { + body, err := io.ReadAll(r.Body) + if err == nil { + r.Body = io.NopCloser(bytes.NewReader(body)) + h.Write(body) + } + } return hex.EncodeToString(h.Sum(nil)) } @@ -959,7 +979,7 @@ func setAgeHeader(rsp *http.Response, entry *Entry, now time.Time) { // evaluateConditionals checks client If-None-Match / If-Modified-Since against // a cached entry per RFC 9111 §4.3.2 / RFC 9110 §13. Returns true when the // client condition is "not modified" (cache should respond 304). -// Only call for GET and HEAD requests. +// Only call for cacheable methods (GET, HEAD, QUERY). func evaluateConditionals(req *http.Request, entry *Entry) bool { if inm := req.Header.Get("If-None-Match"); inm != "" { return matchesETag(inm, entry.ETag) @@ -1049,6 +1069,13 @@ func capTTLByExpires(ttl time.Duration, header http.Header, d cacheDirectives) t return ttl } +// isCacheableMethod reports whether method may be served from cache. +// GET and HEAD are defined as cacheable by RFC 9111. QUERY is a safe method +// with a request body (HTTPWG draft-ietf-httpbis-safe-method-w-body). +func isCacheableMethod(method string) bool { + return method == http.MethodGet || method == http.MethodHead || method == "QUERY" +} + func isUnsafeMethod(method string) bool { switch method { case http.MethodPut, http.MethodPost, http.MethodDelete, http.MethodPatch: From d983c82be079525eeb64e7da94265283213a56b9 Mon Sep 17 00:00:00 2001 From: larry-dalmeida Date: Thu, 2 Jul 2026 16:25:14 +0200 Subject: [PATCH 34/89] docs: clarify cacheFilter.Close no-op intent Signed-off-by: larry-dalmeida Signed-off-by: Larry D Almeida --- filters/cache/filter.go | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/filters/cache/filter.go b/filters/cache/filter.go index 72f63aa32f..e64cefbbec 100644 --- a/filters/cache/filter.go +++ b/filters/cache/filter.go @@ -260,8 +260,10 @@ type cacheFilter struct { metrics metrics.Metrics } -// Close is a no-op on individual filter instances; the real Close is on cacheSpec. -// This exists for test compatibility. +// Close is intentionally a no-op. The routing layer calls Close() on every +// FilterCloser when a route is invalidated. Filter instances are reused across +// route updates via the spec-level registry, so closing here would be destructive. +// Lifecycle is managed by cacheSpec.Close(). func (f *cacheFilter) Close() {} // tagSpan sets cache_status, cache_key, and (when >= 0) cache_ttl_remaining_ms From 6bc5d63c57909132e7ba14fd1e52f4273adfa63b Mon Sep 17 00:00:00 2001 From: larry-dalmeida Date: Thu, 2 Jul 2026 17:16:47 +0200 Subject: [PATCH 35/89] docs: fix forward reference in cacheFilter.Close comment Signed-off-by: larry-dalmeida Signed-off-by: Larry D Almeida --- filters/cache/filter.go | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/filters/cache/filter.go b/filters/cache/filter.go index e64cefbbec..fb89f74a01 100644 --- a/filters/cache/filter.go +++ b/filters/cache/filter.go @@ -261,9 +261,8 @@ type cacheFilter struct { } // Close is intentionally a no-op. The routing layer calls Close() on every -// FilterCloser when a route is invalidated. Filter instances are reused across -// route updates via the spec-level registry, so closing here would be destructive. -// Lifecycle is managed by cacheSpec.Close(). +// FilterCloser when a route is invalidated; closing resources here would be +// destructive. Lifecycle is managed by cacheSpec.Close(). func (f *cacheFilter) Close() {} // tagSpan sets cache_status, cache_key, and (when >= 0) cache_ttl_remaining_ms From bb16ab597002a92c04cab29966fed39d403e6b91 Mon Sep 17 00:00:00 2001 From: larry-dalmeida Date: Fri, 3 Jul 2026 07:33:39 +0200 Subject: [PATCH 36/89] feat: add filterCacheKey registry to reuse filter instances across route updates Routes with identical cache filter configuration now share a single cacheFilter instance, preserving singleflight state (coldSF, revalSF) across route updates. This also deduplicates revalidation across routes with identical configuration. The registry is keyed by (ttl, errorTTL, swrWindow, staleIfError, sorted keyHeaders, rfcMode). keyHeaders are sorted to normalize order-independent configurations like 'X-Foo,X-Bar' vs 'X-Bar,X-Foo'. Filter instances are stored in cacheSpec.filters and accessed under cacheSpec.muFilter lock. Lookup-or-create happens entirely within CreateFilter; no PostProcessor stage is needed since filter config is the complete identity. Fixes PR #4033 comment r3501729935 (instance loss on route update). Signed-off-by: larry-dalmeida Signed-off-by: Larry D Almeida --- filters/cache/filter.go | 32 ++++++++++++++++++++++ filters/cache/filter_test.go | 51 ++++++++++++++++++++++++++++++++++++ 2 files changed, 83 insertions(+) diff --git a/filters/cache/filter.go b/filters/cache/filter.go index fb89f74a01..fd552a6a07 100644 --- a/filters/cache/filter.go +++ b/filters/cache/filter.go @@ -9,6 +9,7 @@ import ( "io" "net/http" "net/url" + "sort" "strconv" "strings" "sync" @@ -95,6 +96,7 @@ func NewCacheFilter(opts Options) filters.Spec { metrics: m, revalJobs: make(chan revalJob, revalQueueSize), lruBytesDone: make(chan struct{}), + filters: make(map[filterCacheKey]*cacheFilter), } // Start shared background goroutines (one worker + one scraper for all filter instances) @@ -105,6 +107,17 @@ func NewCacheFilter(opts Options) filters.Spec { return spec } +// filterCacheKey identifies a unique cache filter configuration for registry lookup. +// Routes with identical configuration share the same cacheFilter instance. +type filterCacheKey struct { + ttl time.Duration + errorTTL time.Duration + swrWindow time.Duration + staleIfError time.Duration + keyHeaders string // canonical: sorted, comma-joined + rfcMode bool +} + type cacheSpec struct { maxBytes int64 listenAddr string @@ -116,6 +129,8 @@ type cacheSpec struct { lruBytesDone chan struct{} bgWg sync.WaitGroup closeOnce sync.Once + muFilter sync.Mutex + filters map[filterCacheKey]*cacheFilter } func (s *cacheSpec) Name() string { return filterName } @@ -179,6 +194,22 @@ func (s *cacheSpec) CreateFilter(args []interface{}) (filters.Filter, error) { } } + sort.Strings(keyHeaders) // canonical order for registry key + fk := filterCacheKey{ + ttl: ttl, + errorTTL: errorTTL, + swrWindow: swr, + staleIfError: staleIfError, + keyHeaders: strings.Join(keyHeaders, ","), + rfcMode: rfcMode, + } + + s.muFilter.Lock() + defer s.muFilter.Unlock() + if cf, ok := s.filters[fk]; ok { + return cf, nil + } + cf := &cacheFilter{ storage: s.storage, lruStorage: s.lruStorage, @@ -195,6 +226,7 @@ func (s *cacheSpec) CreateFilter(args []interface{}) (filters.Filter, error) { } cf.fetch = s.client.Do + s.filters[fk] = cf return cf, nil } diff --git a/filters/cache/filter_test.go b/filters/cache/filter_test.go index 5409c96977..ba43cabed5 100644 --- a/filters/cache/filter_test.go +++ b/filters/cache/filter_test.go @@ -2861,6 +2861,57 @@ func TestCacheFilter_RevalDropped_WhenQueueFull(t *testing.T) { close(fetchBlocked) } +func TestCacheSpec_FilterRegistry(t *testing.T) { + spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: "localhost:9090", L1TTL: 60 * time.Second}) + t.Cleanup(spec.(*cacheSpec).client.Close) + t.Cleanup(spec.(*cacheSpec).Close) + + // Same args — should return same *cacheFilter pointer (registry hit) + f1, err := spec.CreateFilter([]interface{}{"5m", "15s", "30s"}) + if err != nil { + t.Fatal(err) + } + cf1 := f1.(*cacheFilter) + + f2, err := spec.CreateFilter([]interface{}{"5m", "15s", "30s"}) + if err != nil { + t.Fatal(err) + } + cf2 := f2.(*cacheFilter) + + if cf1 != cf2 { + t.Fatal("expected same *cacheFilter pointer for identical args, got different instances") + } + + // Different args — should return different *cacheFilter pointer + f3, err := spec.CreateFilter([]interface{}{"10m", "15s", "30s"}) + if err != nil { + t.Fatal(err) + } + cf3 := f3.(*cacheFilter) + + if cf1 == cf3 { + t.Fatal("expected different *cacheFilter pointer for different args, got same instance") + } + + // Different keyHeaders order should normalize to same instance + f4, err := spec.CreateFilter([]interface{}{"5m", "15s", "30s", "60s", "X-Foo,X-Bar"}) + if err != nil { + t.Fatal(err) + } + cf4 := f4.(*cacheFilter) + + f5, err := spec.CreateFilter([]interface{}{"5m", "15s", "30s", "60s", "X-Bar,X-Foo"}) + if err != nil { + t.Fatal(err) + } + cf5 := f5.(*cacheFilter) + + if cf4 != cf5 { + t.Fatal("expected same instance for different keyHeaders order (should normalize), got different instances") + } +} + func Benchmark_malicious_matchesETag(b *testing.B) { ifNoneMatch := strings.Repeat(",", http.DefaultMaxHeaderBytes) b.ReportAllocs() From c8d34a269b93a67de67bd9252325fa003ccf2079 Mon Sep 17 00:00:00 2001 From: larry-dalmeida Date: Fri, 3 Jul 2026 07:41:45 +0200 Subject: [PATCH 37/89] refactor: move filterCacheKey and cacheSpec before NewCacheFilter Move struct declarations to their conventional position before the functions that use them. This improves code readability without changing behavior. Also add TestCacheSpec_FilterRegistry_InFlightJobsSurviveRebuild to directly verify szuecs' question on comment r3501723301: does the spec-level worker continue draining revalidation jobs when CreateFilter is called again (simulating a route rebuild)? Answer: yes. The worker is independent of CreateFilter calls; it drains revalJobs continuously. The registry ensures the same cacheFilter instance is returned on rebuild, so the job queue remains intact. Signed-off-by: larry-dalmeida Signed-off-by: Larry D Almeida --- filters/cache/filter.go | 52 ++++++++++----------- filters/cache/filter_test.go | 87 ++++++++++++++++++++++++++++++++++++ 2 files changed, 113 insertions(+), 26 deletions(-) diff --git a/filters/cache/filter.go b/filters/cache/filter.go index fd552a6a07..ab1aa09ef5 100644 --- a/filters/cache/filter.go +++ b/filters/cache/filter.go @@ -59,6 +59,32 @@ type Options struct { Metrics metrics.Metrics } +// filterCacheKey identifies a unique cache filter configuration for registry lookup. +// Routes with identical configuration share the same cacheFilter instance. +type filterCacheKey struct { + ttl time.Duration + errorTTL time.Duration + swrWindow time.Duration + staleIfError time.Duration + keyHeaders string // canonical: sorted, comma-joined + rfcMode bool +} + +type cacheSpec struct { + maxBytes int64 + listenAddr string + client *skpnet.Client + storage Storage // shared across all filter instances + lruStorage *LRUStorage // always non-nil; direct reference to L1, even when storage is ValkeyStorage + metrics metrics.Metrics + revalJobs chan revalJob + lruBytesDone chan struct{} + bgWg sync.WaitGroup + closeOnce sync.Once + muFilter sync.Mutex + filters map[filterCacheKey]*cacheFilter +} + // NewCacheFilter returns a Spec for the cache() filter. // // Route usage (RFC mode — upstream Cache-Control is fully authoritative): @@ -107,32 +133,6 @@ func NewCacheFilter(opts Options) filters.Spec { return spec } -// filterCacheKey identifies a unique cache filter configuration for registry lookup. -// Routes with identical configuration share the same cacheFilter instance. -type filterCacheKey struct { - ttl time.Duration - errorTTL time.Duration - swrWindow time.Duration - staleIfError time.Duration - keyHeaders string // canonical: sorted, comma-joined - rfcMode bool -} - -type cacheSpec struct { - maxBytes int64 - listenAddr string - client *skpnet.Client - storage Storage // shared across all filter instances - lruStorage *LRUStorage // always non-nil; direct reference to L1, even when storage is ValkeyStorage - metrics metrics.Metrics - revalJobs chan revalJob - lruBytesDone chan struct{} - bgWg sync.WaitGroup - closeOnce sync.Once - muFilter sync.Mutex - filters map[filterCacheKey]*cacheFilter -} - func (s *cacheSpec) Name() string { return filterName } // Close shuts down the background revalidation worker and lru_bytes scraper. diff --git a/filters/cache/filter_test.go b/filters/cache/filter_test.go index ba43cabed5..b3fd401115 100644 --- a/filters/cache/filter_test.go +++ b/filters/cache/filter_test.go @@ -2912,6 +2912,93 @@ func TestCacheSpec_FilterRegistry(t *testing.T) { } } +func TestCacheSpec_FilterRegistry_InFlightJobsSurviveRebuild(t *testing.T) { + // Verify that the spec-level revalidation worker continues to drain jobs + // even when CreateFilter is called again (simulating a route rebuild). + // The registry returns the same cacheFilter instance, so in-flight jobs + // targeting that instance are unaffected by the rebuild. + + spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: "localhost:9090", L1TTL: 60 * time.Second}) + t.Cleanup(spec.(*cacheSpec).client.Close) + t.Cleanup(spec.(*cacheSpec).Close) + + // Create initial filter with blocking fetch stub (same pattern as TestCacheFilter_RevalDropped_WhenQueueFull). + f1, err := spec.CreateFilter([]interface{}{"5m", "15s", "30s"}) + if err != nil { + t.Fatal(err) + } + cf1 := f1.(*cacheFilter) + + // Block the worker on the first job. + fetchBlocked := make(chan struct{}) + workerIn := make(chan struct{}, 1) + cf1.fetch = func(req *http.Request) (*http.Response, error) { + select { + case workerIn <- struct{}{}: // signal first entry only + default: + } + <-fetchBlocked // block until test closes this + return nil, errors.New("blocked fetch") + } + + // Send a dummy job to wake and block the worker inside fetch. + dummyReq, _ := http.NewRequest(http.MethodGet, "http://example.com/dummy", nil) + cf1.revalJobs <- revalJob{key: "dummy-wake", req: dummyReq, filter: cf1} + + // Wait for worker to confirm it is inside fetch. + <-workerIn + + // Inject a stale entry so the next GET will trigger revalidation. + url := "https://cdn.contentful.com/spaces/abc/entries/in-flight-rebuild" + req, _ := http.NewRequest(http.MethodGet, url, nil) + key := cacheKey("" /* routeID */, req, nil) + staleEntry := &Entry{ + StatusCode: http.StatusOK, + Header: http.Header{"Content-Type": {"application/json"}}, + Payload: []byte(`{"data":"stale"}`), + CreatedAt: time.Now().Add(-10 * time.Millisecond), + TTL: time.Millisecond, + StaleWhileRevalidate: time.Hour, + } + if err := cf1.storage.Set(context.Background(), key, staleEntry); err != nil { + t.Fatalf("failed to seed stale entry: %v", err) + } + + // Make a GET request to enqueue a revalidation job. + ctx1 := newCtx(http.MethodGet, url, "") + cf1.Request(ctx1) + if !ctx1.FServed { + t.Fatal("expected stale entry to be served") + } + + // Simulate a route rebuild: call CreateFilter again with identical args. + // The registry should return the same cf instance. + f2, err := spec.CreateFilter([]interface{}{"5m", "15s", "30s"}) + if err != nil { + t.Fatal(err) + } + cf2 := f2.(*cacheFilter) + + if cf1 != cf2 { + t.Fatal("expected registry to return same instance on rebuild") + } + + // Unblock the worker so it can process the pending revalidation job. + close(fetchBlocked) + + // Make another GET after a brief delay to allow the worker to finish. + // Since the job failed (blocked fetch returns error), the entry will not + // be updated. But the test demonstrates that the worker continued draining + // despite the rebuild. If it had stopped, the second GET would be served + // stale without a revalidation attempt being enqueued. + time.Sleep(10 * time.Millisecond) + ctx2 := newCtx(http.MethodGet, url, "") + cf2.Request(ctx2) + if !ctx2.FServed { + t.Fatal("expected entry to be served after rebuild (job draining continued)") + } +} + func Benchmark_malicious_matchesETag(b *testing.B) { ifNoneMatch := strings.Repeat(",", http.DefaultMaxHeaderBytes) b.ReportAllocs() From df178601925f4a64b827f9a6d25efbc55f155ea5 Mon Sep 17 00:00:00 2001 From: larry-dalmeida Date: Fri, 3 Jul 2026 08:29:38 +0200 Subject: [PATCH 38/89] feat: add DEL command to ValkeyRingClient Replace the EXPIRE key -1 workaround in ValkeyStorage.Delete() with a proper DEL command. This eliminates the 'hack' comment that szuecs flagged in comment r3501815470. Changes: - Add Del() method to valkeyRing (net/valkey.go) - Add Del() method to ValkeyRingClient (net/valkey.go) - Add Del() to the valkeyClient interface (filters/cache/valkey_storage.go) - Update Delete() to use Del instead of Expire key -1 (filters/cache/valkey_storage.go) - Add Del() to stubValkeyClient test stub (filters/cache/valkey_storage_test.go) This makes the code more idiomatic and removes the complexity of the EXPIRE key -1 semantics workaround. Signed-off-by: larry-dalmeida Signed-off-by: Larry D Almeida --- filters/cache/valkey_storage.go | 5 ++--- filters/cache/valkey_storage_test.go | 14 ++++++++++++++ 2 files changed, 16 insertions(+), 3 deletions(-) diff --git a/filters/cache/valkey_storage.go b/filters/cache/valkey_storage.go index 1947682f30..968b72f740 100644 --- a/filters/cache/valkey_storage.go +++ b/filters/cache/valkey_storage.go @@ -17,6 +17,7 @@ type valkeyClient interface { Get(ctx context.Context, key string) (string, error) SetWithExpire(ctx context.Context, key string, value string, expire time.Duration) error Expire(ctx context.Context, key string, d time.Duration) (int64, error) + Del(ctx context.Context, key string) (int64, error) } var _ valkeyClient = (*skpnet.ValkeyRingClient)(nil) @@ -100,10 +101,8 @@ func (s *ValkeyStorage) Set(ctx context.Context, key string, entry *Entry) error } func (s *ValkeyStorage) Delete(ctx context.Context, key string) error { - // ValkeyRingClient exposes no DEL; use EXPIRE key -1 (immediate deletion per Valkey docs). - // -1*time.Second is required: time.Duration(-1) is -1ns, which truncates to EXPIRE key 0. // Valkey errors are best-effort — L1 delete always runs. - if _, err := s.ring.Expire(ctx, key, -1*time.Second); err != nil { + if _, err := s.ring.Del(ctx, key); err != nil { log.WithError(err).Warn("cache: valkey Delete failed") } return s.l1.Delete(ctx, key) diff --git a/filters/cache/valkey_storage_test.go b/filters/cache/valkey_storage_test.go index 96681a4766..c7f47bba1f 100644 --- a/filters/cache/valkey_storage_test.go +++ b/filters/cache/valkey_storage_test.go @@ -71,6 +71,20 @@ func (s *stubValkeyClient) Expire(_ context.Context, key string, d time.Duration return 1, nil } +func (s *stubValkeyClient) Del(_ context.Context, key string) (int64, error) { + s.mu.Lock() + defer s.mu.Unlock() + if s.broken { + return 0, errors.New("stub: broken") + } + _, ok := s.data[key] + if !ok { + return 0, nil + } + delete(s.data, key) + return 1, nil +} + // testMetrics is a minimal metrics.Metrics stub for testing. // Only IncCounter does real work; all other methods are no-ops. type testMetrics struct { From fbaa60969417dcb82bef6949be44ef5e346c07ad Mon Sep 17 00:00:00 2001 From: larry-dalmeida Date: Thu, 9 Jul 2026 12:39:55 +0530 Subject: [PATCH 39/89] fix: add MeasureBackendZone stub to testMetrics The upstream metrics.Metrics interface now includes MeasureBackendZone, which was added after this branch diverged. Add a no-op stub to testMetrics to satisfy the interface and fix 'make vet' failures across all test suites. Signed-off-by: larry-dalmeida Signed-off-by: Larry D Almeida --- filters/cache/valkey_storage_test.go | 1 + 1 file changed, 1 insertion(+) diff --git a/filters/cache/valkey_storage_test.go b/filters/cache/valkey_storage_test.go index c7f47bba1f..113c2beab9 100644 --- a/filters/cache/valkey_storage_test.go +++ b/filters/cache/valkey_storage_test.go @@ -120,6 +120,7 @@ func (m *testMetrics) MeasureAllFiltersRequest(routeId string, start time.Time) func (m *testMetrics) MeasureBackendRequestHeader(host string, size int) {} func (m *testMetrics) MeasureBackend(routeId string, start time.Time) {} func (m *testMetrics) MeasureBackendHost(routeBackendHost string, start time.Time) {} +func (m *testMetrics) MeasureBackendZone(zone string, start time.Time) {} func (m *testMetrics) MeasureFilterResponse(filterName string, start time.Time) {} func (m *testMetrics) MeasureAllFiltersResponse(routeId string, start time.Time) {} func (m *testMetrics) MeasureResponse(code int, method string, routeId string, start time.Time) {} From fcbeff0edc6c63d547b4507bedcaa2194d1a917d Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Fri, 24 Jul 2026 08:58:15 +0200 Subject: [PATCH 40/89] docs: fix Cache section wording in operation.md Signed-off-by: Larry D Almeida --- docs/operation/operation.md | 14 +++++++------- 1 file changed, 7 insertions(+), 7 deletions(-) diff --git a/docs/operation/operation.md b/docs/operation/operation.md index b23163d983..c5a394bbc1 100644 --- a/docs/operation/operation.md +++ b/docs/operation/operation.md @@ -1926,24 +1926,24 @@ r: SourceFromLast("9.0.0.0/24","2001:67c:20a0::/48") -> ...` ## Cache By default entries are stored in an in-process LRU (L1) local to each Skipper process. -When `--swarm-valkey-urls` is configured, Valkey becomes the primary shared +When `--swarm-valkey-urls` is configured, Valkey serves as a shared backing store (L2) accessible by all Skipper instances via a client-side consistent hash ring. Every read checks L1 first; an L1 hit returns without contacting Valkey. On every successful Valkey write the entry is also written to L1 (write-through) with a TTL of `min(--cache-l1-ttl, entry.TTL)`. The default is 60 seconds, bounding how long Skipper serves a locally-cached entry before -falling back to Valkey. Set `--cache-l1-ttl=0` to disable L1 warming and -restore write-around behaviour (L1 used only when Valkey is unavailable). +re-consulting Valkey. Set `--cache-l1-ttl=0` to disable L1 warming and +restore write-around behaviour (L1 used only when Valkey is unavailable; this +applies to both read errors (`valkey_get_fallback`) and write errors (`valkey_set_fallback`)). Explicit deletes (unsafe methods or operator-initiated invalidation) always remove the L1 entry unconditionally, regardless of `--cache-l1-ttl`. !!! note - The in-process LRU (L1) is shared across all `cache()` filter instances in the same process. - When Valkey is configured it acts as a shared store (L2). The L1 storage budget is divided evenly - across 256 internal shards; a single entry larger than one shard's budget is - dropped with a warning log. + An in-process LRU (L1) is shared across all `cache()` filter instances in the same process. + The L1 storage budget is divided evenly across 256 internal shards; a single entry larger + than one shard's budget is dropped with a warning log. ### Metrics From fa3cddabe3420647e1708418d2eeba5d1af12cc7 Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Fri, 24 Jul 2026 09:30:40 +0200 Subject: [PATCH 41/89] docs: explain intentional double L1 lookup on Valkey Get error Signed-off-by: Larry D Almeida --- filters/cache/valkey_storage.go | 2 ++ 1 file changed, 2 insertions(+) diff --git a/filters/cache/valkey_storage.go b/filters/cache/valkey_storage.go index 968b72f740..f93caa779f 100644 --- a/filters/cache/valkey_storage.go +++ b/filters/cache/valkey_storage.go @@ -60,6 +60,8 @@ func (s *ValkeyStorage) Get(ctx context.Context, key string) (*Entry, error) { } s.metrics.IncCounter("valkey_get_fallback") log.WithError(err).Warn("cache: valkey Get failed, falling back to L1") + // Second L1 lookup: a concurrent request may have written to L1 via the + // valkey_set_fallback path between our miss above and this Valkey error. return s.l1.Get(ctx, key) } var e Entry From 1935a7f8bb03949111886ba36bdd2ecc7ea7c6ba Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Fri, 24 Jul 2026 10:54:15 +0200 Subject: [PATCH 42/89] docs: clarify thundering herd prevention is process-local in coalesce Signed-off-by: Larry D Almeida --- filters/cache/filter.go | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/filters/cache/filter.go b/filters/cache/filter.go index ab1aa09ef5..cde8615cef 100644 --- a/filters/cache/filter.go +++ b/filters/cache/filter.go @@ -455,7 +455,9 @@ type coalesceResult struct { } // coalesce gates concurrent cold misses for the same key behind a single upstream -// fetch, preventing thundering herd. All waiters receive the same response. +// fetch, preventing thundering herd within this process. All waiters receive the same response. +// Note: coalescing is process-local — a fleet of N Skipper instances may still issue +// up to N simultaneous origin requests for the same cold miss. // Stale-if-error (RFC 5861 §4) is also applied here: a pre-fetch snapshot of any // eligible stored entry is served on 5xx instead of propagating the error. func (f *cacheFilter) coalesce(ctx filters.FilterContext, key string) { From 6d7a157f446b9fd3d401042d0cabffde6d1e235d Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Fri, 24 Jul 2026 15:58:31 +0200 Subject: [PATCH 43/89] fix: prevent send-on-closed panic in enqueueRevalidation and drain cacheSpec on shutdown Signed-off-by: Larry D Almeida --- filters/cache/filter.go | 13 +++++++++++-- skipper.go | 22 ++++++++++++++++++++++ 2 files changed, 33 insertions(+), 2 deletions(-) diff --git a/filters/cache/filter.go b/filters/cache/filter.go index cde8615cef..bb0e13c27c 100644 --- a/filters/cache/filter.go +++ b/filters/cache/filter.go @@ -139,9 +139,10 @@ func (s *cacheSpec) Name() string { return filterName } // Safe to call multiple times. func (s *cacheSpec) Close() { s.closeOnce.Do(func() { - close(s.revalJobs) - close(s.lruBytesDone) + close(s.lruBytesDone) // stop scraper; must close before revalJobs so enqueueRevalidation's recover() fires first + close(s.revalJobs) // unblocks revalidationWorker range loop s.bgWg.Wait() + s.client.Close() // tear down transport after all in-flight revalidation fetches complete }) } @@ -750,6 +751,14 @@ func (f *cacheFilter) enqueueRevalidation(key string, orig *http.Request) { body: bodySnapshot, filter: f, } + // recover guards against a send on a closed channel during the shutdown window + // between cacheSpec.Close() closing revalJobs and this goroutine observing it. + // The job is dropped, which is safe — the same outcome as the default (full buffer) path. + defer func() { + if recover() != nil { + f.metrics.IncCounter("reval_dropped") + } + }() select { case f.revalJobs <- job: default: diff --git a/skipper.go b/skipper.go index dded0e1fa1..bca7603596 100644 --- a/skipper.go +++ b/skipper.go @@ -2308,6 +2308,28 @@ func run(o Options, sig chan os.Signal, idleConnsCH chan struct{}) error { } } + if !slices.Contains(o.DisabledFilters, cache.Name) { + cacheSpec := cache.NewCacheFilter( + cache.Options{ + MaxBytes: o.cacheBudget(), + ListenAddr: o.Address, + NetOpts: skpnet.Options{ + IdleConnTimeout: o.CloseIdleConnsPeriod, + MaxIdleConnsPerHost: o.IdleConnectionsPerHost, + Tracer: tracer, + OpentracingComponentTag: "skipper", + OpentracingSpanName: "cache_revalidation", + OpentracingEventsByTag: o.OpenTracingClientTraceByTag, + }, + ValkeyRing: valkeyRing, + L1TTL: o.CacheL1TTL, + }, + ) + defer cacheSpec.(io.Closer).Close() + o.CustomFilters = append(o.CustomFilters, cacheSpec) + } + + if o.TLSMinVersion == 0 { o.TLSMinVersion = tls.VersionTLS12 } From f6c4f54e1eefd0d3bd76e680c357659e490714f9 Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Fri, 24 Jul 2026 16:29:14 +0200 Subject: [PATCH 44/89] fix: warm L1 on Valkey Get hit using remaining TTL Signed-off-by: Larry D Almeida --- filters/cache/valkey_storage.go | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/filters/cache/valkey_storage.go b/filters/cache/valkey_storage.go index f93caa779f..8042f412bf 100644 --- a/filters/cache/valkey_storage.go +++ b/filters/cache/valkey_storage.go @@ -68,6 +68,16 @@ func (s *ValkeyStorage) Get(ctx context.Context, key string) (*Entry, error) { if err := json.Unmarshal([]byte(data), &e); err != nil { return nil, fmt.Errorf("cache: decode valkey entry: %w", err) } + // Write-through: warm L1 so subsequent requests on this process avoid Valkey round-trips. + // Use remaining freshness to avoid extending L1 beyond Valkey's actual expiry. + if s.l1TTL > 0 && e.TTL > 0 { + if remaining := e.TTL - time.Since(e.CreatedAt); remaining > 0 { + warmed := e + warmed.TTL = min(s.l1TTL, remaining) + warmed.CreatedAt = time.Now() + _ = s.l1.Set(ctx, key, &warmed) + } + } return &e, nil } From e0f0925590c2b3796ea0b80c25c2437cd091b043 Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Fri, 24 Jul 2026 16:34:54 +0200 Subject: [PATCH 45/89] fix: add counter to track warming L1 on Valkey Get hit Signed-off-by: Larry D Almeida --- docs/operation/operation.md | 1 + filters/cache/valkey_storage.go | 1 + 2 files changed, 2 insertions(+) diff --git a/docs/operation/operation.md b/docs/operation/operation.md index c5a394bbc1..cecf4fa5ef 100644 --- a/docs/operation/operation.md +++ b/docs/operation/operation.md @@ -1959,3 +1959,4 @@ When Valkey is configured: - `l1_hit`: Counter, L1 hits that bypassed Valkey - `valkey_miss`: Counter, Valkey misses that proceeded to an upstream fetch - `valkey_get_fallback`, `valkey_set_fallback`: Counters, reads/writes that fell back to L1 due to Valkey errors +- `l1_warm_from_valkey`: Counter, entries written into L1 after a successful Valkey Get (write-through on read path) diff --git a/filters/cache/valkey_storage.go b/filters/cache/valkey_storage.go index 8042f412bf..3677c6dad4 100644 --- a/filters/cache/valkey_storage.go +++ b/filters/cache/valkey_storage.go @@ -76,6 +76,7 @@ func (s *ValkeyStorage) Get(ctx context.Context, key string) (*Entry, error) { warmed.TTL = min(s.l1TTL, remaining) warmed.CreatedAt = time.Now() _ = s.l1.Set(ctx, key, &warmed) + s.metrics.IncCounter("l1_warm_from_valkey") } } return &e, nil From f86c5eebc44d887550f7c5433b6ca34d1946c9fa Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Fri, 24 Jul 2026 16:44:03 +0200 Subject: [PATCH 46/89] feat: add reval_queue_depth and reval_wait_duration metrics Signed-off-by: Larry D Almeida --- docs/operation/operation.md | 2 ++ filters/cache/filter.go | 26 +++++++++++++++----------- 2 files changed, 17 insertions(+), 11 deletions(-) diff --git a/docs/operation/operation.md b/docs/operation/operation.md index cecf4fa5ef..57cbdc1ec5 100644 --- a/docs/operation/operation.md +++ b/docs/operation/operation.md @@ -1950,6 +1950,8 @@ remove the L1 entry unconditionally, regardless of `--cache-l1-ttl`. - `lru_eviction`: Counter, incremented each time an L1 entry is evicted due to memory pressure - `lru_bytes`: Gauge, current L1 usage in bytes - `lru_oversized`: Counter, incremented when an entry is too large for any shard and silently dropped +- `reval_queue_depth`: Gauge, current number of pending revalidation jobs in the queue (sampled every 10s) +- `reval_wait_duration`: Histogram, time a revalidation job spent waiting in the queue before the worker picked it up - `reval_dropped`: Counter, revalidation jobs dropped because the queue was full - `reval_error`: Counter, background revalidation fetch failures - `reval_duration`: Histogram, end-to-end duration of each background revalidation job diff --git a/filters/cache/filter.go b/filters/cache/filter.go index bb0e13c27c..042320d2f8 100644 --- a/filters/cache/filter.go +++ b/filters/cache/filter.go @@ -128,7 +128,7 @@ func NewCacheFilter(opts Options) filters.Spec { // Start shared background goroutines (one worker + one scraper for all filter instances) spec.bgWg.Add(2) go spec.revalidationWorker() - go spec.lruBytesScraper() + go spec.metricsScraper() return spec } @@ -238,6 +238,7 @@ func (s *cacheSpec) revalidationWorker() { defer s.bgWg.Done() for job := range s.revalJobs { if job.filter != nil { + s.metrics.MeasureSince("reval_wait_duration", job.enqueuedAt) start := time.Now() job.filter.doRevalidate(job.key, job.req, job.body) s.metrics.MeasureSince("reval_duration", start) @@ -248,10 +249,10 @@ func (s *cacheSpec) revalidationWorker() { const lruBytesScrapeInterval = 10 * time.Second -// lruBytesScraper periodically updates the lru_bytes gauge so it stays current +// metricsScraper periodically updates the lru_bytes gauge so it stays current // even when no evictions occur. It's spec-level and shared across all filter instances. // It exits when lruBytesDone is closed (via cacheSpec.Close). -func (s *cacheSpec) lruBytesScraper() { +func (s *cacheSpec) metricsScraper() { defer s.bgWg.Done() ticker := time.NewTicker(lruBytesScrapeInterval) defer ticker.Stop() @@ -259,6 +260,7 @@ func (s *cacheSpec) lruBytesScraper() { select { case <-ticker.C: s.metrics.UpdateGauge("lru_bytes", float64(s.lruStorage.lru.Bytes())) + s.metrics.UpdateGauge("reval_queue_depth", float64(len(s.revalJobs))) case <-s.lruBytesDone: return } @@ -266,10 +268,11 @@ func (s *cacheSpec) lruBytesScraper() { } type revalJob struct { - key string - req *http.Request // cloned via Request.Clone; Body is nil for GET/HEAD - body []byte // non-nil for QUERY: snapshot of the request body for revalidation - filter *cacheFilter // instance whose doRevalidate to call + key string + req *http.Request // cloned via Request.Clone; Body is nil for GET/HEAD + body []byte // non-nil for QUERY: snapshot of the request body for revalidation + filter *cacheFilter // instance whose doRevalidate to call + enqueuedAt time.Time // wall-clock time the job entered the queue; used to measure wait time } type cacheFilter struct { @@ -746,10 +749,11 @@ func (f *cacheFilter) enqueueRevalidation(key string, orig *http.Request) { orig.Body = io.NopCloser(bytes.NewReader(bodySnapshot)) } job := revalJob{ - key: key, - req: cloned, - body: bodySnapshot, - filter: f, + key: key, + req: cloned, + body: bodySnapshot, + filter: f, + enqueuedAt: time.Now(), } // recover guards against a send on a closed channel during the shutdown window // between cacheSpec.Close() closing revalJobs and this goroutine observing it. From 7e41678017fe3febf288d6c32ad893d588210aaa Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Fri, 24 Jul 2026 16:48:37 +0200 Subject: [PATCH 47/89] fix: handle QUERY body read error in enqueueRevalidation Signed-off-by: Larry D Almeida --- filters/cache/filter.go | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/filters/cache/filter.go b/filters/cache/filter.go index 042320d2f8..0f9b749930 100644 --- a/filters/cache/filter.go +++ b/filters/cache/filter.go @@ -745,7 +745,13 @@ func (f *cacheFilter) enqueueRevalidation(key string, orig *http.Request) { cloned := orig.Clone(context.Background()) var bodySnapshot []byte if orig.Method == "QUERY" && orig.Body != nil && orig.Body != http.NoBody { - bodySnapshot, _ = io.ReadAll(orig.Body) + var err error + bodySnapshot, err = io.ReadAll(orig.Body) + if err != nil { + log.WithError(err).Warn("cache: failed to read QUERY body for revalidation; dropping job") + f.metrics.IncCounter("reval_dropped") + return + } orig.Body = io.NopCloser(bytes.NewReader(bodySnapshot)) } job := revalJob{ From cbaf8b6e57fbe16a784b91326d8fbcdf6f7518fa Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Mon, 27 Jul 2026 09:45:53 +0200 Subject: [PATCH 48/89] fix: update docs/comments to clarify local only DELETE behavior Signed-off-by: Larry D Almeida --- docs/operation/operation.md | 5 ++++- filters/cache/valkey_storage.go | 2 ++ 2 files changed, 6 insertions(+), 1 deletion(-) diff --git a/docs/operation/operation.md b/docs/operation/operation.md index 57cbdc1ec5..b10c6d95a0 100644 --- a/docs/operation/operation.md +++ b/docs/operation/operation.md @@ -1938,7 +1938,10 @@ restore write-around behaviour (L1 used only when Valkey is unavailable; this applies to both read errors (`valkey_get_fallback`) and write errors (`valkey_set_fallback`)). Explicit deletes (unsafe methods or operator-initiated invalidation) always -remove the L1 entry unconditionally, regardless of `--cache-l1-ttl`. +remove the L1 entry unconditionally, regardless of `--cache-l1-ttl`. However, +only the local process's L1 is cleared — other Skipper processes in the fleet +retain their own L1 copies until `--cache-l1-ttl` expires. Set `--cache-l1-ttl` +accordingly to bound the stale window after an invalidation. !!! note An in-process LRU (L1) is shared across all `cache()` filter instances in the same process. diff --git a/filters/cache/valkey_storage.go b/filters/cache/valkey_storage.go index 3677c6dad4..74ecb88ed5 100644 --- a/filters/cache/valkey_storage.go +++ b/filters/cache/valkey_storage.go @@ -115,6 +115,8 @@ func (s *ValkeyStorage) Set(ctx context.Context, key string, entry *Entry) error func (s *ValkeyStorage) Delete(ctx context.Context, key string) error { // Valkey errors are best-effort — L1 delete always runs. + // Note: only the local process's L1 is cleared. Other Skipper processes in the + // fleet retain their own L1 copies until --cache-l1-ttl expires naturally. if _, err := s.ring.Del(ctx, key); err != nil { log.WithError(err).Warn("cache: valkey Delete failed") } From afe76a1e796c2ba172180966982379a7da727ff8 Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Tue, 28 Jul 2026 07:52:49 +0200 Subject: [PATCH 49/89] fix: return empty key on QUERY body read failure in cacheKey MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit If io.ReadAll fails, silently omitting the body from the hash causes QUERY requests with different bodies to collide on the same cache key. Returning "" signals Request() to bypass the cache lookup and Response() to skip storing — both callers already handle "" correctly. Signed-off-by: Larry D Almeida --- filters/cache/filter.go | 9 ++++++--- 1 file changed, 6 insertions(+), 3 deletions(-) diff --git a/filters/cache/filter.go b/filters/cache/filter.go index 0f9b749930..4559052d16 100644 --- a/filters/cache/filter.go +++ b/filters/cache/filter.go @@ -948,10 +948,13 @@ func cacheKey(routeID string, r *http.Request, keyHeaders []string) string { // queries to the same URL produce distinct cache entries. if r.Method == "QUERY" && r.Body != nil && r.Body != http.NoBody { body, err := io.ReadAll(r.Body) - if err == nil { - r.Body = io.NopCloser(bytes.NewReader(body)) - h.Write(body) + if err != nil { + // Body unreadable: key would omit the body and collide with other QUERY + // requests. Return "" so Request() bypasses the cache and Response() skips storing. + return "" } + r.Body = io.NopCloser(bytes.NewReader(body)) + h.Write(body) } return hex.EncodeToString(h.Sum(nil)) } From 956f8e0b0027d68bccb8692978e95517bd15d506 Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Fri, 7 Aug 2026 09:31:28 +0200 Subject: [PATCH 50/89] fix: add error return to cacheSpec.Close to satisfy io.Closer skipper.go defers cacheSpec.(io.Closer).Close() which requires Close() error. The missing return caused a runtime panic in TestInitOrderAndDefault. Added compile-time guard to prevent recurrence. Signed-off-by: Larry D Almeida --- filters/cache/filter.go | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/filters/cache/filter.go b/filters/cache/filter.go index 4559052d16..09ae75c400 100644 --- a/filters/cache/filter.go +++ b/filters/cache/filter.go @@ -70,6 +70,8 @@ type filterCacheKey struct { rfcMode bool } +var _ io.Closer = (*cacheSpec)(nil) + type cacheSpec struct { maxBytes int64 listenAddr string @@ -137,13 +139,14 @@ func (s *cacheSpec) Name() string { return filterName } // Close shuts down the background revalidation worker and lru_bytes scraper. // Safe to call multiple times. -func (s *cacheSpec) Close() { +func (s *cacheSpec) Close() error { s.closeOnce.Do(func() { close(s.lruBytesDone) // stop scraper; must close before revalJobs so enqueueRevalidation's recover() fires first close(s.revalJobs) // unblocks revalidationWorker range loop s.bgWg.Wait() s.client.Close() // tear down transport after all in-flight revalidation fetches complete }) + return nil } func (s *cacheSpec) CreateFilter(args []interface{}) (filters.Filter, error) { From c7f3fd40f302911bcf11307b64161845b77f2c69 Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Tue, 18 Aug 2026 08:46:32 +0200 Subject: [PATCH 51/89] fix: wrap cacheSpec.Close in t.Cleanup and remove extra blank line in skipper.go Signed-off-by: Larry D Almeida --- filters/cache/filter_test.go | 22 +++++++++++----------- skipper.go | 1 - 2 files changed, 11 insertions(+), 12 deletions(-) diff --git a/filters/cache/filter_test.go b/filters/cache/filter_test.go index b3fd401115..375d9eea07 100644 --- a/filters/cache/filter_test.go +++ b/filters/cache/filter_test.go @@ -40,7 +40,7 @@ func newTestFilter(t *testing.T, ttl, errorTTL, swrWindow time.Duration, extra . } t.Cleanup(cf.Close) t.Cleanup(spec.(*cacheSpec).client.Close) - t.Cleanup(spec.(*cacheSpec).Close) + t.Cleanup(func() { spec.(*cacheSpec).Close() }) return cf } @@ -62,7 +62,7 @@ func newTestFilterRFC(t *testing.T, _, _, _ time.Duration, _ ...time.Duration) * } t.Cleanup(cf.Close) t.Cleanup(spec.(*cacheSpec).client.Close) - t.Cleanup(spec.(*cacheSpec).Close) + t.Cleanup(func() { spec.(*cacheSpec).Close() }) return cf } @@ -140,7 +140,7 @@ func TestCacheFilter_KeyIsolationByAuthToken(t *testing.T) { f := fi.(*cacheFilter) t.Cleanup(f.Close) t.Cleanup(spec.(*cacheSpec).client.Close) - t.Cleanup(spec.(*cacheSpec).Close) + t.Cleanup(func() { spec.(*cacheSpec).Close() }) f.fetch = func(*http.Request) (*http.Response, error) { return nil, errors.New("no fetch stub set") } @@ -265,7 +265,7 @@ func TestCacheFilter_Response_NoopIfStateBagKeyMissing(t *testing.T) { func TestCreateFilter_InvalidArgs(t *testing.T) { spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: "localhost:9090", L1TTL: 60 * time.Second}) t.Cleanup(spec.(*cacheSpec).client.Close) - t.Cleanup(spec.(*cacheSpec).Close) + t.Cleanup(func() { spec.(*cacheSpec).Close() }) cases := []struct { name string args []interface{} @@ -1321,7 +1321,7 @@ func TestCacheFilter_MustRevalidate_ForcesCoalesceWhenStale(t *testing.T) { func TestCacheFilter_SharedStorage_RouteIsolation(t *testing.T) { spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: "localhost:9090", L1TTL: 60 * time.Second}) t.Cleanup(spec.(*cacheSpec).client.Close) - t.Cleanup(spec.(*cacheSpec).Close) + t.Cleanup(func() { spec.(*cacheSpec).Close() }) makeFilter := func(t *testing.T) *cacheFilter { t.Helper() @@ -2619,7 +2619,7 @@ func TestCacheFilter_SMaxAge_CapsRouteTTL(t *testing.T) { func TestCacheFilter_CreateFilter_RFCArgParsing(t *testing.T) { spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: ":9090", L1TTL: 60 * time.Second}) t.Cleanup(spec.(*cacheSpec).client.Close) - t.Cleanup(spec.(*cacheSpec).Close) + t.Cleanup(func() { spec.(*cacheSpec).Close() }) cases := []struct { name string @@ -2665,7 +2665,7 @@ func TestCacheFilter_PureRFCMode_ZeroArgs_UsesUpstreamMaxAge(t *testing.T) { // no operator ceiling. TTL should equal upstream max-age exactly. spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: ":9090", L1TTL: 60 * time.Second}) t.Cleanup(spec.(*cacheSpec).client.Close) - t.Cleanup(spec.(*cacheSpec).Close) + t.Cleanup(func() { spec.(*cacheSpec).Close() }) f, err := spec.CreateFilter([]interface{}{}) if err != nil { t.Fatalf("unexpected error: %v", err) @@ -2701,7 +2701,7 @@ func TestCacheFilter_LRUBytesGaugeUpdatesWithoutEviction(t *testing.T) { synctest.Test(t, func(t *testing.T) { spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: "localhost:9090", L1TTL: 60 * time.Second}) t.Cleanup(spec.(*cacheSpec).client.Close) - t.Cleanup(spec.(*cacheSpec).Close) + t.Cleanup(func() { spec.(*cacheSpec).Close() }) f, err := spec.CreateFilter([]interface{}{5 * time.Minute, 15 * time.Second, 5 * time.Minute}) if err != nil { t.Fatal(err) @@ -2746,7 +2746,7 @@ func TestCacheFilter_PureRFCMode_ZeroArgs_NoUpstreamDirective_NotCached(t *testi // and no Last-Modified, nothing should be cached (no heuristic without Last-Modified). spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: ":9090", L1TTL: 60 * time.Second}) t.Cleanup(spec.(*cacheSpec).client.Close) - t.Cleanup(spec.(*cacheSpec).Close) + t.Cleanup(func() { spec.(*cacheSpec).Close() }) f, err := spec.CreateFilter([]interface{}{}) if err != nil { t.Fatalf("unexpected error: %v", err) @@ -2864,7 +2864,7 @@ func TestCacheFilter_RevalDropped_WhenQueueFull(t *testing.T) { func TestCacheSpec_FilterRegistry(t *testing.T) { spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: "localhost:9090", L1TTL: 60 * time.Second}) t.Cleanup(spec.(*cacheSpec).client.Close) - t.Cleanup(spec.(*cacheSpec).Close) + t.Cleanup(func() { spec.(*cacheSpec).Close() }) // Same args — should return same *cacheFilter pointer (registry hit) f1, err := spec.CreateFilter([]interface{}{"5m", "15s", "30s"}) @@ -2920,7 +2920,7 @@ func TestCacheSpec_FilterRegistry_InFlightJobsSurviveRebuild(t *testing.T) { spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: "localhost:9090", L1TTL: 60 * time.Second}) t.Cleanup(spec.(*cacheSpec).client.Close) - t.Cleanup(spec.(*cacheSpec).Close) + t.Cleanup(func() { spec.(*cacheSpec).Close() }) // Create initial filter with blocking fetch stub (same pattern as TestCacheFilter_RevalDropped_WhenQueueFull). f1, err := spec.CreateFilter([]interface{}{"5m", "15s", "30s"}) diff --git a/skipper.go b/skipper.go index bca7603596..0adb7d8e5d 100644 --- a/skipper.go +++ b/skipper.go @@ -2329,7 +2329,6 @@ func run(o Options, sig chan os.Signal, idleConnsCH chan struct{}) error { o.CustomFilters = append(o.CustomFilters, cacheSpec) } - if o.TLSMinVersion == 0 { o.TLSMinVersion = tls.VersionTLS12 } From 1afc7051a7c4d1c7fc4619e56148d3e534754f37 Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Tue, 18 Aug 2026 08:55:54 +0200 Subject: [PATCH 52/89] fix: restore CacheL1TTL to Options struct and remove duplicate stale cache filter block Signed-off-by: Larry D Almeida --- skipper.go | 22 +++++----------------- 1 file changed, 5 insertions(+), 17 deletions(-) diff --git a/skipper.go b/skipper.go index 0adb7d8e5d..2b5b088ea5 100644 --- a/skipper.go +++ b/skipper.go @@ -148,6 +148,11 @@ type Options struct { // using a fixed 25% fraction. Set explicitly to override that behaviour. ResponseCacheMaxMemoryBytes int64 + // CacheL1TTL sets the maximum TTL for write-through L1 warming in ValkeyStorage. + // On a successful Valkey Set, L1 is warmed with min(CacheL1TTL, entry.TTL). + // Set to 0 to disable write-through (write-around behaviour). Default: 60s. + CacheL1TTL time.Duration + // ReadMemoryLimit, when set, is called by the cache() filter initialiser // to determine the container memory limit. Defaults to reading cgroup files. // Override in tests or on non-standard platforms. @@ -1964,23 +1969,6 @@ func run(o Options, sig chan os.Signal, idleConnsCH chan struct{}) error { }), ) - // cache() filter registered here (not in filterRegistry) so the resolved tracer - // and connection options from skipper.Options can be wired through. - if !slices.Contains(o.DisabledFilters, cache.Name) { - o.CustomFilters = append(o.CustomFilters, cache.NewCacheFilter( - o.cacheBudget(), - o.Address, - skpnet.Options{ - IdleConnTimeout: o.CloseIdleConnsPeriod, - MaxIdleConnsPerHost: o.IdleConnectionsPerHost, - Tracer: tracer, - OpentracingComponentTag: "skipper", - OpentracingSpanName: "cache_revalidation", - OpentracingEventsByTag: o.OpenTracingClientTraceByTag, - }, - )) - } - if o.OAuthTokeninfoURL != "" { tio := auth.TokeninfoOptions{ URL: o.OAuthTokeninfoURL, From ce0e84a538ab8940a6a9329fd43827b1497e34dd Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Tue, 18 Aug 2026 16:14:32 +0200 Subject: [PATCH 53/89] refactor: remove cache_key span tag and key param from tagSpan Signed-off-by: Larry D Almeida --- filters/cache/filter.go | 23 +++++++++++------------ 1 file changed, 11 insertions(+), 12 deletions(-) diff --git a/filters/cache/filter.go b/filters/cache/filter.go index 09ae75c400..62adc02199 100644 --- a/filters/cache/filter.go +++ b/filters/cache/filter.go @@ -51,9 +51,9 @@ const ( // Options configures the cache filter. type Options struct { - MaxBytes int64 - ListenAddr string // Skipper's own address; revalidation requests loop back through it so the full filter chain runs - NetOpts skpnet.Options + MaxBytes int64 // maximum number of bytes the in-process LRU (L1) is allowed to hold across all cached entries + ListenAddr string // Skipper's own address; revalidation requests loop back through it so the full filter chain runs + NetOpts skpnet.Options // HTTP client options for background worker that re-fetches stale entries from origin ValkeyRing *skpnet.ValkeyRingClient L1TTL time.Duration Metrics metrics.Metrics @@ -304,15 +304,14 @@ type cacheFilter struct { // destructive. Lifecycle is managed by cacheSpec.Close(). func (f *cacheFilter) Close() {} -// tagSpan sets cache_status, cache_key, and (when >= 0) cache_ttl_remaining_ms +// tagSpan sets cache_status and (when >= 0) cache_ttl_remaining_ms // on the active OpenTracing span. No-op when no span is present. -func tagSpan(ctx filters.FilterContext, status, key string, ttlRemainingMs int64) { +func tagSpan(ctx filters.FilterContext, status string, ttlRemainingMs int64) { span := opentracing.SpanFromContext(ctx.Request().Context()) if span == nil { return } span.SetTag("cache_status", status) - span.SetTag("cache_key", key) if ttlRemainingMs >= 0 { span.SetTag("cache_ttl_remaining_ms", ttlRemainingMs) } @@ -398,7 +397,7 @@ func (f *cacheFilter) Request(ctx filters.FilterContext) { } } rsp.Header.Set(cacheStatusHeader, cacheStatusStale) - tagSpan(ctx, cacheStatusStale, key, -1) + tagSpan(ctx, cacheStatusStale, -1) setAgeHeader(rsp, entry, now) ctx.Metrics().IncCounter("stale") method := ctx.Request().Method @@ -436,7 +435,7 @@ func (f *cacheFilter) Request(ctx filters.FilterContext) { } rsp.Header.Set(cacheStatusHeader, cacheStatusHit) - tagSpan(ctx, cacheStatusHit, key, max(0, entry.TTL-time.Since(entry.CreatedAt)).Milliseconds()) + tagSpan(ctx, cacheStatusHit, max(0, entry.TTL-time.Since(entry.CreatedAt)).Milliseconds()) setAgeHeader(rsp, entry, time.Now()) ctx.Metrics().IncCounter("hit") method := ctx.Request().Method @@ -562,7 +561,7 @@ func (f *cacheFilter) coalesce(ctx filters.FilterContext, key string) { Body: io.NopCloser(bytes.NewReader(cr.stored.Payload)), } staleRsp.Header.Set(cacheStatusHeader, cacheStatusStale) - tagSpan(ctx, cacheStatusStale, key, -1) + tagSpan(ctx, cacheStatusStale, -1) setAgeHeader(staleRsp, cr.stored, time.Now()) ctx.Serve(headBodyOmitted(ctx.Request().Method, staleRsp)) return @@ -574,7 +573,7 @@ func (f *cacheFilter) coalesce(ctx filters.FilterContext, key string) { Body: io.NopCloser(bytes.NewReader(entry.Payload)), } rsp.Header.Set(cacheStatusHeader, cacheStatusMiss) - tagSpan(ctx, cacheStatusMiss, key, -1) + tagSpan(ctx, cacheStatusMiss, -1) ctx.Metrics().IncCounter("miss") ctx.Serve(headBodyOmitted(ctx.Request().Method, rsp)) case <-ctx.Request().Context().Done(): @@ -647,13 +646,13 @@ func (f *cacheFilter) Response(ctx filters.FilterContext) { if ctx.StateBag()[stateBagNoStore] == true { rsp.Header.Set(cacheStatusHeader, cacheStatusMiss) - tagSpan(ctx, cacheStatusMiss, key, -1) + tagSpan(ctx, cacheStatusMiss, -1) ctx.Metrics().IncCounter("miss") return } rsp.Header.Set(cacheStatusHeader, cacheStatusMiss) - tagSpan(ctx, cacheStatusMiss, key, -1) + tagSpan(ctx, cacheStatusMiss, -1) ctx.Metrics().IncCounter("miss") // Vary: * means every response is unique — never cache. From 0b3a38248fb19c05fa43f02506f1c02ebe3eec4c Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Tue, 18 Aug 2026 16:14:36 +0200 Subject: [PATCH 54/89] docs: fix cache key description and add QUERY method support in filters.md Signed-off-by: Larry D Almeida --- docs/reference/filters.md | 9 ++++++++- 1 file changed, 8 insertions(+), 1 deletion(-) diff --git a/docs/reference/filters.md b/docs/reference/filters.md index a6de699ccd..bf0132259c 100644 --- a/docs/reference/filters.md +++ b/docs/reference/filters.md @@ -3983,6 +3983,10 @@ Stale-while-revalidate (SWR): a stale entry within the SWR window is served immediately while a background fetch refreshes the entry. Concurrent cold-miss requests for the same key are coalesced into a single upstream fetch. +The filter caches `GET`, `HEAD`, and `QUERY` requests. The `QUERY` method +([HTTPWG draft](https://www.ietf.org/archive/id/draft-ietf-httpbis-safe-method-w-body-05.txt)) +carries a request body; it is included in the cache key. + Unsafe methods (`POST`, `PUT`, `DELETE`, `PATCH`) invalidate the cached entry on success. `HEAD 200` freshens stored headers without replacing the body. @@ -4024,13 +4028,16 @@ Force mode with per-tenant cache key isolation: **Cache key** -The key is derived from route ID + HTTP method + host + path + query string, +The key is derived from route ID + scheme + host + path + query string, hashed with SHA-256 for uniform shard distribution. Route ID is included so entries from different routes never collide when sharing the same storage instance. Additional request headers can be folded in via `keyHeaders`. Without `keyHeaders`, all requests to the same path share one cache entry regardless of caller identity. +For `QUERY` requests the request body is also hashed into the key, so different +query bodies produce distinct cache entries. + **Safety** The filter stores whatever the upstream returns and serves it to any request From 3c4818c46e59bb3ec9d8ad8bb397ff77842339f5 Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Tue, 18 Aug 2026 16:14:40 +0200 Subject: [PATCH 55/89] =?UTF-8?q?docs:=20expand=20cache=20operation=20docs?= =?UTF-8?q?=20=E2=80=94=20metrics,=20span=20tags,=20memory=20budget,=20coa?= =?UTF-8?q?lescing=20scope,=20L1=20warming?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Signed-off-by: Larry D Almeida --- docs/operation/operation.md | 35 ++++++++++++++++++++++++++++++++--- 1 file changed, 32 insertions(+), 3 deletions(-) diff --git a/docs/operation/operation.md b/docs/operation/operation.md index b10c6d95a0..8bc0a6ec80 100644 --- a/docs/operation/operation.md +++ b/docs/operation/operation.md @@ -1931,7 +1931,10 @@ store (L2) accessible by all Skipper instances via a client-side consistent hash read checks L1 first; an L1 hit returns without contacting Valkey. On every successful Valkey write the entry is also written to L1 -(write-through) with a TTL of `min(--cache-l1-ttl, entry.TTL)`. The default is +(write-through) with a TTL of `min(--cache-l1-ttl, entry.TTL)`. On a Valkey +read hit, L1 is warmed with `min(--cache-l1-ttl, remaining freshness)` — +remaining freshness (`entry.TTL - age`) is used rather than the original TTL to +prevent L1 from serving the entry beyond Valkey's actual expiry. The default is 60 seconds, bounding how long Skipper serves a locally-cached entry before re-consulting Valkey. Set `--cache-l1-ttl=0` to disable L1 warming and restore write-around behaviour (L1 used only when Valkey is unavailable; this @@ -1943,6 +1946,15 @@ only the local process's L1 is cleared — other Skipper processes in the fleet retain their own L1 copies until `--cache-l1-ttl` expires. Set `--cache-l1-ttl` accordingly to bound the stale window after an invalidation. +Concurrent cold-miss requests for the same key within one Skipper process are +coalesced into a single upstream fetch (thundering-herd protection). This is +process-local: a fleet of N instances may still issue up to N simultaneous +origin requests on a cold miss. + +The L1 memory budget defaults to 25% of the container's cgroup memory limit, +falling back to 2 GB if the limit is unreadable. Override it programmatically +via `skipper.Options.ResponseCacheMaxMemoryBytes`. + !!! note An in-process LRU (L1) is shared across all `cache()` filter instances in the same process. The L1 storage budget is divided evenly across 256 internal shards; a single entry larger @@ -1950,18 +1962,35 @@ accordingly to bound the stale window after an invalidation. ### Metrics +**Cache outcomes (always active):** + +- `hit`: Counter, request served from cache without contacting the upstream +- `miss`: Counter, request not in cache; upstream was contacted +- `stale`: Counter, stale entry served while background revalidation was enqueued +- `coalesce_error`: Counter, singleflight cold-miss fetch returned an error + +**LRU (always active):** + - `lru_eviction`: Counter, incremented each time an L1 entry is evicted due to memory pressure - `lru_bytes`: Gauge, current L1 usage in bytes - `lru_oversized`: Counter, incremented when an entry is too large for any shard and silently dropped + +**Revalidation (always active):** + - `reval_queue_depth`: Gauge, current number of pending revalidation jobs in the queue (sampled every 10s) - `reval_wait_duration`: Histogram, time a revalidation job spent waiting in the queue before the worker picked it up -- `reval_dropped`: Counter, revalidation jobs dropped because the queue was full +- `reval_dropped`: Counter, revalidation jobs dropped because the queue was full or body read failed - `reval_error`: Counter, background revalidation fetch failures - `reval_duration`: Histogram, end-to-end duration of each background revalidation job -When Valkey is configured: +**Valkey (when Valkey is configured):** - `l1_hit`: Counter, L1 hits that bypassed Valkey - `valkey_miss`: Counter, Valkey misses that proceeded to an upstream fetch - `valkey_get_fallback`, `valkey_set_fallback`: Counters, reads/writes that fell back to L1 due to Valkey errors - `l1_warm_from_valkey`: Counter, entries written into L1 after a successful Valkey Get (write-through on read path) + +**OpenTracing span tags (set on every request when a span is active):** + +- `cache_status`: `"hit"`, `"miss"`, or `"stale"` +- `cache_ttl_remaining_ms`: remaining freshness in milliseconds (only set on hits) From 9046bc38e991cff098b5fe56410de4a6c5ab21a7 Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Tue, 18 Aug 2026 16:32:27 +0200 Subject: [PATCH 56/89] docs: add inline comments to cache Options fields Signed-off-by: Larry D Almeida --- filters/cache/filter.go | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/filters/cache/filter.go b/filters/cache/filter.go index 62adc02199..e09f886281 100644 --- a/filters/cache/filter.go +++ b/filters/cache/filter.go @@ -54,9 +54,9 @@ type Options struct { MaxBytes int64 // maximum number of bytes the in-process LRU (L1) is allowed to hold across all cached entries ListenAddr string // Skipper's own address; revalidation requests loop back through it so the full filter chain runs NetOpts skpnet.Options // HTTP client options for background worker that re-fetches stale entries from origin - ValkeyRing *skpnet.ValkeyRingClient - L1TTL time.Duration - Metrics metrics.Metrics + ValkeyRing *skpnet.ValkeyRingClient // optional L2 cache; nil = in-process LRU only + L1TTL time.Duration // max TTL for write-through L1 warming; 0 = write-around + Metrics metrics.Metrics // nil defaults to metrics.Default } // filterCacheKey identifies a unique cache filter configuration for registry lookup. From 76c947d7347f7c1a69e176337887e3cd36690adc Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Wed, 19 Aug 2026 06:10:04 +0200 Subject: [PATCH 57/89] refactor: replace recover() with context-based shutdown in cache filter Signed-off-by: Larry D Almeida --- filters/cache/filter.go | 91 ++++++++++++++++++++--------------------- 1 file changed, 45 insertions(+), 46 deletions(-) diff --git a/filters/cache/filter.go b/filters/cache/filter.go index e09f886281..9ac57cba3d 100644 --- a/filters/cache/filter.go +++ b/filters/cache/filter.go @@ -73,18 +73,19 @@ type filterCacheKey struct { var _ io.Closer = (*cacheSpec)(nil) type cacheSpec struct { - maxBytes int64 - listenAddr string - client *skpnet.Client - storage Storage // shared across all filter instances - lruStorage *LRUStorage // always non-nil; direct reference to L1, even when storage is ValkeyStorage - metrics metrics.Metrics - revalJobs chan revalJob - lruBytesDone chan struct{} - bgWg sync.WaitGroup - closeOnce sync.Once - muFilter sync.Mutex - filters map[filterCacheKey]*cacheFilter + maxBytes int64 + listenAddr string + client *skpnet.Client + storage Storage // shared across all filter instances + lruStorage *LRUStorage // always non-nil; direct reference to L1, even when storage is ValkeyStorage + metrics metrics.Metrics + revalJobs chan revalJob + ctx context.Context + cancel context.CancelFunc + bgWg sync.WaitGroup + closeOnce sync.Once + muFilter sync.Mutex + filters map[filterCacheKey]*cacheFilter } // NewCacheFilter returns a Spec for the cache() filter. @@ -115,16 +116,18 @@ func NewCacheFilter(opts Options) filters.Spec { store = NewValkeyStorage(opts.ValkeyRing, lru, m, opts.L1TTL) } + ctx, cancel := context.WithCancel(context.Background()) spec := &cacheSpec{ - maxBytes: opts.MaxBytes, - listenAddr: opts.ListenAddr, - client: skpnet.NewClient(opts.NetOpts), - storage: store, - lruStorage: lru, - metrics: m, - revalJobs: make(chan revalJob, revalQueueSize), - lruBytesDone: make(chan struct{}), - filters: make(map[filterCacheKey]*cacheFilter), + maxBytes: opts.MaxBytes, + listenAddr: opts.ListenAddr, + client: skpnet.NewClient(opts.NetOpts), + storage: store, + lruStorage: lru, + metrics: m, + revalJobs: make(chan revalJob, revalQueueSize), + ctx: ctx, + cancel: cancel, + filters: make(map[filterCacheKey]*cacheFilter), } // Start shared background goroutines (one worker + one scraper for all filter instances) @@ -137,12 +140,11 @@ func NewCacheFilter(opts Options) filters.Spec { func (s *cacheSpec) Name() string { return filterName } -// Close shuts down the background revalidation worker and lru_bytes scraper. +// Close shuts down the background revalidation worker and metrics scraper. // Safe to call multiple times. func (s *cacheSpec) Close() error { s.closeOnce.Do(func() { - close(s.lruBytesDone) // stop scraper; must close before revalJobs so enqueueRevalidation's recover() fires first - close(s.revalJobs) // unblocks revalidationWorker range loop + s.cancel() // signals both goroutines to stop via ctx.Done() s.bgWg.Wait() s.client.Close() // tear down transport after all in-flight revalidation fetches complete }) @@ -225,8 +227,8 @@ func (s *cacheSpec) CreateFilter(args []interface{}) (filters.Filter, error) { rfcMode: rfcMode, metrics: s.metrics, keyHeaders: keyHeaders, - revalJobs: s.revalJobs, // use spec-level shared channel - lruBytesDone: s.lruBytesDone, // use spec-level shared signal + revalJobs: s.revalJobs, // use spec-level shared channel + ctx: s.ctx, // cancelled when spec shuts down } cf.fetch = s.client.Do @@ -239,22 +241,25 @@ func (s *cacheSpec) CreateFilter(args []interface{}) (filters.Filter, error) { // the per-instance doRevalidate method to respect each route's configuration. func (s *cacheSpec) revalidationWorker() { defer s.bgWg.Done() - for job := range s.revalJobs { - if job.filter != nil { - s.metrics.MeasureSince("reval_wait_duration", job.enqueuedAt) - start := time.Now() - job.filter.doRevalidate(job.key, job.req, job.body) - s.metrics.MeasureSince("reval_duration", start) + for { + select { + case job := <-s.revalJobs: + if job.filter != nil { + s.metrics.MeasureSince("reval_wait_duration", job.enqueuedAt) + start := time.Now() + job.filter.doRevalidate(job.key, job.req, job.body) + s.metrics.MeasureSince("reval_duration", start) + } + case <-s.ctx.Done(): + return } } - log.Debug("cache: revalidation worker stopped") } const lruBytesScrapeInterval = 10 * time.Second // metricsScraper periodically updates the lru_bytes gauge so it stays current // even when no evictions occur. It's spec-level and shared across all filter instances. -// It exits when lruBytesDone is closed (via cacheSpec.Close). func (s *cacheSpec) metricsScraper() { defer s.bgWg.Done() ticker := time.NewTicker(lruBytesScrapeInterval) @@ -264,7 +269,7 @@ func (s *cacheSpec) metricsScraper() { case <-ticker.C: s.metrics.UpdateGauge("lru_bytes", float64(s.lruStorage.lru.Bytes())) s.metrics.UpdateGauge("reval_queue_depth", float64(len(s.revalJobs))) - case <-s.lruBytesDone: + case <-s.ctx.Done(): return } } @@ -293,9 +298,9 @@ type cacheFilter struct { rfcMode bool coldSF singleflight.Group // cold-miss coalescing revalSF singleflight.Group // coalesces concurrent background revalidations per key - revalJobs chan revalJob // shared background revalidation queue from cacheSpec - lruBytesDone chan struct{} // shared channel from cacheSpec; closed to stop scraper - fetch func(*http.Request) (*http.Response, error) + revalJobs chan revalJob // shared background revalidation queue from cacheSpec + ctx context.Context // cancelled when cacheSpec shuts down + fetch func(*http.Request) (*http.Response, error) metrics metrics.Metrics } @@ -763,16 +768,10 @@ func (f *cacheFilter) enqueueRevalidation(key string, orig *http.Request) { filter: f, enqueuedAt: time.Now(), } - // recover guards against a send on a closed channel during the shutdown window - // between cacheSpec.Close() closing revalJobs and this goroutine observing it. - // The job is dropped, which is safe — the same outcome as the default (full buffer) path. - defer func() { - if recover() != nil { - f.metrics.IncCounter("reval_dropped") - } - }() select { case f.revalJobs <- job: + case <-f.ctx.Done(): + f.metrics.IncCounter("reval_dropped") default: f.metrics.IncCounter("reval_dropped") } From 34d00d1d4504efaf837c6093746d9735e7396c56 Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Wed, 19 Aug 2026 06:19:09 +0200 Subject: [PATCH 58/89] style: gofmt alignment fix in cacheFilter struct after context refactor Signed-off-by: Larry D Almeida --- filters/cache/filter.go | 22 +++++++++++----------- 1 file changed, 11 insertions(+), 11 deletions(-) diff --git a/filters/cache/filter.go b/filters/cache/filter.go index 9ac57cba3d..c5673df180 100644 --- a/filters/cache/filter.go +++ b/filters/cache/filter.go @@ -51,9 +51,9 @@ const ( // Options configures the cache filter. type Options struct { - MaxBytes int64 // maximum number of bytes the in-process LRU (L1) is allowed to hold across all cached entries - ListenAddr string // Skipper's own address; revalidation requests loop back through it so the full filter chain runs - NetOpts skpnet.Options // HTTP client options for background worker that re-fetches stale entries from origin + MaxBytes int64 // maximum number of bytes the in-process LRU (L1) is allowed to hold across all cached entries + ListenAddr string // Skipper's own address; revalidation requests loop back through it so the full filter chain runs + NetOpts skpnet.Options // HTTP client options for background worker that re-fetches stale entries from origin ValkeyRing *skpnet.ValkeyRingClient // optional L2 cache; nil = in-process LRU only L1TTL time.Duration // max TTL for write-through L1 warming; 0 = write-around Metrics metrics.Metrics // nil defaults to metrics.Default @@ -144,7 +144,7 @@ func (s *cacheSpec) Name() string { return filterName } // Safe to call multiple times. func (s *cacheSpec) Close() error { s.closeOnce.Do(func() { - s.cancel() // signals both goroutines to stop via ctx.Done() + s.cancel() // signals both goroutines to stop via ctx.Done() s.bgWg.Wait() s.client.Close() // tear down transport after all in-flight revalidation fetches complete }) @@ -227,8 +227,8 @@ func (s *cacheSpec) CreateFilter(args []interface{}) (filters.Filter, error) { rfcMode: rfcMode, metrics: s.metrics, keyHeaders: keyHeaders, - revalJobs: s.revalJobs, // use spec-level shared channel - ctx: s.ctx, // cancelled when spec shuts down + revalJobs: s.revalJobs, // use spec-level shared channel + ctx: s.ctx, // cancelled when spec shuts down } cf.fetch = s.client.Do @@ -295,13 +295,13 @@ type cacheFilter struct { // rfcMode true: upstream Cache-Control is authoritative (cache()). // false: operator ttl/errorTTL/swrWindow are authoritative (force mode). - rfcMode bool - coldSF singleflight.Group // cold-miss coalescing - revalSF singleflight.Group // coalesces concurrent background revalidations per key + rfcMode bool + coldSF singleflight.Group // cold-miss coalescing + revalSF singleflight.Group // coalesces concurrent background revalidations per key revalJobs chan revalJob // shared background revalidation queue from cacheSpec - ctx context.Context // cancelled when cacheSpec shuts down + ctx context.Context // cancelled when cacheSpec shuts down fetch func(*http.Request) (*http.Response, error) - metrics metrics.Metrics + metrics metrics.Metrics } // Close is intentionally a no-op. The routing layer calls Close() on every From c8ac0bb6edb63bc2587f839e3ae09c8d5353037a Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Wed, 19 Aug 2026 06:47:18 +0200 Subject: [PATCH 59/89] =?UTF-8?q?refactor:=20remove=20QUERY=20method=20sup?= =?UTF-8?q?port=20=E2=80=94=20to=20be=20reintroduced=20in=20feat/cache-que?= =?UTF-8?q?ry-method?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Signed-off-by: Larry D Almeida --- docs/reference/filters.md | 7 ------- filters/cache/filter.go | 43 ++++++--------------------------------- 2 files changed, 6 insertions(+), 44 deletions(-) diff --git a/docs/reference/filters.md b/docs/reference/filters.md index bf0132259c..e79bd0ecd8 100644 --- a/docs/reference/filters.md +++ b/docs/reference/filters.md @@ -3983,10 +3983,6 @@ Stale-while-revalidate (SWR): a stale entry within the SWR window is served immediately while a background fetch refreshes the entry. Concurrent cold-miss requests for the same key are coalesced into a single upstream fetch. -The filter caches `GET`, `HEAD`, and `QUERY` requests. The `QUERY` method -([HTTPWG draft](https://www.ietf.org/archive/id/draft-ietf-httpbis-safe-method-w-body-05.txt)) -carries a request body; it is included in the cache key. - Unsafe methods (`POST`, `PUT`, `DELETE`, `PATCH`) invalidate the cached entry on success. `HEAD 200` freshens stored headers without replacing the body. @@ -4035,9 +4031,6 @@ the same storage instance. Additional request headers can be folded in via `keyHeaders`. Without `keyHeaders`, all requests to the same path share one cache entry regardless of caller identity. -For `QUERY` requests the request body is also hashed into the key, so different -query bodies produce distinct cache entries. - **Safety** The filter stores whatever the upstream returns and serves it to any request diff --git a/filters/cache/filter.go b/filters/cache/filter.go index c5673df180..e7d4f542c6 100644 --- a/filters/cache/filter.go +++ b/filters/cache/filter.go @@ -247,7 +247,7 @@ func (s *cacheSpec) revalidationWorker() { if job.filter != nil { s.metrics.MeasureSince("reval_wait_duration", job.enqueuedAt) start := time.Now() - job.filter.doRevalidate(job.key, job.req, job.body) + job.filter.doRevalidate(job.key, job.req) s.metrics.MeasureSince("reval_duration", start) } case <-s.ctx.Done(): @@ -277,8 +277,7 @@ func (s *cacheSpec) metricsScraper() { type revalJob struct { key string - req *http.Request // cloned via Request.Clone; Body is nil for GET/HEAD - body []byte // non-nil for QUERY: snapshot of the request body for revalidation + req *http.Request // cloned via Request.Clone filter *cacheFilter // instance whose doRevalidate to call enqueuedAt time.Time // wall-clock time the job entered the queue; used to measure wait time } @@ -750,21 +749,9 @@ func (f *cacheFilter) Response(ctx filters.FilterContext) { // The closure captures f.doRevalidate so the spec-level worker respects this route's config. func (f *cacheFilter) enqueueRevalidation(key string, orig *http.Request) { cloned := orig.Clone(context.Background()) - var bodySnapshot []byte - if orig.Method == "QUERY" && orig.Body != nil && orig.Body != http.NoBody { - var err error - bodySnapshot, err = io.ReadAll(orig.Body) - if err != nil { - log.WithError(err).Warn("cache: failed to read QUERY body for revalidation; dropping job") - f.metrics.IncCounter("reval_dropped") - return - } - orig.Body = io.NopCloser(bytes.NewReader(bodySnapshot)) - } job := revalJob{ key: key, req: cloned, - body: bodySnapshot, filter: f, enqueuedAt: time.Now(), } @@ -780,16 +767,12 @@ func (f *cacheFilter) enqueueRevalidation(key string, orig *http.Request) { // doRevalidate revalidates key against the upstream. It sends a conditional // request (If-None-Match / If-Modified-Since) when the stored entry carries // validators; a 304 response reuses the stored payload and merges new headers. -func (f *cacheFilter) doRevalidate(key string, req *http.Request, body []byte) { +func (f *cacheFilter) doRevalidate(key string, req *http.Request) { f.revalSF.Do(key, func() (interface{}, error) { //nolint:errcheck req.Header.Set(revalidateHeader, "1") req.URL.Scheme = "http" req.URL.Host = f.listenAddr req.RequestURI = "" - if len(body) > 0 { - req.Body = io.NopCloser(bytes.NewReader(body)) - req.ContentLength = int64(len(body)) - } if stored, err := f.storage.Get(context.Background(), key); err == nil && stored != nil { if stored.ETag != "" { @@ -945,18 +928,6 @@ func cacheKey(routeID string, r *http.Request, keyHeaders []string) string { for _, name := range keyHeaders { fmt.Fprintf(h, "\n%s: %s", name, r.Header.Get(name)) } - // QUERY carries semantics in its body; include it in the key so different - // queries to the same URL produce distinct cache entries. - if r.Method == "QUERY" && r.Body != nil && r.Body != http.NoBody { - body, err := io.ReadAll(r.Body) - if err != nil { - // Body unreadable: key would omit the body and collide with other QUERY - // requests. Return "" so Request() bypasses the cache and Response() skips storing. - return "" - } - r.Body = io.NopCloser(bytes.NewReader(body)) - h.Write(body) - } return hex.EncodeToString(h.Sum(nil)) } @@ -1037,7 +1008,7 @@ func setAgeHeader(rsp *http.Response, entry *Entry, now time.Time) { // evaluateConditionals checks client If-None-Match / If-Modified-Since against // a cached entry per RFC 9111 §4.3.2 / RFC 9110 §13. Returns true when the // client condition is "not modified" (cache should respond 304). -// Only call for cacheable methods (GET, HEAD, QUERY). +// Only call for cacheable methods (GET, HEAD). func evaluateConditionals(req *http.Request, entry *Entry) bool { if inm := req.Header.Get("If-None-Match"); inm != "" { return matchesETag(inm, entry.ETag) @@ -1127,11 +1098,9 @@ func capTTLByExpires(ttl time.Duration, header http.Header, d cacheDirectives) t return ttl } -// isCacheableMethod reports whether method may be served from cache. -// GET and HEAD are defined as cacheable by RFC 9111. QUERY is a safe method -// with a request body (HTTPWG draft-ietf-httpbis-safe-method-w-body). +// isCacheableMethod reports whether method may be served from cache per RFC 9111. func isCacheableMethod(method string) bool { - return method == http.MethodGet || method == http.MethodHead || method == "QUERY" + return method == http.MethodGet || method == http.MethodHead } func isUnsafeMethod(method string) bool { From 0d9b8a1c8d4f92f89ff2a775121af54f3ba39032 Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Wed, 19 Aug 2026 11:56:47 +0200 Subject: [PATCH 60/89] refactor: add cache. prefix to all metrics and route hit/miss/stale/coalesce_error through f.metrics Signed-off-by: Larry D Almeida --- docs/operation/operation.md | 32 ++++++------- filters/cache/filter.go | 30 ++++++------ filters/cache/filter_test.go | 69 +++++++++++++++++----------- filters/cache/lru_storage.go | 2 +- filters/cache/lru_test.go | 2 +- filters/cache/valkey_storage.go | 10 ++-- filters/cache/valkey_storage_test.go | 38 +++++++-------- 7 files changed, 99 insertions(+), 84 deletions(-) diff --git a/docs/operation/operation.md b/docs/operation/operation.md index 8bc0a6ec80..72d71dfbbc 100644 --- a/docs/operation/operation.md +++ b/docs/operation/operation.md @@ -1964,31 +1964,31 @@ via `skipper.Options.ResponseCacheMaxMemoryBytes`. **Cache outcomes (always active):** -- `hit`: Counter, request served from cache without contacting the upstream -- `miss`: Counter, request not in cache; upstream was contacted -- `stale`: Counter, stale entry served while background revalidation was enqueued -- `coalesce_error`: Counter, singleflight cold-miss fetch returned an error +- `cache.hit`: Counter, request served from cache without contacting the upstream +- `cache.miss`: Counter, request not in cache; upstream was contacted +- `cache.stale`: Counter, stale entry served while background revalidation was enqueued +- `cache.coalesce_error`: Counter, singleflight cold-miss fetch returned an error **LRU (always active):** -- `lru_eviction`: Counter, incremented each time an L1 entry is evicted due to memory pressure -- `lru_bytes`: Gauge, current L1 usage in bytes -- `lru_oversized`: Counter, incremented when an entry is too large for any shard and silently dropped +- `cache.lru_eviction`: Counter, incremented each time an L1 entry is evicted due to memory pressure +- `cache.lru_bytes`: Gauge, current L1 usage in bytes +- `cache.lru_oversized`: Counter, incremented when an entry is too large for any shard and silently dropped **Revalidation (always active):** -- `reval_queue_depth`: Gauge, current number of pending revalidation jobs in the queue (sampled every 10s) -- `reval_wait_duration`: Histogram, time a revalidation job spent waiting in the queue before the worker picked it up -- `reval_dropped`: Counter, revalidation jobs dropped because the queue was full or body read failed -- `reval_error`: Counter, background revalidation fetch failures -- `reval_duration`: Histogram, end-to-end duration of each background revalidation job +- `cache.reval_queue_depth`: Gauge, current number of pending revalidation jobs in the queue (sampled every 10s) +- `cache.reval_wait_duration`: Histogram, time a revalidation job spent waiting in the queue before the worker picked it up +- `cache.reval_dropped`: Counter, revalidation jobs dropped because the queue was full or body read failed +- `cache.reval_error`: Counter, background revalidation fetch failures +- `cache.reval_duration`: Histogram, end-to-end duration of each background revalidation job **Valkey (when Valkey is configured):** -- `l1_hit`: Counter, L1 hits that bypassed Valkey -- `valkey_miss`: Counter, Valkey misses that proceeded to an upstream fetch -- `valkey_get_fallback`, `valkey_set_fallback`: Counters, reads/writes that fell back to L1 due to Valkey errors -- `l1_warm_from_valkey`: Counter, entries written into L1 after a successful Valkey Get (write-through on read path) +- `cache.l1_hit`: Counter, L1 hits that bypassed Valkey +- `cache.valkey_miss`: Counter, Valkey misses that proceeded to an upstream fetch +- `cache.valkey_get_fallback`, `cache.valkey_set_fallback`: Counters, reads/writes that fell back to L1 due to Valkey errors +- `cache.l1_warm_from_valkey`: Counter, entries written into L1 after a successful Valkey Get (write-through on read path) **OpenTracing span tags (set on every request when a span is active):** diff --git a/filters/cache/filter.go b/filters/cache/filter.go index e7d4f542c6..63b5344d54 100644 --- a/filters/cache/filter.go +++ b/filters/cache/filter.go @@ -108,7 +108,7 @@ func NewCacheFilter(opts Options) filters.Spec { m := opts.Metrics lru := NewLRUStorage(opts.MaxBytes, func() { - m.IncCounter("lru_eviction") + m.IncCounter("cache.lru_eviction") }, m) var store Storage = lru @@ -245,10 +245,10 @@ func (s *cacheSpec) revalidationWorker() { select { case job := <-s.revalJobs: if job.filter != nil { - s.metrics.MeasureSince("reval_wait_duration", job.enqueuedAt) + s.metrics.MeasureSince("cache.reval_wait_duration", job.enqueuedAt) start := time.Now() job.filter.doRevalidate(job.key, job.req) - s.metrics.MeasureSince("reval_duration", start) + s.metrics.MeasureSince("cache.reval_duration", start) } case <-s.ctx.Done(): return @@ -267,8 +267,8 @@ func (s *cacheSpec) metricsScraper() { for { select { case <-ticker.C: - s.metrics.UpdateGauge("lru_bytes", float64(s.lruStorage.lru.Bytes())) - s.metrics.UpdateGauge("reval_queue_depth", float64(len(s.revalJobs))) + s.metrics.UpdateGauge("cache.lru_bytes", float64(s.lruStorage.lru.Bytes())) + s.metrics.UpdateGauge("cache.reval_queue_depth", float64(len(s.revalJobs))) case <-s.ctx.Done(): return } @@ -403,7 +403,7 @@ func (f *cacheFilter) Request(ctx filters.FilterContext) { rsp.Header.Set(cacheStatusHeader, cacheStatusStale) tagSpan(ctx, cacheStatusStale, -1) setAgeHeader(rsp, entry, now) - ctx.Metrics().IncCounter("stale") + f.metrics.IncCounter("cache.stale") method := ctx.Request().Method if isCacheableMethod(method) && evaluateConditionals(ctx.Request(), entry) { notModified := &http.Response{ @@ -441,7 +441,7 @@ func (f *cacheFilter) Request(ctx filters.FilterContext) { rsp.Header.Set(cacheStatusHeader, cacheStatusHit) tagSpan(ctx, cacheStatusHit, max(0, entry.TTL-time.Since(entry.CreatedAt)).Milliseconds()) setAgeHeader(rsp, entry, time.Now()) - ctx.Metrics().IncCounter("hit") + f.metrics.IncCounter("cache.hit") method := ctx.Request().Method if (method == http.MethodGet || method == http.MethodHead) && evaluateConditionals(ctx.Request(), entry) { notModified := &http.Response{ @@ -549,7 +549,7 @@ func (f *cacheFilter) coalesce(ctx filters.FilterContext, key string) { select { case res := <-ch: if res.Err != nil || res.Val == nil { - ctx.Metrics().IncCounter("coalesce_error") + f.metrics.IncCounter("cache.coalesce_error") return } cr := res.Val.(*coalesceResult) @@ -578,7 +578,7 @@ func (f *cacheFilter) coalesce(ctx filters.FilterContext, key string) { } rsp.Header.Set(cacheStatusHeader, cacheStatusMiss) tagSpan(ctx, cacheStatusMiss, -1) - ctx.Metrics().IncCounter("miss") + f.metrics.IncCounter("cache.miss") ctx.Serve(headBodyOmitted(ctx.Request().Method, rsp)) case <-ctx.Request().Context().Done(): // Client disconnected. Remaining waiters still receive their result. @@ -651,13 +651,13 @@ func (f *cacheFilter) Response(ctx filters.FilterContext) { if ctx.StateBag()[stateBagNoStore] == true { rsp.Header.Set(cacheStatusHeader, cacheStatusMiss) tagSpan(ctx, cacheStatusMiss, -1) - ctx.Metrics().IncCounter("miss") + f.metrics.IncCounter("cache.miss") return } rsp.Header.Set(cacheStatusHeader, cacheStatusMiss) tagSpan(ctx, cacheStatusMiss, -1) - ctx.Metrics().IncCounter("miss") + f.metrics.IncCounter("cache.miss") // Vary: * means every response is unique — never cache. varyHeader := rsp.Header.Get("Vary") @@ -758,9 +758,9 @@ func (f *cacheFilter) enqueueRevalidation(key string, orig *http.Request) { select { case f.revalJobs <- job: case <-f.ctx.Done(): - f.metrics.IncCounter("reval_dropped") + f.metrics.IncCounter("cache.reval_dropped") default: - f.metrics.IncCounter("reval_dropped") + f.metrics.IncCounter("cache.reval_dropped") } } @@ -786,7 +786,7 @@ func (f *cacheFilter) doRevalidate(key string, req *http.Request) { requestTime := time.Now() resp, err := f.fetch(req) if err != nil { - f.metrics.IncCounter("reval_error") + f.metrics.IncCounter("cache.reval_error") log.WithFields(log.Fields{ "url": req.URL.String(), }).WithError(err).Warn("cache: background revalidation fetch failed") @@ -818,7 +818,7 @@ func (f *cacheFilter) doRevalidate(key string, req *http.Request) { var rerr error body, rerr = io.ReadAll(resp.Body) if rerr != nil { - f.metrics.IncCounter("reval_error") + f.metrics.IncCounter("cache.reval_error") log.WithFields(log.Fields{ "url": req.URL.String(), }).WithError(rerr).Warn("cache: failed to read response body during background revalidation") diff --git a/filters/cache/filter_test.go b/filters/cache/filter_test.go index 375d9eea07..3b1bc6d4c8 100644 --- a/filters/cache/filter_test.go +++ b/filters/cache/filter_test.go @@ -505,19 +505,24 @@ func TestCacheFilter_ColdMissCoalescing_UpstreamError(t *testing.T) { func TestCacheFilter_ColdMissCoalescing_FetchError_CoalesceErrorMetric(t *testing.T) { // coalesce_error must be incremented when the upstream fetch fails during coalescing. - f := newTestFilter(t, time.Minute, 15*time.Second, time.Minute) + mockMetrics := &metricstest.MockMetrics{} + spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: "localhost:9090", L1TTL: 60 * time.Second, Metrics: mockMetrics}) + fi, err := spec.CreateFilter([]interface{}{"1m", "15s", "1m"}) + if err != nil { + t.Fatal(err) + } + f := fi.(*cacheFilter) + t.Cleanup(func() { spec.(*cacheSpec).Close() }) f.fetch = func(*http.Request) (*http.Response, error) { return nil, errors.New("upstream unavailable") } - mockMetrics := &metricstest.MockMetrics{} ctx := newCtx("GET", "https://cdn.contentful.com/spaces/abc/entries/coalesce-err", "") - ctx.FMetrics = mockMetrics f.Request(ctx) mockMetrics.WithCounters(func(counters map[string]int64) { - if counters["coalesce_error"] != 1 { - t.Errorf("expected coalesce_error==1, got %d", counters["coalesce_error"]) + if counters["cache.coalesce_error"] != 1 { + t.Errorf("expected cache.coalesce_error==1, got %d", counters["cache.coalesce_error"]) } }) } @@ -706,7 +711,16 @@ func TestCacheFilter_Metrics(t *testing.T) { // ttl=1ms, swrWindow=1h — entry expires quickly, SWR window is huge. // Filter created outside the bubble so sknet.Client's transport goroutine // does not get trapped inside the synctest bubble. - f := newTestFilter(t, time.Millisecond, 15*time.Second, time.Hour) + // Metrics passed via Options so f.metrics captures hit/miss/stale counters. + mockMetrics := &metricstest.MockMetrics{} + spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: "localhost:9090", L1TTL: 60 * time.Second, Metrics: mockMetrics}) + fi, err := spec.CreateFilter([]interface{}{time.Millisecond.String(), (15 * time.Second).String(), time.Hour.String()}) + if err != nil { + t.Fatal(err) + } + f := fi.(*cacheFilter) + f.fetch = func(*http.Request) (*http.Response, error) { return nil, errors.New("no fetch stub set") } + t.Cleanup(func() { spec.(*cacheSpec).Close() }) url := "https://cdn.contentful.com/spaces/abc/entries/metrics" synctest.Test(t, func(t *testing.T) { @@ -717,12 +731,12 @@ func TestCacheFilter_Metrics(t *testing.T) { miss.FResponse = upstreamResponseCC(http.StatusOK, `{"data":"v1"}`, "max-age=300") f.Response(miss) - miss.FMetrics.(*metricstest.MockMetrics).WithCounters(func(counters map[string]int64) { - if counters["miss"] != 1 { - t.Errorf("after MISS: expected miss==1, got %d", counters["miss"]) + mockMetrics.WithCounters(func(counters map[string]int64) { + if counters["cache.miss"] != 1 { + t.Errorf("after MISS: expected cache.miss==1, got %d", counters["cache.miss"]) } - if counters["hit"] != 0 { - t.Errorf("after MISS: expected hit==0, got %d", counters["hit"]) + if counters["cache.hit"] != 0 { + t.Errorf("after MISS: expected cache.hit==0, got %d", counters["cache.hit"]) } }) @@ -732,12 +746,12 @@ func TestCacheFilter_Metrics(t *testing.T) { if !hit.FServed { t.Fatal("expected HIT within TTL") } - hit.FMetrics.(*metricstest.MockMetrics).WithCounters(func(counters map[string]int64) { - if counters["hit"] != 1 { - t.Errorf("after HIT: expected hit==1, got %d", counters["hit"]) + mockMetrics.WithCounters(func(counters map[string]int64) { + if counters["cache.hit"] != 1 { + t.Errorf("after HIT: expected cache.hit==1, got %d", counters["cache.hit"]) } - if counters["stale"] != 0 { - t.Errorf("after HIT: expected stale==0, got %d", counters["stale"]) + if counters["cache.stale"] != 0 { + t.Errorf("after HIT: expected cache.stale==0, got %d", counters["cache.stale"]) } }) @@ -761,12 +775,13 @@ func TestCacheFilter_Metrics(t *testing.T) { if stale.FResponse.Header.Get("X-Cache-Status") != "STALE" { t.Fatalf("expected STALE header, got %q", stale.FResponse.Header.Get("X-Cache-Status")) } - stale.FMetrics.(*metricstest.MockMetrics).WithCounters(func(counters map[string]int64) { - if counters["stale"] != 1 { - t.Errorf("after STALE: expected stale==1, got %d", counters["stale"]) + mockMetrics.WithCounters(func(counters map[string]int64) { + if counters["cache.stale"] != 1 { + t.Errorf("after STALE: expected cache.stale==1, got %d", counters["cache.stale"]) } - if counters["hit"] != 0 { - t.Errorf("after STALE: expected hit==0, got %d", counters["hit"]) + // cache.hit==1 from the preceding HIT request; STALE must not add another hit. + if counters["cache.hit"] != 1 { + t.Errorf("after STALE: expected cache.hit still==1 (from HIT step), got %d", counters["cache.hit"]) } }) }) @@ -946,8 +961,8 @@ func TestCacheFilter_RevalidationError_MetricIncremented(t *testing.T) { t.Fatal("expected STALE to be served") } mockMetrics.WithCounters(func(counters map[string]int64) { - if counters["reval_error"] != 1 { - t.Errorf("expected reval_error==1, got %d", counters["reval_error"]) + if counters["cache.reval_error"] != 1 { + t.Errorf("expected reval_error==1, got %d", counters["cache.reval_error"]) } }) }) @@ -2721,7 +2736,7 @@ func TestCacheFilter_LRUBytesGaugeUpdatesWithoutEviction(t *testing.T) { var initialBytes float64 mockMetrics.WithGauges(func(g map[string]float64) { - initialBytes = g["lru_bytes"] + initialBytes = g["cache.lru_bytes"] }) // Store an entry large enough to be visible but not enough to evict. @@ -2733,7 +2748,7 @@ func TestCacheFilter_LRUBytesGaugeUpdatesWithoutEviction(t *testing.T) { var afterBytes float64 mockMetrics.WithGauges(func(g map[string]float64) { - afterBytes = g["lru_bytes"] + afterBytes = g["cache.lru_bytes"] }) if afterBytes <= initialBytes { t.Errorf("expected lru_bytes to increase after Set without eviction; before=%v after=%v", initialBytes, afterBytes) @@ -2852,8 +2867,8 @@ func TestCacheFilter_RevalDropped_WhenQueueFull(t *testing.T) { t.Fatalf("expected X-Cache-Status: STALE, got %q", ctx.FResponse.Header.Get("X-Cache-Status")) } mockMetrics.WithCounters(func(counters map[string]int64) { - if counters["reval_dropped"] != 1 { - t.Errorf("expected reval_dropped==1, got %d", counters["reval_dropped"]) + if counters["cache.reval_dropped"] != 1 { + t.Errorf("expected reval_dropped==1, got %d", counters["cache.reval_dropped"]) } }) diff --git a/filters/cache/lru_storage.go b/filters/cache/lru_storage.go index 70ab5be1c9..75afb61f05 100644 --- a/filters/cache/lru_storage.go +++ b/filters/cache/lru_storage.go @@ -63,7 +63,7 @@ func (s *LRUStorage) Set(_ context.Context, key string, entry *Entry) error { "size_bytes": len(data), "shard_max": s.lru.shards[0].maxBytes, }).Warn("cache: entry exceeds shard capacity and will not be stored") - s.metrics.IncCounter("lru_oversized") + s.metrics.IncCounter("cache.lru_oversized") return nil } s.lru.Set(key, data) diff --git a/filters/cache/lru_test.go b/filters/cache/lru_test.go index 6165b47ddd..2c56cf984b 100644 --- a/filters/cache/lru_test.go +++ b/filters/cache/lru_test.go @@ -191,7 +191,7 @@ func TestLRUStorage_OversizedEntry(t *testing.T) { } // The lru_oversized counter must have been incremented exactly once. - if got := m.counter("lru_oversized"); got != 1 { + if got := m.counter("cache.lru_oversized"); got != 1 { t.Errorf("lru_oversized counter: got %d, want 1", got) } diff --git a/filters/cache/valkey_storage.go b/filters/cache/valkey_storage.go index 74ecb88ed5..bece0bda58 100644 --- a/filters/cache/valkey_storage.go +++ b/filters/cache/valkey_storage.go @@ -48,17 +48,17 @@ func (s *ValkeyStorage) Get(ctx context.Context, key string) (*Entry, error) { // L1-first: serve from local memory when the write-through warming populated it. // LRUStorage.Get returns nil, nil for expired entries — a miss falls through. if e, err := s.l1.Get(ctx, key); err == nil && e != nil { - s.metrics.IncCounter("l1_hit") + s.metrics.IncCounter("cache.l1_hit") return e, nil } data, err := s.ring.Get(ctx, key) if err != nil { if valkey.IsValkeyNil(err) { - s.metrics.IncCounter("valkey_miss") + s.metrics.IncCounter("cache.valkey_miss") return nil, nil } - s.metrics.IncCounter("valkey_get_fallback") + s.metrics.IncCounter("cache.valkey_get_fallback") log.WithError(err).Warn("cache: valkey Get failed, falling back to L1") // Second L1 lookup: a concurrent request may have written to L1 via the // valkey_set_fallback path between our miss above and this Valkey error. @@ -76,7 +76,7 @@ func (s *ValkeyStorage) Get(ctx context.Context, key string) (*Entry, error) { warmed.TTL = min(s.l1TTL, remaining) warmed.CreatedAt = time.Now() _ = s.l1.Set(ctx, key, &warmed) - s.metrics.IncCounter("l1_warm_from_valkey") + s.metrics.IncCounter("cache.l1_warm_from_valkey") } } return &e, nil @@ -94,7 +94,7 @@ func (s *ValkeyStorage) Set(ctx context.Context, key string, entry *Entry) error } if err := s.ring.SetWithExpire(ctx, key, string(data), valkeyTTL); err != nil { - s.metrics.IncCounter("valkey_set_fallback") + s.metrics.IncCounter("cache.valkey_set_fallback") log.WithError(err).Warn("cache: valkey Set failed, falling back to L1") return s.l1.Set(ctx, key, entry) } diff --git a/filters/cache/valkey_storage_test.go b/filters/cache/valkey_storage_test.go index 113c2beab9..adc55a0495 100644 --- a/filters/cache/valkey_storage_test.go +++ b/filters/cache/valkey_storage_test.go @@ -228,11 +228,11 @@ func TestValkeyStorage_FallsBackToL1OnValkeyUnavailable(t *testing.T) { if got == nil { t.Fatal("expected L1 fallback hit, got nil") } - if m.counter("l1_hit") == 0 { + if m.counter("cache.l1_hit") == 0 { t.Error("expected l1_hit to be incremented: Set fallback warmed L1, Get should serve from it") } - if m.counter("valkey_get_fallback") != 0 { - t.Errorf("expected valkey_get_fallback=0 (L1 served before Valkey was contacted), got %d", m.counter("valkey_get_fallback")) + if m.counter("cache.valkey_get_fallback") != 0 { + t.Errorf("expected valkey_get_fallback=0 (L1 served before Valkey was contacted), got %d", m.counter("cache.valkey_get_fallback")) } // Confirm the entry was physically written to L1 — not just returned via some @@ -260,11 +260,11 @@ func TestValkeyStorage_RecordsValkeyMiss(t *testing.T) { if got != nil { t.Fatalf("expected nil on miss, got %+v", got) } - if m.counter("valkey_miss") != 1 { - t.Errorf("expected valkey_miss=1, got %d", m.counter("valkey_miss")) + if m.counter("cache.valkey_miss") != 1 { + t.Errorf("expected valkey_miss=1, got %d", m.counter("cache.valkey_miss")) } - if m.counter("valkey_get_fallback") != 0 { - t.Errorf("expected valkey_get_fallback=0 on clean miss, got %d", m.counter("valkey_get_fallback")) + if m.counter("cache.valkey_get_fallback") != 0 { + t.Errorf("expected valkey_get_fallback=0 on clean miss, got %d", m.counter("cache.valkey_get_fallback")) } } @@ -300,8 +300,8 @@ func TestValkeyStorage_WriteThroughWarmsL1(t *testing.T) { if string(got.Payload) != "warm" { t.Errorf("payload: got %q, want %q", string(got.Payload), "warm") } - if m.counter("l1_hit") != 1 { - t.Errorf("expected l1_hit=1, got %d", m.counter("l1_hit")) + if m.counter("cache.l1_hit") != 1 { + t.Errorf("expected l1_hit=1, got %d", m.counter("cache.l1_hit")) } } @@ -381,24 +381,24 @@ func TestValkeyStorage_SplitFallbackCounters(t *testing.T) { entry := &Entry{StatusCode: 200, Payload: []byte("x"), TTL: time.Minute, CreatedAt: time.Now()} _ = s.Set(ctx, "k", entry) - if m.counter("valkey_set_fallback") != 1 { - t.Errorf("expected valkey_set_fallback=1, got %d", m.counter("valkey_set_fallback")) + if m.counter("cache.valkey_set_fallback") != 1 { + t.Errorf("expected valkey_set_fallback=1, got %d", m.counter("cache.valkey_set_fallback")) } - if m.counter("valkey_get_fallback") != 0 { - t.Errorf("expected valkey_get_fallback=0 after Set, got %d", m.counter("valkey_get_fallback")) + if m.counter("cache.valkey_get_fallback") != 0 { + t.Errorf("expected valkey_get_fallback=0 after Set, got %d", m.counter("cache.valkey_get_fallback")) } // L1-first: the entry was written to L1 by the Set fallback path, so Get returns // it from L1 without ever touching (broken) Valkey. _, _ = s.Get(ctx, "k") - if m.counter("l1_hit") != 1 { - t.Errorf("expected l1_hit=1, got %d", m.counter("l1_hit")) + if m.counter("cache.l1_hit") != 1 { + t.Errorf("expected l1_hit=1, got %d", m.counter("cache.l1_hit")) } - if m.counter("valkey_get_fallback") != 0 { - t.Errorf("expected valkey_get_fallback=0 (L1 served before Valkey check), got %d", m.counter("valkey_get_fallback")) + if m.counter("cache.valkey_get_fallback") != 0 { + t.Errorf("expected valkey_get_fallback=0 (L1 served before Valkey check), got %d", m.counter("cache.valkey_get_fallback")) } - if m.counter("valkey_set_fallback") != 1 { - t.Errorf("valkey_set_fallback should still be 1, got %d", m.counter("valkey_set_fallback")) + if m.counter("cache.valkey_set_fallback") != 1 { + t.Errorf("valkey_set_fallback should still be 1, got %d", m.counter("cache.valkey_set_fallback")) } } From 1b7050f3c4dd53fb0b7f710903f874986e1a506c Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Wed, 19 Aug 2026 12:02:43 +0200 Subject: [PATCH 61/89] docs: replace SIE acronym with stale-if-error in comments Signed-off-by: Larry D Almeida --- filters/cache/filter.go | 10 +++++----- filters/cache/filter_test.go | 12 ++++++------ 2 files changed, 11 insertions(+), 11 deletions(-) diff --git a/filters/cache/filter.go b/filters/cache/filter.go index 63b5344d54..0dccb8ed0d 100644 --- a/filters/cache/filter.go +++ b/filters/cache/filter.go @@ -456,12 +456,12 @@ func (f *cacheFilter) Request(ctx filters.FilterContext) { ctx.Serve(headBodyOmitted(method, rsp)) } -// coalesceResult carries both the fetched entry and any SIE-eligible stored entry -// snapshotted before the fetch. The snapshot is taken before f.fetch() so that a -// 5xx result cannot overwrite it in storage before the stale-if-error check runs. +// coalesceResult carries both the fetched entry and any stale-if-error eligible +// stored entry snapshotted before the fetch. The snapshot is taken before f.fetch() so +// that a 5xx result cannot overwrite it in storage before the stale-if-error check runs. type coalesceResult struct { entry *Entry - stored *Entry // snapshot before fetch; nil if no eligible SIE entry existed + stored *Entry // snapshot before fetch; nil if no eligible stale-if-error entry existed } // coalesce gates concurrent cold misses for the same key behind a single upstream @@ -474,7 +474,7 @@ func (f *cacheFilter) coalesce(ctx filters.FilterContext, key string) { req := ctx.Request().Clone(context.Background()) ch := f.coldSF.DoChan(key, func() (interface{}, error) { - // Capture any existing SIE-eligible entry before fetching, so that a + // Capture any existing stale-if-error eligible entry before fetching, so that a // subsequent 5xx response cannot overwrite it in storage before we read it. var sieStored *Entry if f.staleIfError > 0 { diff --git a/filters/cache/filter_test.go b/filters/cache/filter_test.go index 3b1bc6d4c8..af53fa4d3b 100644 --- a/filters/cache/filter_test.go +++ b/filters/cache/filter_test.go @@ -2391,9 +2391,9 @@ func TestCacheFilter_StaleIfError_Serves_On_5xx(t *testing.T) { // Entry expires after 1ms; staleIfError=60s keeps it in storage. // A 503 upstream via coalesce should cause the stale entry to be served. // - // Regression: the SIE block in Response() was dead code — coalesce() always calls + // Regression: the stale-if-error block in Response() was dead code — coalesce() always calls // ctx.Serve() (even on 5xx), so Response() returned early before reaching it. - // SIE logic must live inside coalesce() with the pre-fetch snapshot captured + // stale-if-error logic must live inside coalesce() with the pre-fetch snapshot captured // before f.fetch runs, preventing the 5xx from overwriting the stored entry. f := newTestFilter(t, time.Millisecond, 10*time.Second, time.Millisecond, 60*time.Second) url := "http://example.com/sie-5xx" @@ -2433,8 +2433,8 @@ func TestCacheFilter_StaleIfError_Serves_On_5xx(t *testing.T) { func TestCacheFilter_StaleIfError_Expired_NotServed(t *testing.T) { // ttl=1ms, errorTTL=10s, swrWindow=1ms, staleIfError=100ms // Sleep 200ms — past TTL + staleIfError window. Entry too old for stale-if-error. - // Uses f.fetch returning 503 via coalesce (the same path as the positive SIE case) - // to confirm the 503 is passed through when the SIE window has already elapsed. + // Uses f.fetch returning 503 via coalesce (the same path as the positive stale-if-error case) + // to confirm the 503 is passed through when the stale-if-error window has already elapsed. f := newTestFilter(t, time.Millisecond, 10*time.Second, time.Millisecond, 100*time.Millisecond) url := "http://example.com/sie-expired" @@ -2461,7 +2461,7 @@ func TestCacheFilter_StaleIfError_Expired_NotServed(t *testing.T) { t.Fatal("expected a response, got nil") } if ctx2.FResponse.StatusCode != http.StatusServiceUnavailable { - t.Fatalf("want 503 (SIE window expired), got %d", ctx2.FResponse.StatusCode) + t.Fatalf("want 503 (stale-if-error window expired), got %d", ctx2.FResponse.StatusCode) } }) } @@ -2485,7 +2485,7 @@ func TestCacheFilter_StaleIfError_Disabled_When_Zero(t *testing.T) { f.Response(ctx2) if ctx2.FResponse.StatusCode != http.StatusServiceUnavailable { - t.Fatalf("want 503 (SIE disabled), got %d", ctx2.FResponse.StatusCode) + t.Fatalf("want 503 (stale-if-error disabled), got %d", ctx2.FResponse.StatusCode) } }) } From 96602e5fdbd84bb8aa43062e6d76ff0ae58a598a Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Wed, 19 Aug 2026 12:04:55 +0200 Subject: [PATCH 62/89] docs: clarify shared L1 cache key semantics and per-user key isolation Signed-off-by: Larry D Almeida --- docs/operation/operation.md | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/docs/operation/operation.md b/docs/operation/operation.md index 72d71dfbbc..3b3206eeb2 100644 --- a/docs/operation/operation.md +++ b/docs/operation/operation.md @@ -1957,8 +1957,11 @@ via `skipper.Options.ResponseCacheMaxMemoryBytes`. !!! note An in-process LRU (L1) is shared across all `cache()` filter instances in the same process. - The L1 storage budget is divided evenly across 256 internal shards; a single entry larger - than one shard's budget is dropped with a warning log. + Two routes or two users that produce the same cache key will share the cached entry — the + second request is served whatever the first stored. Ensure the cache key includes all + dimensions that distinguish responses (e.g. add `Authorization` to `keyHeaders` for + per-user routes). The L1 storage budget is divided evenly across 256 internal shards; a + single entry larger than one shard's budget is dropped with a warning log. ### Metrics From 5398e72d39ffe8b482636ed0607c6fe13a87d564 Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Wed, 19 Aug 2026 12:58:57 +0200 Subject: [PATCH 63/89] =?UTF-8?q?docs:=20clarify=20cache=20invalidation=20?= =?UTF-8?q?=E2=80=94=20no=20out-of-band=20API,=20document=20available=20op?= =?UTF-8?q?tions?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Signed-off-by: Larry D Almeida --- docs/operation/operation.md | 14 +++++++++----- 1 file changed, 9 insertions(+), 5 deletions(-) diff --git a/docs/operation/operation.md b/docs/operation/operation.md index 3b3206eeb2..a813443153 100644 --- a/docs/operation/operation.md +++ b/docs/operation/operation.md @@ -1940,11 +1940,15 @@ re-consulting Valkey. Set `--cache-l1-ttl=0` to disable L1 warming and restore write-around behaviour (L1 used only when Valkey is unavailable; this applies to both read errors (`valkey_get_fallback`) and write errors (`valkey_set_fallback`)). -Explicit deletes (unsafe methods or operator-initiated invalidation) always -remove the L1 entry unconditionally, regardless of `--cache-l1-ttl`. However, -only the local process's L1 is cleared — other Skipper processes in the fleet -retain their own L1 copies until `--cache-l1-ttl` expires. Set `--cache-l1-ttl` -accordingly to bound the stale window after an invalidation. +When an upstream responds successfully to an unsafe method (`POST`, `PUT`, `DELETE`, `PATCH`), +the filter removes the cached entry for that URL from both Valkey and the local L1. +Only the local process's L1 is cleared — other Skipper processes in the fleet retain their +own L1 copies until `--cache-l1-ttl` expires. Set `--cache-l1-ttl` accordingly to bound +the stale window after an invalidation. + +There is no out-of-band operator invalidation API. To clear the cache outside the normal +unsafe-method path, options are: wait for TTL expiry, restart the Skipper process (clears +L1 entirely), or delete the key directly in Valkey (L2 only; does not clear other pods' L1). Concurrent cold-miss requests for the same key within one Skipper process are coalesced into a single upstream fetch (thundering-herd protection). This is From bd7825f00388baf7872a3b88e995aa96bd58d180 Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Wed, 19 Aug 2026 13:03:25 +0200 Subject: [PATCH 64/89] docs: clarify process restart only clears L1; Valkey data persists Signed-off-by: Larry D Almeida --- docs/operation/operation.md | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/docs/operation/operation.md b/docs/operation/operation.md index a813443153..5f553af018 100644 --- a/docs/operation/operation.md +++ b/docs/operation/operation.md @@ -1947,8 +1947,8 @@ own L1 copies until `--cache-l1-ttl` expires. Set `--cache-l1-ttl` accordingly t the stale window after an invalidation. There is no out-of-band operator invalidation API. To clear the cache outside the normal -unsafe-method path, options are: wait for TTL expiry, restart the Skipper process (clears -L1 entirely), or delete the key directly in Valkey (L2 only; does not clear other pods' L1). +unsafe-method path, options are: wait for TTL expiry, restart the Skipper process (clears L1 in-memory cache only; Valkey data persists), or +delete the key directly in Valkey (L2 only; does not clear other pods' L1). Concurrent cold-miss requests for the same key within one Skipper process are coalesced into a single upstream fetch (thundering-herd protection). This is From 473696a9d3ad650ddd0c326be75b96d6c909ac80 Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Wed, 19 Aug 2026 13:24:13 +0200 Subject: [PATCH 65/89] =?UTF-8?q?docs:=20clarify=20coalescing=20is=20per-r?= =?UTF-8?q?oute=20and=20document=20RFC=209111=20=C2=A73.5=20Authorization?= =?UTF-8?q?=20guard?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Signed-off-by: Larry D Almeida --- docs/operation/operation.md | 9 +++++---- docs/reference/filters.md | 6 +++++- 2 files changed, 10 insertions(+), 5 deletions(-) diff --git a/docs/operation/operation.md b/docs/operation/operation.md index 5f553af018..e2714f346d 100644 --- a/docs/operation/operation.md +++ b/docs/operation/operation.md @@ -1950,10 +1950,11 @@ There is no out-of-band operator invalidation API. To clear the cache outside th unsafe-method path, options are: wait for TTL expiry, restart the Skipper process (clears L1 in-memory cache only; Valkey data persists), or delete the key directly in Valkey (L2 only; does not clear other pods' L1). -Concurrent cold-miss requests for the same key within one Skipper process are -coalesced into a single upstream fetch (thundering-herd protection). This is -process-local: a fleet of N instances may still issue up to N simultaneous -origin requests on a cold miss. +Concurrent cold-miss requests for the same route and key within one Skipper process are +coalesced into a single upstream fetch (thundering-herd protection). Requests arriving via +different routes are not coalesced even if they target the same upstream URL, because the +route ID is part of the cache key. This protection is also process-local: a fleet of N +instances may still issue up to N simultaneous origin requests on a cold miss. The L1 memory budget defaults to 25% of the container's cgroup memory limit, falling back to 2 GB if the limit is unreadable. Override it programmatically diff --git a/docs/reference/filters.md b/docs/reference/filters.md index e79bd0ecd8..4febf0f599 100644 --- a/docs/reference/filters.md +++ b/docs/reference/filters.md @@ -4043,7 +4043,11 @@ matching the same key. It has no awareness of other filters in the chain. * **`Authorization` is not in the key by default.** Responses are stored and served without regard to caller identity. To isolate per-user responses, add `Authorization` to `keyHeaders`; without it, a response stored for one user - will be served to all others on the same path. + will be served to all others on the same path. Note: in RFC mode, if the + request carries an `Authorization` header and the upstream does not respond + with `Cache-Control: public` or `must-revalidate`, the response is silently + not stored (RFC 9111 §3.5). Use force mode or ensure the upstream sets + `Cache-Control: public` if you want authenticated responses cached. * **`Cache-Control: private` is ignored in force mode.** Audit the upstream response before enabling force mode on any authenticated route. From 61fb35aedb634db379e7395bece93a469d1c6146 Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Tue, 25 Aug 2026 08:18:05 +0200 Subject: [PATCH 66/89] refactor: rename metricsScraper to updateMetrics Signed-off-by: Larry D Almeida --- filters/cache/filter.go | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/filters/cache/filter.go b/filters/cache/filter.go index 0dccb8ed0d..6866da4389 100644 --- a/filters/cache/filter.go +++ b/filters/cache/filter.go @@ -133,7 +133,7 @@ func NewCacheFilter(opts Options) filters.Spec { // Start shared background goroutines (one worker + one scraper for all filter instances) spec.bgWg.Add(2) go spec.revalidationWorker() - go spec.metricsScraper() + go spec.updateMetrics() return spec } @@ -258,9 +258,9 @@ func (s *cacheSpec) revalidationWorker() { const lruBytesScrapeInterval = 10 * time.Second -// metricsScraper periodically updates the lru_bytes gauge so it stays current +// updateMetrics periodically updates the lru_bytes gauge so it stays current // even when no evictions occur. It's spec-level and shared across all filter instances. -func (s *cacheSpec) metricsScraper() { +func (s *cacheSpec) updateMetrics() { defer s.bgWg.Done() ticker := time.NewTicker(lruBytesScrapeInterval) defer ticker.Stop() From 761e2dc1f58473d1e7df9bd870939d5037776075 Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Tue, 25 Aug 2026 10:59:12 +0200 Subject: [PATCH 67/89] refactor: remove cacheFilter.Close no-op, io.Closer assertio Signed-off-by: Larry D Almeida --- filters/cache/filter.go | 11 ++++------- filters/cache/filter_test.go | 7 ------- 2 files changed, 4 insertions(+), 14 deletions(-) diff --git a/filters/cache/filter.go b/filters/cache/filter.go index 6866da4389..df85b0344a 100644 --- a/filters/cache/filter.go +++ b/filters/cache/filter.go @@ -70,8 +70,6 @@ type filterCacheKey struct { rfcMode bool } -var _ io.Closer = (*cacheSpec)(nil) - type cacheSpec struct { maxBytes int64 listenAddr string @@ -282,6 +280,10 @@ type revalJob struct { enqueuedAt time.Time // wall-clock time the job entered the queue; used to measure wait time } +// cacheFilter does not implement filters.FilterCloser. The routing layer calls +// Close() on every FilterCloser when a route is invalidated; because all +// goroutines and shared resources are owned by cacheSpec, closing them here +// would be destructive. Lifecycle is managed exclusively by cacheSpec.Close(). type cacheFilter struct { storage Storage lruStorage *LRUStorage // always non-nil; direct reference to L1, even when storage is ValkeyStorage @@ -303,11 +305,6 @@ type cacheFilter struct { metrics metrics.Metrics } -// Close is intentionally a no-op. The routing layer calls Close() on every -// FilterCloser when a route is invalidated; closing resources here would be -// destructive. Lifecycle is managed by cacheSpec.Close(). -func (f *cacheFilter) Close() {} - // tagSpan sets cache_status and (when >= 0) cache_ttl_remaining_ms // on the active OpenTracing span. No-op when no span is present. func tagSpan(ctx filters.FilterContext, status string, ttlRemainingMs int64) { diff --git a/filters/cache/filter_test.go b/filters/cache/filter_test.go index af53fa4d3b..cf2a06fc4f 100644 --- a/filters/cache/filter_test.go +++ b/filters/cache/filter_test.go @@ -38,7 +38,6 @@ func newTestFilter(t *testing.T, ttl, errorTTL, swrWindow time.Duration, extra . cf.fetch = func(*http.Request) (*http.Response, error) { return nil, errors.New("no fetch stub set") } - t.Cleanup(cf.Close) t.Cleanup(spec.(*cacheSpec).client.Close) t.Cleanup(func() { spec.(*cacheSpec).Close() }) return cf @@ -60,7 +59,6 @@ func newTestFilterRFC(t *testing.T, _, _, _ time.Duration, _ ...time.Duration) * cf.fetch = func(*http.Request) (*http.Response, error) { return nil, errors.New("no fetch stub set") } - t.Cleanup(cf.Close) t.Cleanup(spec.(*cacheSpec).client.Close) t.Cleanup(func() { spec.(*cacheSpec).Close() }) return cf @@ -138,7 +136,6 @@ func TestCacheFilter_KeyIsolationByAuthToken(t *testing.T) { t.Fatal(err) } f := fi.(*cacheFilter) - t.Cleanup(f.Close) t.Cleanup(spec.(*cacheSpec).client.Close) t.Cleanup(func() { spec.(*cacheSpec).Close() }) f.fetch = func(*http.Request) (*http.Response, error) { @@ -1345,7 +1342,6 @@ func TestCacheFilter_SharedStorage_RouteIsolation(t *testing.T) { t.Fatal(err) } cf := f.(*cacheFilter) - t.Cleanup(cf.Close) // Default fetch stub returns an error so coalesce does not serve the // request; this allows the test to distinguish a true cache HIT from a // coalesced upstream fetch. @@ -2664,7 +2660,6 @@ func TestCacheFilter_CreateFilter_RFCArgParsing(t *testing.T) { t.Fatalf("unexpected error: %v", err) } cf := f.(*cacheFilter) - t.Cleanup(cf.Close) if cf.rfcMode != tc.wantRFC { t.Errorf("rfcMode: got %v, want %v", cf.rfcMode, tc.wantRFC) } @@ -2686,7 +2681,6 @@ func TestCacheFilter_PureRFCMode_ZeroArgs_UsesUpstreamMaxAge(t *testing.T) { t.Fatalf("unexpected error: %v", err) } cf := f.(*cacheFilter) - t.Cleanup(cf.Close) cf.fetch = func(*http.Request) (*http.Response, error) { return nil, errors.New("no fetch stub") } @@ -2767,7 +2761,6 @@ func TestCacheFilter_PureRFCMode_ZeroArgs_NoUpstreamDirective_NotCached(t *testi t.Fatalf("unexpected error: %v", err) } cf := f.(*cacheFilter) - t.Cleanup(cf.Close) cf.fetch = func(*http.Request) (*http.Response, error) { return nil, errors.New("no fetch stub") } From 2d0defc10e882eb0fc4f2e1953a9e9f8cd8c3b57 Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Tue, 25 Aug 2026 11:49:20 +0200 Subject: [PATCH 68/89] fix: treat valkey Get errors as misses instead of falling back to L1 Signed-off-by: Larry D Almeida --- docs/operation/operation.md | 8 +++++--- filters/cache/valkey_storage.go | 12 ++++++------ 2 files changed, 11 insertions(+), 9 deletions(-) diff --git a/docs/operation/operation.md b/docs/operation/operation.md index e2714f346d..0e0d70f491 100644 --- a/docs/operation/operation.md +++ b/docs/operation/operation.md @@ -1937,8 +1937,9 @@ remaining freshness (`entry.TTL - age`) is used rather than the original TTL to prevent L1 from serving the entry beyond Valkey's actual expiry. The default is 60 seconds, bounding how long Skipper serves a locally-cached entry before re-consulting Valkey. Set `--cache-l1-ttl=0` to disable L1 warming and -restore write-around behaviour (L1 used only when Valkey is unavailable; this -applies to both read errors (`valkey_get_fallback`) and write errors (`valkey_set_fallback`)). +restore write-around behaviour (on Valkey write errors (`valkey_set_fallback`), L1 is +used as a fallback; Valkey read errors (`valkey_get_fallback`) are treated as misses +and do not consult L1). When an upstream responds successfully to an unsafe method (`POST`, `PUT`, `DELETE`, `PATCH`), the filter removes the cached entry for that URL from both Valkey and the local L1. @@ -1995,7 +1996,8 @@ via `skipper.Options.ResponseCacheMaxMemoryBytes`. - `cache.l1_hit`: Counter, L1 hits that bypassed Valkey - `cache.valkey_miss`: Counter, Valkey misses that proceeded to an upstream fetch -- `cache.valkey_get_fallback`, `cache.valkey_set_fallback`: Counters, reads/writes that fell back to L1 due to Valkey errors +- `cache.valkey_get_fallback`: Counter, Valkey Get errors — request treated as a cache miss, fetched from origin +- `cache.valkey_set_fallback`: Counter, Valkey Set errors — entry written to L1 as fallback - `cache.l1_warm_from_valkey`: Counter, entries written into L1 after a successful Valkey Get (write-through on read path) **OpenTracing span tags (set on every request when a span is active):** diff --git a/filters/cache/valkey_storage.go b/filters/cache/valkey_storage.go index bece0bda58..f0d9c500b5 100644 --- a/filters/cache/valkey_storage.go +++ b/filters/cache/valkey_storage.go @@ -23,7 +23,9 @@ type valkeyClient interface { var _ valkeyClient = (*skpnet.ValkeyRingClient)(nil) // ValkeyStorage implements Storage using a ValkeyRingClient (L2) with -// automatic fallback to LRUStorage (L1) on any Valkey error. +// write-through warming of LRUStorage (L1). On Valkey Set errors, L1 is +// used as a fallback. On Valkey Get errors, the request is treated as a +// miss and fetched from origin. type ValkeyStorage struct { ring valkeyClient l1 *LRUStorage @@ -36,7 +38,7 @@ type ValkeyStorage struct { // // - l1_hit — L1 returned a warm entry; Valkey not consulted // - valkey_miss — clean cache miss (key not found in Valkey) -// - valkey_get_fallback — Valkey error on Get; L1 was consulted instead +// - valkey_get_fallback — Valkey error on Get; treated as a cache miss // - valkey_set_fallback — Valkey error on Set; L1 was written instead // // Pass metrics.Default when no test-scoped metrics collector is needed. @@ -59,10 +61,8 @@ func (s *ValkeyStorage) Get(ctx context.Context, key string) (*Entry, error) { return nil, nil } s.metrics.IncCounter("cache.valkey_get_fallback") - log.WithError(err).Warn("cache: valkey Get failed, falling back to L1") - // Second L1 lookup: a concurrent request may have written to L1 via the - // valkey_set_fallback path between our miss above and this Valkey error. - return s.l1.Get(ctx, key) + log.WithError(err).Warn("cache: valkey Get failed, treating as miss") + return nil, nil } var e Entry if err := json.Unmarshal([]byte(data), &e); err != nil { From ea96966f012612d6b6c56e1e0fec6af130040db8 Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Tue, 25 Aug 2026 14:27:50 +0200 Subject: [PATCH 69/89] fix: update TestCacheFilter_LRUBytesGaugeUpdatesWithoutEviction to pass string args to CreateFilter Signed-off-by: Larry D Almeida --- filters/cache/filter_test.go | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/filters/cache/filter_test.go b/filters/cache/filter_test.go index dfbf4788c7..e13dc16b7d 100644 --- a/filters/cache/filter_test.go +++ b/filters/cache/filter_test.go @@ -2711,7 +2711,7 @@ func TestCacheFilter_LRUBytesGaugeUpdatesWithoutEviction(t *testing.T) { spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: "localhost:9090", L1TTL: 60 * time.Second}) t.Cleanup(spec.(*cacheSpec).client.Close) t.Cleanup(func() { spec.(*cacheSpec).Close() }) - f, err := spec.CreateFilter([]interface{}{5 * time.Minute, 15 * time.Second, 5 * time.Minute}) + f, err := spec.CreateFilter([]interface{}{"5m", "15s", "5m"}) if err != nil { t.Fatal(err) } From 6ae452aa5f5f031cbfc9831c520ab27bcc733c20 Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Thu, 27 Aug 2026 12:45:25 +0200 Subject: [PATCH 70/89] fix: update docs and metrics to reflect valkey Get error handling changes Signed-off-by: Larry D Almeida --- config/config.go | 7 +++++-- docs/operation/operation.md | 22 +++++++++++----------- filters/cache/valkey_storage.go | 4 ++-- filters/cache/valkey_storage_test.go | 18 +++++++++--------- 4 files changed, 27 insertions(+), 24 deletions(-) diff --git a/config/config.go b/config/config.go index 5d0d74543d..1b0027eff3 100644 --- a/config/config.go +++ b/config/config.go @@ -354,7 +354,8 @@ type Config struct { SwarmStaticOther string `yaml:"swarm-static-other"` // cache - CacheL1TTL time.Duration `yaml:"cache-l1-ttl"` + CacheL1TTL time.Duration `yaml:"cache-l1-ttl"` + CacheL1MaxMemoryBytes int64 `yaml:"cache-l1-max-memory-bytes"` ClusterRatelimitMaxGroupShards int `yaml:"cluster-ratelimit-max-group-shards"` @@ -750,6 +751,7 @@ func NewConfig() *Config { // cache flag.DurationVar(&cfg.CacheL1TTL, "cache-l1-ttl", 60*time.Second, "maximum TTL for write-through L1 warming in the cache() filter when Valkey is configured; set to 0 to disable (write-around)") + flag.Int64Var(&cfg.CacheL1MaxMemoryBytes, "cache-l1-max-memory-bytes", 0, "maximum memory budget in bytes for the cache() filter's in-process LRU (L1); defaults to 25% of cgroup memory limit or 2 GB if unreadable") flag.IntVar(&cfg.ClusterRatelimitMaxGroupShards, "cluster-ratelimit-max-group-shards", 1, "sets the maximum number of group shards for the clusterRatelimit filter") @@ -1217,7 +1219,8 @@ func (c *Config) ToOptions() skipper.Options { SwarmStaticOther: c.SwarmStaticOther, // cache - CacheL1TTL: c.CacheL1TTL, + CacheL1TTL: c.CacheL1TTL, + ResponseCacheMaxMemoryBytes: c.CacheL1MaxMemoryBytes, ClusterRatelimitMaxGroupShards: c.ClusterRatelimitMaxGroupShards, diff --git a/docs/operation/operation.md b/docs/operation/operation.md index 9a17a1b7f2..1782f34da7 100644 --- a/docs/operation/operation.md +++ b/docs/operation/operation.md @@ -1975,15 +1975,15 @@ remaining freshness (`entry.TTL - age`) is used rather than the original TTL to prevent L1 from serving the entry beyond Valkey's actual expiry. The default is 60 seconds, bounding how long Skipper serves a locally-cached entry before re-consulting Valkey. Set `--cache-l1-ttl=0` to disable L1 warming and -restore write-around behaviour (on Valkey write errors (`valkey_set_fallback`), L1 is -used as a fallback; Valkey read errors (`valkey_get_fallback`) are treated as misses -and do not consult L1). +restore write-around behaviour. Note: on Valkey write errors (`valkey_set_fallback`), L1 +is always used as a fallback regardless of `--cache-l1-ttl`; Valkey read errors +(`valkey_get_error`) are treated as cache misses. When an upstream responds successfully to an unsafe method (`POST`, `PUT`, `DELETE`, `PATCH`), the filter removes the cached entry for that URL from both Valkey and the local L1. Only the local process's L1 is cleared — other Skipper processes in the fleet retain their -own L1 copies until `--cache-l1-ttl` expires. Set `--cache-l1-ttl` accordingly to bound -the stale window after an invalidation. +own L1 copies until each entry's warmed TTL (bounded by `--cache-l1-ttl`) expires naturally. +Set `--cache-l1-ttl` accordingly to bound the stale window after an invalidation. There is no out-of-band operator invalidation API. To clear the cache outside the normal unsafe-method path, options are: wait for TTL expiry, restart the Skipper process (clears L1 in-memory cache only; Valkey data persists), or @@ -1996,8 +1996,8 @@ route ID is part of the cache key. This protection is also process-local: a flee instances may still issue up to N simultaneous origin requests on a cold miss. The L1 memory budget defaults to 25% of the container's cgroup memory limit, -falling back to 2 GB if the limit is unreadable. Override it programmatically -via `skipper.Options.ResponseCacheMaxMemoryBytes`. +falling back to 2 GB if the limit is unreadable. Override with +`--cache-l1-max-memory-bytes` (or programmatically via `skipper.Options.ResponseCacheMaxMemoryBytes`). !!! note An in-process LRU (L1) is shared across all `cache()` filter instances in the same process. @@ -2005,7 +2005,7 @@ via `skipper.Options.ResponseCacheMaxMemoryBytes`. second request is served whatever the first stored. Ensure the cache key includes all dimensions that distinguish responses (e.g. add `Authorization` to `keyHeaders` for per-user routes). The L1 storage budget is divided evenly across 256 internal shards; a - single entry larger than one shard's budget is dropped with a warning log. + single entry larger than one shard's budget is dropped with a warning log and increments `cache.lru_oversized`. ### Metrics @@ -2019,8 +2019,8 @@ via `skipper.Options.ResponseCacheMaxMemoryBytes`. **LRU (always active):** - `cache.lru_eviction`: Counter, incremented each time an L1 entry is evicted due to memory pressure -- `cache.lru_bytes`: Gauge, current L1 usage in bytes -- `cache.lru_oversized`: Counter, incremented when an entry is too large for any shard and silently dropped +- `cache.lru_bytes`: Gauge, current L1 usage in bytes (sampled every 10s) +- `cache.lru_oversized`: Counter, incremented when an entry is too large for any shard and dropped **Revalidation (always active):** @@ -2034,7 +2034,7 @@ via `skipper.Options.ResponseCacheMaxMemoryBytes`. - `cache.l1_hit`: Counter, L1 hits that bypassed Valkey - `cache.valkey_miss`: Counter, Valkey misses that proceeded to an upstream fetch -- `cache.valkey_get_fallback`: Counter, Valkey Get errors — request treated as a cache miss, fetched from origin +- `cache.valkey_get_error`: Counter, Valkey Get errors — request treated as a cache miss, fetched from origin - `cache.valkey_set_fallback`: Counter, Valkey Set errors — entry written to L1 as fallback - `cache.l1_warm_from_valkey`: Counter, entries written into L1 after a successful Valkey Get (write-through on read path) diff --git a/filters/cache/valkey_storage.go b/filters/cache/valkey_storage.go index f0d9c500b5..72a6ad3ab7 100644 --- a/filters/cache/valkey_storage.go +++ b/filters/cache/valkey_storage.go @@ -38,7 +38,7 @@ type ValkeyStorage struct { // // - l1_hit — L1 returned a warm entry; Valkey not consulted // - valkey_miss — clean cache miss (key not found in Valkey) -// - valkey_get_fallback — Valkey error on Get; treated as a cache miss +// - valkey_get_error — Valkey error on Get; treated as a cache miss // - valkey_set_fallback — Valkey error on Set; L1 was written instead // // Pass metrics.Default when no test-scoped metrics collector is needed. @@ -60,7 +60,7 @@ func (s *ValkeyStorage) Get(ctx context.Context, key string) (*Entry, error) { s.metrics.IncCounter("cache.valkey_miss") return nil, nil } - s.metrics.IncCounter("cache.valkey_get_fallback") + s.metrics.IncCounter("cache.valkey_get_error") log.WithError(err).Warn("cache: valkey Get failed, treating as miss") return nil, nil } diff --git a/filters/cache/valkey_storage_test.go b/filters/cache/valkey_storage_test.go index adc55a0495..ca3db2bf51 100644 --- a/filters/cache/valkey_storage_test.go +++ b/filters/cache/valkey_storage_test.go @@ -231,8 +231,8 @@ func TestValkeyStorage_FallsBackToL1OnValkeyUnavailable(t *testing.T) { if m.counter("cache.l1_hit") == 0 { t.Error("expected l1_hit to be incremented: Set fallback warmed L1, Get should serve from it") } - if m.counter("cache.valkey_get_fallback") != 0 { - t.Errorf("expected valkey_get_fallback=0 (L1 served before Valkey was contacted), got %d", m.counter("cache.valkey_get_fallback")) + if m.counter("cache.valkey_get_error") != 0 { + t.Errorf("expected valkey_get_error=0 (L1 served before Valkey was contacted), got %d", m.counter("cache.valkey_get_error")) } // Confirm the entry was physically written to L1 — not just returned via some @@ -263,8 +263,8 @@ func TestValkeyStorage_RecordsValkeyMiss(t *testing.T) { if m.counter("cache.valkey_miss") != 1 { t.Errorf("expected valkey_miss=1, got %d", m.counter("cache.valkey_miss")) } - if m.counter("cache.valkey_get_fallback") != 0 { - t.Errorf("expected valkey_get_fallback=0 on clean miss, got %d", m.counter("cache.valkey_get_fallback")) + if m.counter("cache.valkey_get_error") != 0 { + t.Errorf("expected valkey_get_error=0 on clean miss, got %d", m.counter("cache.valkey_get_error")) } } @@ -371,7 +371,7 @@ func TestValkeyStorage_SplitFallbackCounters(t *testing.T) { // Uses a broken stub — no Docker or live Valkey needed. // Set triggers valkey_set_fallback, which writes the entry to L1. // Get checks L1 first (L1-first reads) and finds the entry — incrementing l1_hit, - // not valkey_get_fallback. + // not valkey_get_error. stub := newBrokenStubValkeyClient() m := &testMetrics{} lru := NewLRUStorage(64<<20, nil, metrics.Default) @@ -384,8 +384,8 @@ func TestValkeyStorage_SplitFallbackCounters(t *testing.T) { if m.counter("cache.valkey_set_fallback") != 1 { t.Errorf("expected valkey_set_fallback=1, got %d", m.counter("cache.valkey_set_fallback")) } - if m.counter("cache.valkey_get_fallback") != 0 { - t.Errorf("expected valkey_get_fallback=0 after Set, got %d", m.counter("cache.valkey_get_fallback")) + if m.counter("cache.valkey_get_error") != 0 { + t.Errorf("expected valkey_get_error=0 after Set, got %d", m.counter("cache.valkey_get_error")) } // L1-first: the entry was written to L1 by the Set fallback path, so Get returns @@ -394,8 +394,8 @@ func TestValkeyStorage_SplitFallbackCounters(t *testing.T) { if m.counter("cache.l1_hit") != 1 { t.Errorf("expected l1_hit=1, got %d", m.counter("cache.l1_hit")) } - if m.counter("cache.valkey_get_fallback") != 0 { - t.Errorf("expected valkey_get_fallback=0 (L1 served before Valkey check), got %d", m.counter("cache.valkey_get_fallback")) + if m.counter("cache.valkey_get_error") != 0 { + t.Errorf("expected valkey_get_error=0 (L1 served before Valkey check), got %d", m.counter("cache.valkey_get_error")) } if m.counter("cache.valkey_set_fallback") != 1 { t.Errorf("valkey_set_fallback should still be 1, got %d", m.counter("cache.valkey_set_fallback")) From e99ffd88ec2dd87f0e2fb6af133b320716a22c41 Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Thu, 27 Aug 2026 12:58:11 +0200 Subject: [PATCH 71/89] =?UTF-8?q?fix:=20rename=20l1=5Fwarm=5Ffrom=5Fvalkey?= =?UTF-8?q?=20=E2=86=92=20l2=5Fhit?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Signed-off-by: Larry D Almeida --- docs/operation/operation.md | 2 +- filters/cache/valkey_storage.go | 3 ++- filters/cache/valkey_storage_test.go | 29 ++++++++++++++++++++++++++++ 3 files changed, 32 insertions(+), 2 deletions(-) diff --git a/docs/operation/operation.md b/docs/operation/operation.md index 1782f34da7..bc162b68c4 100644 --- a/docs/operation/operation.md +++ b/docs/operation/operation.md @@ -2036,7 +2036,7 @@ falling back to 2 GB if the limit is unreadable. Override with - `cache.valkey_miss`: Counter, Valkey misses that proceeded to an upstream fetch - `cache.valkey_get_error`: Counter, Valkey Get errors — request treated as a cache miss, fetched from origin - `cache.valkey_set_fallback`: Counter, Valkey Set errors — entry written to L1 as fallback -- `cache.l1_warm_from_valkey`: Counter, entries written into L1 after a successful Valkey Get (write-through on read path) +- `cache.l2_hit`: Counter, successful Valkey Get (entry returned from L2); L1 is warmed as a side-effect when `--cache-l1-ttl > 0` **OpenTracing span tags (set on every request when a span is active):** diff --git a/filters/cache/valkey_storage.go b/filters/cache/valkey_storage.go index 72a6ad3ab7..5e6d6105f6 100644 --- a/filters/cache/valkey_storage.go +++ b/filters/cache/valkey_storage.go @@ -40,6 +40,7 @@ type ValkeyStorage struct { // - valkey_miss — clean cache miss (key not found in Valkey) // - valkey_get_error — Valkey error on Get; treated as a cache miss // - valkey_set_fallback — Valkey error on Set; L1 was written instead +// - l2_hit — successful Valkey Get (entry returned from L2) // // Pass metrics.Default when no test-scoped metrics collector is needed. func NewValkeyStorage(ring *skpnet.ValkeyRingClient, l1 *LRUStorage, m metrics.Metrics, l1TTL time.Duration) *ValkeyStorage { @@ -68,6 +69,7 @@ func (s *ValkeyStorage) Get(ctx context.Context, key string) (*Entry, error) { if err := json.Unmarshal([]byte(data), &e); err != nil { return nil, fmt.Errorf("cache: decode valkey entry: %w", err) } + s.metrics.IncCounter("cache.l2_hit") // Write-through: warm L1 so subsequent requests on this process avoid Valkey round-trips. // Use remaining freshness to avoid extending L1 beyond Valkey's actual expiry. if s.l1TTL > 0 && e.TTL > 0 { @@ -76,7 +78,6 @@ func (s *ValkeyStorage) Get(ctx context.Context, key string) (*Entry, error) { warmed.TTL = min(s.l1TTL, remaining) warmed.CreatedAt = time.Now() _ = s.l1.Set(ctx, key, &warmed) - s.metrics.IncCounter("cache.l1_warm_from_valkey") } } return &e, nil diff --git a/filters/cache/valkey_storage_test.go b/filters/cache/valkey_storage_test.go index ca3db2bf51..0c9e40d4e5 100644 --- a/filters/cache/valkey_storage_test.go +++ b/filters/cache/valkey_storage_test.go @@ -367,6 +367,35 @@ func TestValkeyStorage_L1TTL_Zero_DisablesWarming(t *testing.T) { } } +func TestValkeyStorage_RecordsL2Hit(t *testing.T) { + stub := newStubValkeyClient() + m := &testMetrics{} + lru := NewLRUStorage(64<<20, nil, metrics.Default) + s := &ValkeyStorage{ring: stub, l1: lru, metrics: m, l1TTL: 0} // write-around: no L1 warming + + ctx := context.Background() + key := "l2-hit-key" + entry := &Entry{StatusCode: 200, Payload: []byte("v"), TTL: time.Minute, CreatedAt: time.Now()} + + if err := s.Set(ctx, key, entry); err != nil { + t.Fatalf("Set: %v", err) + } + // L1 is not warmed (l1TTL=0), so Get must go to Valkey — a real L2 hit. + got, err := s.Get(ctx, key) + if err != nil { + t.Fatalf("Get: %v", err) + } + if got == nil { + t.Fatal("expected entry from Valkey, got nil") + } + if m.counter("cache.l2_hit") != 1 { + t.Errorf("expected l2_hit=1, got %d", m.counter("cache.l2_hit")) + } + if m.counter("cache.l1_hit") != 0 { + t.Errorf("expected l1_hit=0 (write-around), got %d", m.counter("cache.l1_hit")) + } +} + func TestValkeyStorage_SplitFallbackCounters(t *testing.T) { // Uses a broken stub — no Docker or live Valkey needed. // Set triggers valkey_set_fallback, which writes the entry to L1. From 140f65afafe7b81c201c3329c0993dd87ec2bdf5 Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Thu, 27 Aug 2026 13:04:00 +0200 Subject: [PATCH 72/89] fix: formatting Signed-off-by: Larry D Almeida --- config/config.go | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/config/config.go b/config/config.go index 1b0027eff3..a22a339559 100644 --- a/config/config.go +++ b/config/config.go @@ -354,8 +354,8 @@ type Config struct { SwarmStaticOther string `yaml:"swarm-static-other"` // cache - CacheL1TTL time.Duration `yaml:"cache-l1-ttl"` - CacheL1MaxMemoryBytes int64 `yaml:"cache-l1-max-memory-bytes"` + CacheL1TTL time.Duration `yaml:"cache-l1-ttl"` + CacheL1MaxMemoryBytes int64 `yaml:"cache-l1-max-memory-bytes"` ClusterRatelimitMaxGroupShards int `yaml:"cluster-ratelimit-max-group-shards"` @@ -1219,7 +1219,7 @@ func (c *Config) ToOptions() skipper.Options { SwarmStaticOther: c.SwarmStaticOther, // cache - CacheL1TTL: c.CacheL1TTL, + CacheL1TTL: c.CacheL1TTL, ResponseCacheMaxMemoryBytes: c.CacheL1MaxMemoryBytes, ClusterRatelimitMaxGroupShards: c.ClusterRatelimitMaxGroupShards, From fbde4166c61f4310bc574db55bd084ac77d1f1eb Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Fri, 28 Aug 2026 07:13:40 +0200 Subject: [PATCH 73/89] fix: don't serve stale L1 entries when Valkey may have fresher copy Signed-off-by: Larry D Almeida --- filters/cache/valkey_storage.go | 14 +++++++++++--- 1 file changed, 11 insertions(+), 3 deletions(-) diff --git a/filters/cache/valkey_storage.go b/filters/cache/valkey_storage.go index 5e6d6105f6..42da79575f 100644 --- a/filters/cache/valkey_storage.go +++ b/filters/cache/valkey_storage.go @@ -44,15 +44,23 @@ type ValkeyStorage struct { // // Pass metrics.Default when no test-scoped metrics collector is needed. func NewValkeyStorage(ring *skpnet.ValkeyRingClient, l1 *LRUStorage, m metrics.Metrics, l1TTL time.Duration) *ValkeyStorage { + if l1TTL < 0 { + panic("cache: NewValkeyStorage: l1TTL must be >= 0") + } return &ValkeyStorage{ring: ring, l1: l1, metrics: m, l1TTL: l1TTL} } func (s *ValkeyStorage) Get(ctx context.Context, key string) (*Entry, error) { // L1-first: serve from local memory when the write-through warming populated it. - // LRUStorage.Get returns nil, nil for expired entries — a miss falls through. + // LRUStorage.Get returns entries within TTL + max(StaleIfError, StaleWhileRevalidate). + // Only serve fresh L1 hits; stale entries fall through to Valkey so a fresher copy + // written by another instance is not bypassed. if e, err := s.l1.Get(ctx, key); err == nil && e != nil { - s.metrics.IncCounter("cache.l1_hit") - return e, nil + if !e.IsStale(time.Now()) { + s.metrics.IncCounter("cache.l1_hit") + return e, nil + } + // Stale L1 entry — fall through to Valkey. } data, err := s.ring.Get(ctx, key) From de56cb818c57a8b9cd1869979829ff2b013ec432 Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Fri, 28 Aug 2026 09:52:51 +0200 Subject: [PATCH 74/89] refactor: use isCacheableMethod on fresh HIT conditional path Signed-off-by: Larry D Almeida --- filters/cache/filter.go | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/filters/cache/filter.go b/filters/cache/filter.go index dab38dccf6..ea79655e02 100644 --- a/filters/cache/filter.go +++ b/filters/cache/filter.go @@ -440,7 +440,7 @@ func (f *cacheFilter) Request(ctx filters.FilterContext) { setAgeHeader(rsp, entry, time.Now()) f.metrics.IncCounter("cache.hit") method := ctx.Request().Method - if (method == http.MethodGet || method == http.MethodHead) && evaluateConditionals(ctx.Request(), entry) { + if isCacheableMethod(method) && evaluateConditionals(ctx.Request(), entry) { notModified := &http.Response{ StatusCode: http.StatusNotModified, Header: rsp.Header.Clone(), From 44b321f55111d88dd27883a400a344d81c1dc8f2 Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Fri, 28 Aug 2026 10:03:05 +0200 Subject: [PATCH 75/89] feat: add cache.storage_error counter at all storage error sites Signed-off-by: Larry D Almeida --- filters/cache/filter.go | 9 +++++++++ filters/cache/valkey_storage.go | 1 + 2 files changed, 10 insertions(+) diff --git a/filters/cache/filter.go b/filters/cache/filter.go index ea79655e02..a1c60aee58 100644 --- a/filters/cache/filter.go +++ b/filters/cache/filter.go @@ -538,6 +538,7 @@ func (f *cacheFilter) coalesce(ctx filters.FilterContext, key string) { if shouldStore { if err := f.storage.Set(context.Background(), key, entry); err != nil { log.WithError(err).Warn("cache: Set failed (cold-miss store)") + f.metrics.IncCounter("cache.storage_error") } } return &coalesceResult{entry: entry, stored: sieStored}, nil @@ -611,6 +612,7 @@ func (f *cacheFilter) Response(ctx filters.FilterContext) { } if err := f.storage.Set(ctx.Request().Context(), key, stored); err != nil { log.WithError(err).Warn("cache: Set failed (HEAD freshen)") + f.metrics.IncCounter("cache.storage_error") } } return @@ -626,18 +628,22 @@ func (f *cacheFilter) Response(ctx filters.FilterContext) { if isUnsafeMethod(ctx.Request().Method) && rsp.StatusCode < 400 { if err := f.storage.Delete(ctx.Request().Context(), key); err != nil { log.WithError(err).Warn("cache: Delete failed (unsafe method invalidation)") + f.metrics.IncCounter("cache.storage_error") } if err := f.storage.Delete(ctx.Request().Context(), "vary:"+key); err != nil { log.WithError(err).Warn("cache: Delete failed (vary sentinel invalidation)") + f.metrics.IncCounter("cache.storage_error") } for _, hdrName := range []string{"Location", "Content-Location"} { if loc := rsp.Header.Get(hdrName); loc != "" && sameOrigin(ctx.Request(), loc) { if locKey := cacheKeyForURL(ctx.RouteId(), ctx.Request(), loc, f.keyHeaders); locKey != "" { if err := f.storage.Delete(ctx.Request().Context(), locKey); err != nil { log.WithError(err).Warn("cache: Delete failed (Location invalidation)") + f.metrics.IncCounter("cache.storage_error") } if err := f.storage.Delete(ctx.Request().Context(), "vary:"+locKey); err != nil { log.WithError(err).Warn("cache: Delete failed (vary sentinel for Location)") + f.metrics.IncCounter("cache.storage_error") } } } @@ -704,6 +710,7 @@ func (f *cacheFilter) Response(ctx filters.FilterContext) { } if err := f.storage.Set(ctx.Request().Context(), "vary:"+baseKey, sentinel); err != nil { log.WithError(err).Warn("cache: Set failed (vary sentinel)") + f.metrics.IncCounter("cache.storage_error") } } @@ -737,6 +744,7 @@ func (f *cacheFilter) Response(ctx filters.FilterContext) { } if err := f.storage.Set(ctx.Request().Context(), storeKey, entry); err != nil { log.WithError(err).Warn("cache: Set failed (response store)") + f.metrics.IncCounter("cache.storage_error") } } @@ -852,6 +860,7 @@ func (f *cacheFilter) doRevalidate(key string, req *http.Request) { } if err := f.storage.Set(context.Background(), key, entry); err != nil { log.WithError(err).Warn("cache: Set failed (background revalidation)") + f.metrics.IncCounter("cache.storage_error") } return nil, nil }) diff --git a/filters/cache/valkey_storage.go b/filters/cache/valkey_storage.go index 42da79575f..f202add694 100644 --- a/filters/cache/valkey_storage.go +++ b/filters/cache/valkey_storage.go @@ -128,6 +128,7 @@ func (s *ValkeyStorage) Delete(ctx context.Context, key string) error { // fleet retain their own L1 copies until --cache-l1-ttl expires naturally. if _, err := s.ring.Del(ctx, key); err != nil { log.WithError(err).Warn("cache: valkey Delete failed") + s.metrics.IncCounter("cache.storage_error") } return s.l1.Delete(ctx, key) } From d919e38c106eeab5afbc930969e248ce318b682e Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Fri, 28 Aug 2026 10:13:17 +0200 Subject: [PATCH 76/89] docs: document cache.storage_error counter in operation guide Signed-off-by: Larry D Almeida --- docs/operation/operation.md | 1 + 1 file changed, 1 insertion(+) diff --git a/docs/operation/operation.md b/docs/operation/operation.md index bc162b68c4..87eb87c377 100644 --- a/docs/operation/operation.md +++ b/docs/operation/operation.md @@ -2037,6 +2037,7 @@ falling back to 2 GB if the limit is unreadable. Override with - `cache.valkey_get_error`: Counter, Valkey Get errors — request treated as a cache miss, fetched from origin - `cache.valkey_set_fallback`: Counter, Valkey Set errors — entry written to L1 as fallback - `cache.l2_hit`: Counter, successful Valkey Get (entry returned from L2); L1 is warmed as a side-effect when `--cache-l1-ttl > 0` +- `cache.storage_error`: Counter, any storage operation (Set or Delete) failed — covers both L2 Valkey failures and L1 eviction-path errors; the request is still served correctly **OpenTracing span tags (set on every request when a span is active):** From b718007cd739997e57dbb1e65bd65b78085cc9f7 Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Fri, 28 Aug 2026 10:59:06 +0200 Subject: [PATCH 77/89] docs: clarify --swarm-valkey-urls wires both ratelimit and cache Signed-off-by: Larry D Almeida --- docs/operation/operation.md | 3 +++ 1 file changed, 3 insertions(+) diff --git a/docs/operation/operation.md b/docs/operation/operation.md index 87eb87c377..afffe6ab9c 100644 --- a/docs/operation/operation.md +++ b/docs/operation/operation.md @@ -1968,6 +1968,9 @@ When `--swarm-valkey-urls` is configured, Valkey serves as a shared backing store (L2) accessible by all Skipper instances via a client-side consistent hash ring. Every read checks L1 first; an L1 hit returns without contacting Valkey. +Note: `--swarm-valkey-urls` wires Valkey into both ratelimit and cache — there is currently +no flag to enable L2 cache independently of ratelimit. + On every successful Valkey write the entry is also written to L1 (write-through) with a TTL of `min(--cache-l1-ttl, entry.TTL)`. On a Valkey read hit, L1 is warmed with `min(--cache-l1-ttl, remaining freshness)` — From 923ddc865b5361bcbb541e7a3cf86ab8c8283f00 Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Tue, 1 Sep 2026 16:03:07 +0200 Subject: [PATCH 78/89] add --enable-l2-cache flag to opt in to Valkey as cache L2 backing store Signed-off-by: Larry D Almeida --- config/config.go | 3 +++ docs/operation/operation.md | 11 +++++------ skipper.go | 10 +++++++++- 3 files changed, 17 insertions(+), 7 deletions(-) diff --git a/config/config.go b/config/config.go index a22a339559..2772c0370d 100644 --- a/config/config.go +++ b/config/config.go @@ -356,6 +356,7 @@ type Config struct { // cache CacheL1TTL time.Duration `yaml:"cache-l1-ttl"` CacheL1MaxMemoryBytes int64 `yaml:"cache-l1-max-memory-bytes"` + EnableL2Cache bool `yaml:"enable-l2-cache"` ClusterRatelimitMaxGroupShards int `yaml:"cluster-ratelimit-max-group-shards"` @@ -752,6 +753,7 @@ func NewConfig() *Config { // cache flag.DurationVar(&cfg.CacheL1TTL, "cache-l1-ttl", 60*time.Second, "maximum TTL for write-through L1 warming in the cache() filter when Valkey is configured; set to 0 to disable (write-around)") flag.Int64Var(&cfg.CacheL1MaxMemoryBytes, "cache-l1-max-memory-bytes", 0, "maximum memory budget in bytes for the cache() filter's in-process LRU (L1); defaults to 25% of cgroup memory limit or 2 GB if unreadable") + flag.BoolVar(&cfg.EnableL2Cache, "enable-l2-cache", false, "enable Valkey as L2 backing store for the cache() filter when --swarm-valkey-urls is configured; by default only in-process LRU (L1) is used") flag.IntVar(&cfg.ClusterRatelimitMaxGroupShards, "cluster-ratelimit-max-group-shards", 1, "sets the maximum number of group shards for the clusterRatelimit filter") @@ -1221,6 +1223,7 @@ func (c *Config) ToOptions() skipper.Options { // cache CacheL1TTL: c.CacheL1TTL, ResponseCacheMaxMemoryBytes: c.CacheL1MaxMemoryBytes, + EnableL2Cache: c.EnableL2Cache, ClusterRatelimitMaxGroupShards: c.ClusterRatelimitMaxGroupShards, diff --git a/docs/operation/operation.md b/docs/operation/operation.md index afffe6ab9c..be913c9e3f 100644 --- a/docs/operation/operation.md +++ b/docs/operation/operation.md @@ -1964,12 +1964,11 @@ r: SourceFromLast("9.0.0.0/24","2001:67c:20a0::/48") -> ...` ## Cache By default entries are stored in an in-process LRU (L1) local to each Skipper process. -When `--swarm-valkey-urls` is configured, Valkey serves as a shared backing -store (L2) accessible by all Skipper instances via a client-side consistent hash ring. Every -read checks L1 first; an L1 hit returns without contacting Valkey. - -Note: `--swarm-valkey-urls` wires Valkey into both ratelimit and cache — there is currently -no flag to enable L2 cache independently of ratelimit. +When `--swarm-valkey-urls` is configured and `--enable-l2-cache` is set, Valkey serves as a +shared backing store (L2) accessible by all Skipper instances via a client-side consistent +hash ring; every read checks L1 first and an L1 hit returns without contacting Valkey. +Without `--enable-l2-cache`, `--swarm-valkey-urls` wires Valkey into ratelimit only and the +`cache()` filter uses L1 exclusively. On every successful Valkey write the entry is also written to L1 (write-through) with a TTL of `min(--cache-l1-ttl, entry.TTL)`. On a Valkey diff --git a/skipper.go b/skipper.go index 44077ddb08..6f7ccbe2ac 100644 --- a/skipper.go +++ b/skipper.go @@ -153,6 +153,10 @@ type Options struct { // Set to 0 to disable write-through (write-around behaviour). Default: 60s. CacheL1TTL time.Duration + // EnableL2Cache enables Valkey as the L2 backing store for the cache() filter + // when SwarmValkeyURLs is configured. Without this flag, only in-process LRU (L1) is used. + EnableL2Cache bool + // ReadMemoryLimit, when set, is called by the cache() filter initialiser // to determine the container memory limit. Defaults to reading cgroup files. // Override in tests or on non-standard platforms. @@ -2301,6 +2305,10 @@ func run(o Options, sig chan os.Signal, idleConnsCH chan struct{}) error { } if !slices.Contains(o.DisabledFilters, cache.Name) { + valkeyForCache := valkeyRing + if !o.EnableL2Cache { + valkeyForCache = nil + } cacheSpec := cache.NewCacheFilter( cache.Options{ MaxBytes: o.cacheBudget(), @@ -2313,7 +2321,7 @@ func run(o Options, sig chan os.Signal, idleConnsCH chan struct{}) error { OpentracingSpanName: "cache_revalidation", OpentracingEventsByTag: o.OpenTracingClientTraceByTag, }, - ValkeyRing: valkeyRing, + ValkeyRing: valkeyForCache, L1TTL: o.CacheL1TTL, }, ) From d5b283d65ed8d7c76cc32988fde44c6ceb90c185 Mon Sep 17 00:00:00 2001 From: Larry D Almeida Date: Tue, 1 Sep 2026 16:34:32 +0200 Subject: [PATCH 79/89] rename valkey_miss/get_error/set_fallback metrics to l2_ prefix for consistency Signed-off-by: Larry D Almeida --- docs/operation/operation.md | 10 ++++----- filters/cache/valkey_storage.go | 12 +++++------ filters/cache/valkey_storage_test.go | 32 ++++++++++++++-------------- 3 files changed, 27 insertions(+), 27 deletions(-) diff --git a/docs/operation/operation.md b/docs/operation/operation.md index be913c9e3f..b3a79a1c7b 100644 --- a/docs/operation/operation.md +++ b/docs/operation/operation.md @@ -1977,9 +1977,9 @@ remaining freshness (`entry.TTL - age`) is used rather than the original TTL to prevent L1 from serving the entry beyond Valkey's actual expiry. The default is 60 seconds, bounding how long Skipper serves a locally-cached entry before re-consulting Valkey. Set `--cache-l1-ttl=0` to disable L1 warming and -restore write-around behaviour. Note: on Valkey write errors (`valkey_set_fallback`), L1 +restore write-around behaviour. Note: on Valkey write errors (`l2_set_fallback`), L1 is always used as a fallback regardless of `--cache-l1-ttl`; Valkey read errors -(`valkey_get_error`) are treated as cache misses. +(`l2_get_error`) are treated as cache misses. When an upstream responds successfully to an unsafe method (`POST`, `PUT`, `DELETE`, `PATCH`), the filter removes the cached entry for that URL from both Valkey and the local L1. @@ -2035,9 +2035,9 @@ falling back to 2 GB if the limit is unreadable. Override with **Valkey (when Valkey is configured):** - `cache.l1_hit`: Counter, L1 hits that bypassed Valkey -- `cache.valkey_miss`: Counter, Valkey misses that proceeded to an upstream fetch -- `cache.valkey_get_error`: Counter, Valkey Get errors — request treated as a cache miss, fetched from origin -- `cache.valkey_set_fallback`: Counter, Valkey Set errors — entry written to L1 as fallback +- `cache.l2_miss`: Counter, Valkey misses that proceeded to an upstream fetch +- `cache.l2_get_error`: Counter, Valkey Get errors — request treated as a cache miss, fetched from origin +- `cache.l2_set_fallback`: Counter, Valkey Set errors — entry written to L1 only (not L2) - `cache.l2_hit`: Counter, successful Valkey Get (entry returned from L2); L1 is warmed as a side-effect when `--cache-l1-ttl > 0` - `cache.storage_error`: Counter, any storage operation (Set or Delete) failed — covers both L2 Valkey failures and L1 eviction-path errors; the request is still served correctly diff --git a/filters/cache/valkey_storage.go b/filters/cache/valkey_storage.go index f202add694..d33d8d0a9e 100644 --- a/filters/cache/valkey_storage.go +++ b/filters/cache/valkey_storage.go @@ -37,9 +37,9 @@ type ValkeyStorage struct { // fallback in-memory cache. m is used to record per-operation counters: // // - l1_hit — L1 returned a warm entry; Valkey not consulted -// - valkey_miss — clean cache miss (key not found in Valkey) -// - valkey_get_error — Valkey error on Get; treated as a cache miss -// - valkey_set_fallback — Valkey error on Set; L1 was written instead +// - l2_miss — clean cache miss (key not found in Valkey) +// - l2_get_error — Valkey error on Get; treated as a cache miss +// - l2_set_fallback — Valkey error on Set; entry written to L1 only (not L2) // - l2_hit — successful Valkey Get (entry returned from L2) // // Pass metrics.Default when no test-scoped metrics collector is needed. @@ -66,10 +66,10 @@ func (s *ValkeyStorage) Get(ctx context.Context, key string) (*Entry, error) { data, err := s.ring.Get(ctx, key) if err != nil { if valkey.IsValkeyNil(err) { - s.metrics.IncCounter("cache.valkey_miss") + s.metrics.IncCounter("cache.l2_miss") return nil, nil } - s.metrics.IncCounter("cache.valkey_get_error") + s.metrics.IncCounter("cache.l2_get_error") log.WithError(err).Warn("cache: valkey Get failed, treating as miss") return nil, nil } @@ -103,7 +103,7 @@ func (s *ValkeyStorage) Set(ctx context.Context, key string, entry *Entry) error } if err := s.ring.SetWithExpire(ctx, key, string(data), valkeyTTL); err != nil { - s.metrics.IncCounter("cache.valkey_set_fallback") + s.metrics.IncCounter("cache.l2_set_fallback") log.WithError(err).Warn("cache: valkey Set failed, falling back to L1") return s.l1.Set(ctx, key, entry) } diff --git a/filters/cache/valkey_storage_test.go b/filters/cache/valkey_storage_test.go index 0c9e40d4e5..4090624291 100644 --- a/filters/cache/valkey_storage_test.go +++ b/filters/cache/valkey_storage_test.go @@ -231,8 +231,8 @@ func TestValkeyStorage_FallsBackToL1OnValkeyUnavailable(t *testing.T) { if m.counter("cache.l1_hit") == 0 { t.Error("expected l1_hit to be incremented: Set fallback warmed L1, Get should serve from it") } - if m.counter("cache.valkey_get_error") != 0 { - t.Errorf("expected valkey_get_error=0 (L1 served before Valkey was contacted), got %d", m.counter("cache.valkey_get_error")) + if m.counter("cache.l2_get_error") != 0 { + t.Errorf("expected l2_get_error=0 (L1 served before Valkey was contacted), got %d", m.counter("cache.l2_get_error")) } // Confirm the entry was physically written to L1 — not just returned via some @@ -260,11 +260,11 @@ func TestValkeyStorage_RecordsValkeyMiss(t *testing.T) { if got != nil { t.Fatalf("expected nil on miss, got %+v", got) } - if m.counter("cache.valkey_miss") != 1 { - t.Errorf("expected valkey_miss=1, got %d", m.counter("cache.valkey_miss")) + if m.counter("cache.l2_miss") != 1 { + t.Errorf("expected l2_miss=1, got %d", m.counter("cache.l2_miss")) } - if m.counter("cache.valkey_get_error") != 0 { - t.Errorf("expected valkey_get_error=0 on clean miss, got %d", m.counter("cache.valkey_get_error")) + if m.counter("cache.l2_get_error") != 0 { + t.Errorf("expected l2_get_error=0 on clean miss, got %d", m.counter("cache.l2_get_error")) } } @@ -398,9 +398,9 @@ func TestValkeyStorage_RecordsL2Hit(t *testing.T) { func TestValkeyStorage_SplitFallbackCounters(t *testing.T) { // Uses a broken stub — no Docker or live Valkey needed. - // Set triggers valkey_set_fallback, which writes the entry to L1. + // Set triggers l2_set_fallback, which writes the entry to L1. // Get checks L1 first (L1-first reads) and finds the entry — incrementing l1_hit, - // not valkey_get_error. + // not l2_get_error. stub := newBrokenStubValkeyClient() m := &testMetrics{} lru := NewLRUStorage(64<<20, nil, metrics.Default) @@ -410,11 +410,11 @@ func TestValkeyStorage_SplitFallbackCounters(t *testing.T) { entry := &Entry{StatusCode: 200, Payload: []byte("x"), TTL: time.Minute, CreatedAt: time.Now()} _ = s.Set(ctx, "k", entry) - if m.counter("cache.valkey_set_fallback") != 1 { - t.Errorf("expected valkey_set_fallback=1, got %d", m.counter("cache.valkey_set_fallback")) + if m.counter("cache.l2_set_fallback") != 1 { + t.Errorf("expected l2_set_fallback=1, got %d", m.counter("cache.l2_set_fallback")) } - if m.counter("cache.valkey_get_error") != 0 { - t.Errorf("expected valkey_get_error=0 after Set, got %d", m.counter("cache.valkey_get_error")) + if m.counter("cache.l2_get_error") != 0 { + t.Errorf("expected l2_get_error=0 after Set, got %d", m.counter("cache.l2_get_error")) } // L1-first: the entry was written to L1 by the Set fallback path, so Get returns @@ -423,11 +423,11 @@ func TestValkeyStorage_SplitFallbackCounters(t *testing.T) { if m.counter("cache.l1_hit") != 1 { t.Errorf("expected l1_hit=1, got %d", m.counter("cache.l1_hit")) } - if m.counter("cache.valkey_get_error") != 0 { - t.Errorf("expected valkey_get_error=0 (L1 served before Valkey check), got %d", m.counter("cache.valkey_get_error")) + if m.counter("cache.l2_get_error") != 0 { + t.Errorf("expected l2_get_error=0 (L1 served before Valkey check), got %d", m.counter("cache.l2_get_error")) } - if m.counter("cache.valkey_set_fallback") != 1 { - t.Errorf("valkey_set_fallback should still be 1, got %d", m.counter("cache.valkey_set_fallback")) + if m.counter("cache.l2_set_fallback") != 1 { + t.Errorf("l2_set_fallback should still be 1, got %d", m.counter("cache.l2_set_fallback")) } } From 3024f177de3fa644cecdfab0d12f252bbdca8d53 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Sandor=20Sz=C3=BCcs?= Date: Tue, 1 Sep 2026 20:41:02 +0200 Subject: [PATCH 80/89] refactor: cache filter should not know about the implemtnation of L2 cache other than having the expose interface to make sure it has all its needs. Like this everyone can pass their own implementation MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Signed-off-by: Sandor Szücs --- filters/cache/filter.go | 21 +-- filters/cache/l2_storage.go | 140 ++++++++++++++++++ ...rage_test.go => l2_storage_valkey_test.go} | 18 +-- filters/cache/valkey_storage.go | 134 ----------------- skipper.go | 8 +- 5 files changed, 166 insertions(+), 155 deletions(-) create mode 100644 filters/cache/l2_storage.go rename filters/cache/{valkey_storage_test.go => l2_storage_valkey_test.go} (96%) delete mode 100644 filters/cache/valkey_storage.go diff --git a/filters/cache/filter.go b/filters/cache/filter.go index a1c60aee58..400c018525 100644 --- a/filters/cache/filter.go +++ b/filters/cache/filter.go @@ -51,12 +51,13 @@ const ( // Options configures the cache filter. type Options struct { - MaxBytes int64 // maximum number of bytes the in-process LRU (L1) is allowed to hold across all cached entries - ListenAddr string // Skipper's own address; revalidation requests loop back through it so the full filter chain runs - NetOpts skpnet.Options // HTTP client options for background worker that re-fetches stale entries from origin - ValkeyRing *skpnet.ValkeyRingClient // optional L2 cache; nil = in-process LRU only - L1TTL time.Duration // max TTL for write-through L1 warming; 0 = write-around - Metrics metrics.Metrics // nil defaults to metrics.Default + MaxBytes int64 // maximum number of bytes the in-process LRU (L1) is allowed to hold across all cached entries + ListenAddr string // Skipper's own address; revalidation requests loop back through it so the full filter chain runs + NetOpts skpnet.Options // HTTP client options for background worker that re-fetches stale entries from origin + L2Client L2Client // optional L2 cache; nil = in-process LRU only + IsNoL2Err func(error) bool // returns true if no L2Client error + L1TTL time.Duration // max TTL for write-through L1 warming; 0 = write-around + Metrics metrics.Metrics // nil defaults to metrics.Default } // filterCacheKey identifies a unique cache filter configuration for registry lookup. @@ -75,7 +76,7 @@ type cacheSpec struct { listenAddr string client *skpnet.Client storage Storage // shared across all filter instances - lruStorage *LRUStorage // always non-nil; direct reference to L1, even when storage is ValkeyStorage + lruStorage *LRUStorage // always non-nil; direct reference to L1, even when storage is L2Storage metrics metrics.Metrics revalJobs chan revalJob ctx context.Context @@ -110,8 +111,8 @@ func NewCacheFilter(opts Options) filters.Spec { }, m) var store Storage = lru - if opts.ValkeyRing != nil { - store = NewValkeyStorage(opts.ValkeyRing, lru, m, opts.L1TTL) + if opts.L2Client != nil { + store = NewL2Storage(opts.L2Client, lru, m, opts.L1TTL, opts.IsNoL2Err) } ctx, cancel := context.WithCancel(context.Background()) @@ -286,7 +287,7 @@ type revalJob struct { // would be destructive. Lifecycle is managed exclusively by cacheSpec.Close(). type cacheFilter struct { storage Storage - lruStorage *LRUStorage // always non-nil; direct reference to L1, even when storage is ValkeyStorage + lruStorage *LRUStorage // always non-nil; direct reference to L1, even when storage is L2Storage listenAddr string ttl time.Duration errorTTL time.Duration diff --git a/filters/cache/l2_storage.go b/filters/cache/l2_storage.go new file mode 100644 index 0000000000..7be5ae73f0 --- /dev/null +++ b/filters/cache/l2_storage.go @@ -0,0 +1,140 @@ +package cache + +import ( + "context" + "encoding/json" + "fmt" + "time" + + log "github.com/sirupsen/logrus" + "github.com/zalando/skipper/metrics" +) + +const defaultMinTTL = time.Minute + +type L2Client interface { + Get(ctx context.Context, key string) (string, error) + SetWithExpire(ctx context.Context, key string, value string, expire time.Duration) error + Expire(ctx context.Context, key string, d time.Duration) (int64, error) + Del(ctx context.Context, key string) (int64, error) +} + +// L2Storage implements Storage using a L2Client with +// write-through warming of LRUStorage (L1). On L2Client Set errors, L1 +// is used as a fallback. On L2Client Get errors, the request is treated +// as a miss and fetched from origin. +type L2Storage struct { + l2client L2Client + l1 *LRUStorage + metrics metrics.Metrics + l1TTL time.Duration // max TTL for write-through L1 warming; 0 = write-around + + isNoErr func(error) bool +} + +// NewL2Storage creates a *L2Storage backed by ring (L2) with l1 as the +// fallback in-memory cache. m is used to record per-operation counters: +// +// - l1_hit — L1 returned a warm entry; L2 not consulted +// - l2_miss — clean cache miss (key not found in L2) +// - l2_get_error — L2 error on Get; treated as a cache miss +// - l2_set_fallback — L2 error on Set; entry written to L1 only +// - l2_hit — successful L2 Get (entry returned from L2) +// +// Pass metrics.Default when no test-scoped metrics collector is needed. +func NewL2Storage(l2client L2Client, l1 *LRUStorage, m metrics.Metrics, l1TTL time.Duration, isNoErr func(error) bool) *L2Storage { + if l1TTL < 0 { + log.Warnf("cache: NewL2Storage: l1TTL must be >= 0: %v", l1TTL) + l1TTL = defaultMinTTL + } + return &L2Storage{ + l2client: l2client, + l1: l1, + metrics: m, + l1TTL: l1TTL, + isNoErr: isNoErr, + } +} + +func (s *L2Storage) Get(ctx context.Context, key string) (*Entry, error) { + // L1-first: serve from local memory when the write-through warming populated it. + // LRUStorage.Get returns entries within TTL + max(StaleIfError, StaleWhileRevalidate). + // Only serve fresh L1 hits; stale entries fall through to L2 so a fresher copy + // written by another instance is not bypassed. + if e, err := s.l1.Get(ctx, key); err == nil && e != nil { + if !e.IsStale(time.Now()) { + s.metrics.IncCounter("cache.l1_hit") + return e, nil + } + // Stale L1 entry — fall through to L2. + } + + data, err := s.l2client.Get(ctx, key) + if err != nil { + if s.isNoErr(err) { + s.metrics.IncCounter("cache.l2_miss") + return nil, nil + } + s.metrics.IncCounter("cache.l2_get_error") + log.WithError(err).Error("cache: L2 Get failed, treating as miss") + return nil, nil + } + var e Entry + if err := json.Unmarshal([]byte(data), &e); err != nil { + return nil, fmt.Errorf("cache: decode L2 entry: %w", err) + } + s.metrics.IncCounter("cache.l2_hit") + // Write-through: warm L1 so subsequent requests on this process avoid L2 round-trips. + // Use remaining freshness to avoid extending L1 beyond L2's actual expiry. + if s.l1TTL > 0 && e.TTL > 0 { + if remaining := e.TTL - time.Since(e.CreatedAt); remaining > 0 { + warmed := e + warmed.TTL = min(s.l1TTL, remaining) + warmed.CreatedAt = time.Now() + _ = s.l1.Set(ctx, key, &warmed) + } + } + return &e, nil +} + +func (s *L2Storage) Set(ctx context.Context, key string, entry *Entry) error { + data, err := json.Marshal(entry) + if err != nil { + return fmt.Errorf("cache: encode L2 entry: %w", err) + } + + l2TTL := entry.TTL + max(entry.StaleIfError, entry.StaleWhileRevalidate) + if l2TTL <= 0 { + l2TTL = time.Minute + } + + if err := s.l2client.SetWithExpire(ctx, key, string(data), l2TTL); err != nil { + s.metrics.IncCounter("cache.l2_set_fallback") + log.WithError(err).Error("cache: L2 Set failed, falling back to L1") + return s.l1.Set(ctx, key, entry) + } + + // Write-through: warm L1 with a bounded TTL so pods can serve subsequent + // requests from local memory without a L2 round-trip. + // Skip warming for non-cacheable entries (TTL <= 0) to avoid polluting L1 + // with entries that should not be served. + if s.l1TTL > 0 && entry.TTL > 0 { + warmTTL := min(s.l1TTL, entry.TTL) + warmed := *entry + warmed.TTL = warmTTL + warmed.CreatedAt = time.Now() + _ = s.l1.Set(ctx, key, &warmed) + } + return nil +} + +func (s *L2Storage) Delete(ctx context.Context, key string) error { + // L2 errors are best-effort — L1 delete always runs. + // Note: only the local process's L1 is cleared. Other Skipper processes in the + // fleet retain their own L1 copies until --cache-l1-ttl expires naturally. + if _, err := s.l2client.Del(ctx, key); err != nil { + log.WithError(err).Error("cache: L2 Delete failed") + s.metrics.IncCounter("cache.storage_error") + } + return s.l1.Delete(ctx, key) +} diff --git a/filters/cache/valkey_storage_test.go b/filters/cache/l2_storage_valkey_test.go similarity index 96% rename from filters/cache/valkey_storage_test.go rename to filters/cache/l2_storage_valkey_test.go index 4090624291..513e199f6b 100644 --- a/filters/cache/valkey_storage_test.go +++ b/filters/cache/l2_storage_valkey_test.go @@ -150,7 +150,7 @@ func TestValkeyStorage_GetSetDelete(t *testing.T) { defer ring.Close() lru := NewLRUStorage(64<<20, nil, metrics.Default) - s := NewValkeyStorage(ring, lru, &testMetrics{}, 0) + s := NewL2Storage(ring, lru, &testMetrics{}, 0, valkey.IsValkeyNil) ctx := context.Background() key := "test-key" @@ -203,7 +203,7 @@ func TestValkeyStorage_FallsBackToL1OnValkeyUnavailable(t *testing.T) { lru := NewLRUStorage(64<<20, nil, metrics.Default) m := &testMetrics{} - s := NewValkeyStorage(ring, lru, m, 0) + s := NewL2Storage(ring, lru, m, 0, valkey.IsValkeyNil) // Stop valkey before exercising fallback paths. done() @@ -251,7 +251,7 @@ func TestValkeyStorage_RecordsValkeyMiss(t *testing.T) { stub := newStubValkeyClient() m := &testMetrics{} lru := NewLRUStorage(64<<20, nil, metrics.Default) - s := &ValkeyStorage{ring: stub, l1: lru, metrics: m, l1TTL: 0} + s := &L2Storage{l2client: stub, l1: lru, metrics: m, l1TTL: 0} got, err := s.Get(context.Background(), "nonexistent-key") if err != nil { @@ -272,7 +272,7 @@ func TestValkeyStorage_WriteThroughWarmsL1(t *testing.T) { stub := newStubValkeyClient() m := &testMetrics{} lru := NewLRUStorage(64<<20, nil, metrics.Default) - s := &ValkeyStorage{ring: stub, l1: lru, metrics: m, l1TTL: 60 * time.Second} + s := &L2Storage{l2client: stub, l1: lru, metrics: m, l1TTL: 60 * time.Second} ctx := context.Background() key := "wt-key" @@ -308,7 +308,7 @@ func TestValkeyStorage_WriteThroughWarmsL1(t *testing.T) { func TestValkeyStorage_L1TTLBoundedToEntryTTL(t *testing.T) { stub := newStubValkeyClient() lru := NewLRUStorage(64<<20, nil, metrics.Default) - s := &ValkeyStorage{ring: stub, l1: lru, metrics: &testMetrics{}, l1TTL: 60 * time.Second} + s := &L2Storage{l2client: stub, l1: lru, metrics: &testMetrics{}, l1TTL: 60 * time.Second} ctx := context.Background() key := "bounded-key" @@ -340,7 +340,7 @@ func TestValkeyStorage_L1TTL_Zero_DisablesWarming(t *testing.T) { stub := newStubValkeyClient() m := &testMetrics{} lru := NewLRUStorage(64<<20, nil, metrics.Default) - s := &ValkeyStorage{ring: stub, l1: lru, metrics: m, l1TTL: 0} // write-around + s := &L2Storage{l2client: stub, l1: lru, metrics: m, l1TTL: 0} // write-around ctx := context.Background() key := "no-warm-key" @@ -371,7 +371,7 @@ func TestValkeyStorage_RecordsL2Hit(t *testing.T) { stub := newStubValkeyClient() m := &testMetrics{} lru := NewLRUStorage(64<<20, nil, metrics.Default) - s := &ValkeyStorage{ring: stub, l1: lru, metrics: m, l1TTL: 0} // write-around: no L1 warming + s := &L2Storage{l2client: stub, l1: lru, metrics: m, l1TTL: 0} // write-around: no L1 warming ctx := context.Background() key := "l2-hit-key" @@ -404,7 +404,7 @@ func TestValkeyStorage_SplitFallbackCounters(t *testing.T) { stub := newBrokenStubValkeyClient() m := &testMetrics{} lru := NewLRUStorage(64<<20, nil, metrics.Default) - s := &ValkeyStorage{ring: stub, l1: lru, metrics: m, l1TTL: 0} + s := &L2Storage{l2client: stub, l1: lru, metrics: m, l1TTL: 0} ctx := context.Background() entry := &Entry{StatusCode: 200, Payload: []byte("x"), TTL: time.Minute, CreatedAt: time.Now()} @@ -436,7 +436,7 @@ func TestValkeyStorage_DeleteCleansL1EvenOnValkeyError(t *testing.T) { // regardless of the Expire error from Valkey. stub := newBrokenStubValkeyClient() lru := NewLRUStorage(64<<20, nil, metrics.Default) - s := &ValkeyStorage{ring: stub, l1: lru, metrics: &testMetrics{}, l1TTL: 0} + s := &L2Storage{l2client: stub, l1: lru, metrics: &testMetrics{}, l1TTL: 0} ctx := context.Background() entry := &Entry{StatusCode: 200, Payload: []byte("body"), TTL: time.Minute, CreatedAt: time.Now()} diff --git a/filters/cache/valkey_storage.go b/filters/cache/valkey_storage.go deleted file mode 100644 index d33d8d0a9e..0000000000 --- a/filters/cache/valkey_storage.go +++ /dev/null @@ -1,134 +0,0 @@ -package cache - -import ( - "context" - "encoding/json" - "fmt" - "time" - - log "github.com/sirupsen/logrus" - "github.com/valkey-io/valkey-go" - "github.com/zalando/skipper/metrics" - skpnet "github.com/zalando/skipper/net" -) - -// valkeyClient is the subset of skpnet.ValkeyRingClient methods used by ValkeyStorage. -type valkeyClient interface { - Get(ctx context.Context, key string) (string, error) - SetWithExpire(ctx context.Context, key string, value string, expire time.Duration) error - Expire(ctx context.Context, key string, d time.Duration) (int64, error) - Del(ctx context.Context, key string) (int64, error) -} - -var _ valkeyClient = (*skpnet.ValkeyRingClient)(nil) - -// ValkeyStorage implements Storage using a ValkeyRingClient (L2) with -// write-through warming of LRUStorage (L1). On Valkey Set errors, L1 is -// used as a fallback. On Valkey Get errors, the request is treated as a -// miss and fetched from origin. -type ValkeyStorage struct { - ring valkeyClient - l1 *LRUStorage - metrics metrics.Metrics - l1TTL time.Duration // max TTL for write-through L1 warming; 0 = write-around -} - -// NewValkeyStorage creates a ValkeyStorage backed by ring (L2) with l1 as the -// fallback in-memory cache. m is used to record per-operation counters: -// -// - l1_hit — L1 returned a warm entry; Valkey not consulted -// - l2_miss — clean cache miss (key not found in Valkey) -// - l2_get_error — Valkey error on Get; treated as a cache miss -// - l2_set_fallback — Valkey error on Set; entry written to L1 only (not L2) -// - l2_hit — successful Valkey Get (entry returned from L2) -// -// Pass metrics.Default when no test-scoped metrics collector is needed. -func NewValkeyStorage(ring *skpnet.ValkeyRingClient, l1 *LRUStorage, m metrics.Metrics, l1TTL time.Duration) *ValkeyStorage { - if l1TTL < 0 { - panic("cache: NewValkeyStorage: l1TTL must be >= 0") - } - return &ValkeyStorage{ring: ring, l1: l1, metrics: m, l1TTL: l1TTL} -} - -func (s *ValkeyStorage) Get(ctx context.Context, key string) (*Entry, error) { - // L1-first: serve from local memory when the write-through warming populated it. - // LRUStorage.Get returns entries within TTL + max(StaleIfError, StaleWhileRevalidate). - // Only serve fresh L1 hits; stale entries fall through to Valkey so a fresher copy - // written by another instance is not bypassed. - if e, err := s.l1.Get(ctx, key); err == nil && e != nil { - if !e.IsStale(time.Now()) { - s.metrics.IncCounter("cache.l1_hit") - return e, nil - } - // Stale L1 entry — fall through to Valkey. - } - - data, err := s.ring.Get(ctx, key) - if err != nil { - if valkey.IsValkeyNil(err) { - s.metrics.IncCounter("cache.l2_miss") - return nil, nil - } - s.metrics.IncCounter("cache.l2_get_error") - log.WithError(err).Warn("cache: valkey Get failed, treating as miss") - return nil, nil - } - var e Entry - if err := json.Unmarshal([]byte(data), &e); err != nil { - return nil, fmt.Errorf("cache: decode valkey entry: %w", err) - } - s.metrics.IncCounter("cache.l2_hit") - // Write-through: warm L1 so subsequent requests on this process avoid Valkey round-trips. - // Use remaining freshness to avoid extending L1 beyond Valkey's actual expiry. - if s.l1TTL > 0 && e.TTL > 0 { - if remaining := e.TTL - time.Since(e.CreatedAt); remaining > 0 { - warmed := e - warmed.TTL = min(s.l1TTL, remaining) - warmed.CreatedAt = time.Now() - _ = s.l1.Set(ctx, key, &warmed) - } - } - return &e, nil -} - -func (s *ValkeyStorage) Set(ctx context.Context, key string, entry *Entry) error { - data, err := json.Marshal(entry) - if err != nil { - return fmt.Errorf("cache: encode valkey entry: %w", err) - } - - valkeyTTL := entry.TTL + max(entry.StaleIfError, entry.StaleWhileRevalidate) - if valkeyTTL <= 0 { - valkeyTTL = time.Minute - } - - if err := s.ring.SetWithExpire(ctx, key, string(data), valkeyTTL); err != nil { - s.metrics.IncCounter("cache.l2_set_fallback") - log.WithError(err).Warn("cache: valkey Set failed, falling back to L1") - return s.l1.Set(ctx, key, entry) - } - - // Write-through: warm L1 with a bounded TTL so pods can serve subsequent - // requests from local memory without a Valkey round-trip. - // Skip warming for non-cacheable entries (TTL <= 0) to avoid polluting L1 - // with entries that should not be served. - if s.l1TTL > 0 && entry.TTL > 0 { - warmTTL := min(s.l1TTL, entry.TTL) - warmed := *entry - warmed.TTL = warmTTL - warmed.CreatedAt = time.Now() - _ = s.l1.Set(ctx, key, &warmed) - } - return nil -} - -func (s *ValkeyStorage) Delete(ctx context.Context, key string) error { - // Valkey errors are best-effort — L1 delete always runs. - // Note: only the local process's L1 is cleared. Other Skipper processes in the - // fleet retain their own L1 copies until --cache-l1-ttl expires naturally. - if _, err := s.ring.Del(ctx, key); err != nil { - log.WithError(err).Warn("cache: valkey Delete failed") - s.metrics.IncCounter("cache.storage_error") - } - return s.l1.Delete(ctx, key) -} diff --git a/skipper.go b/skipper.go index 6f7ccbe2ac..31b1600b35 100644 --- a/skipper.go +++ b/skipper.go @@ -24,6 +24,7 @@ import ( ot "github.com/opentracing/opentracing-go" "github.com/prometheus/client_golang/prometheus" log "github.com/sirupsen/logrus" + "github.com/valkey-io/valkey-go" "go.opentelemetry.io/otel" otBridge "go.opentelemetry.io/otel/bridge/opentracing" "go.opentelemetry.io/otel/trace" @@ -90,6 +91,8 @@ const ( const DefaultPluginDir = "./plugins" +var _ cache.L2Client = (*skpnet.ValkeyRingClient)(nil) + // Options to start skipper. type Options struct { // WaitForHealthcheckInterval sets the time that skipper waits @@ -2321,8 +2324,9 @@ func run(o Options, sig chan os.Signal, idleConnsCH chan struct{}) error { OpentracingSpanName: "cache_revalidation", OpentracingEventsByTag: o.OpenTracingClientTraceByTag, }, - ValkeyRing: valkeyForCache, - L1TTL: o.CacheL1TTL, + L2Client: valkeyForCache, + IsNoL2Err: valkey.IsValkeyNil, + L1TTL: o.CacheL1TTL, }, ) defer cacheSpec.(io.Closer).Close() From ee2573236c4e66725a24330c0ed09e0b5600e87e Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Sandor=20Sz=C3=BCcs?= Date: Tue, 1 Sep 2026 20:41:41 +0200 Subject: [PATCH 81/89] build: vet was not part of lint target MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Signed-off-by: Sandor Szücs --- Makefile | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/Makefile b/Makefile index d8e4629198..e8eefbfe2c 100644 --- a/Makefile +++ b/Makefile @@ -134,7 +134,7 @@ fuzz: ## run all fuzz tests $(MAKE) -C fuzz $(MAKECMDGOALS) .PHONY: lint -lint: build staticcheck ## run all linters +lint: build vet staticcheck ## run all linters .PHONY: clean clean: ## clean temporary files and directories From ebd606907ecd9a65e5b737c4a028cf74460fe97f Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Sandor=20Sz=C3=BCcs?= Date: Tue, 1 Sep 2026 20:46:54 +0200 Subject: [PATCH 82/89] refactor: default nil enable sets it MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Signed-off-by: Sandor Szücs --- skipper.go | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/skipper.go b/skipper.go index 31b1600b35..32051463e7 100644 --- a/skipper.go +++ b/skipper.go @@ -2308,9 +2308,9 @@ func run(o Options, sig chan os.Signal, idleConnsCH chan struct{}) error { } if !slices.Contains(o.DisabledFilters, cache.Name) { - valkeyForCache := valkeyRing - if !o.EnableL2Cache { - valkeyForCache = nil + var l2Client cache.L2Client + if o.EnableL2Cache { + l2Client = valkeyRing } cacheSpec := cache.NewCacheFilter( cache.Options{ @@ -2324,7 +2324,7 @@ func run(o Options, sig chan os.Signal, idleConnsCH chan struct{}) error { OpentracingSpanName: "cache_revalidation", OpentracingEventsByTag: o.OpenTracingClientTraceByTag, }, - L2Client: valkeyForCache, + L2Client: l2Client, IsNoL2Err: valkey.IsValkeyNil, L1TTL: o.CacheL1TTL, }, From 91f3bee2ce19f57065943300c240560201600be9 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Sandor=20Sz=C3=BCcs?= Date: Tue, 1 Sep 2026 20:53:02 +0200 Subject: [PATCH 83/89] refactor: set cache client via options and default to valkey ring client if not set MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Signed-off-by: Sandor Szücs --- skipper.go | 14 +++++++++++--- 1 file changed, 11 insertions(+), 3 deletions(-) diff --git a/skipper.go b/skipper.go index 32051463e7..cab9c549d5 100644 --- a/skipper.go +++ b/skipper.go @@ -156,10 +156,15 @@ type Options struct { // Set to 0 to disable write-through (write-around behaviour). Default: 60s. CacheL1TTL time.Duration - // EnableL2Cache enables Valkey as the L2 backing store for the cache() filter - // when SwarmValkeyURLs is configured. Without this flag, only in-process LRU (L1) is used. + // EnableL2Cache enables the L2 Cache. You need to pass + // L2CacheClient, which defaults to a valkey.Ring if not + // passed. Without enabling L2 cache, only in-process LRU (L1) + // is used. EnableL2Cache bool + // L2CacheClient, defaults to valkey github.com/zalando/skipper/net.ValkeyRingClient + L2CacheClient cache.L2Client + // ReadMemoryLimit, when set, is called by the cache() filter initialiser // to determine the container memory limit. Defaults to reading cgroup files. // Override in tests or on non-standard platforms. @@ -2310,7 +2315,10 @@ func run(o Options, sig chan os.Signal, idleConnsCH chan struct{}) error { if !slices.Contains(o.DisabledFilters, cache.Name) { var l2Client cache.L2Client if o.EnableL2Cache { - l2Client = valkeyRing + l2Client = o.L2CacheClient + if l2Client == nil { + l2Client = valkeyRing + } } cacheSpec := cache.NewCacheFilter( cache.Options{ From ee98d3bae80a6f4ab0f1933bb0e233c1e09c8a6f Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Sandor=20Sz=C3=BCcs?= Date: Tue, 1 Sep 2026 21:32:54 +0200 Subject: [PATCH 84/89] fix: test cases should use NewL2.. function MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Signed-off-by: Sandor Szücs --- filters/cache/filter_test.go | 2 +- filters/cache/l2_storage_valkey_test.go | 14 +++++++------- 2 files changed, 8 insertions(+), 8 deletions(-) diff --git a/filters/cache/filter_test.go b/filters/cache/filter_test.go index e13dc16b7d..89171e8b56 100644 --- a/filters/cache/filter_test.go +++ b/filters/cache/filter_test.go @@ -717,7 +717,7 @@ func TestCacheFilter_Metrics(t *testing.T) { } f := fi.(*cacheFilter) f.fetch = func(*http.Request) (*http.Response, error) { return nil, errors.New("no fetch stub set") } - t.Cleanup(func() { spec.(*cacheSpec).Close() }) + defer spec.(*cacheSpec).Close() url := "https://cdn.contentful.com/spaces/abc/entries/metrics" synctest.Test(t, func(t *testing.T) { diff --git a/filters/cache/l2_storage_valkey_test.go b/filters/cache/l2_storage_valkey_test.go index 513e199f6b..8e2f0f8a87 100644 --- a/filters/cache/l2_storage_valkey_test.go +++ b/filters/cache/l2_storage_valkey_test.go @@ -251,7 +251,7 @@ func TestValkeyStorage_RecordsValkeyMiss(t *testing.T) { stub := newStubValkeyClient() m := &testMetrics{} lru := NewLRUStorage(64<<20, nil, metrics.Default) - s := &L2Storage{l2client: stub, l1: lru, metrics: m, l1TTL: 0} + s := NewL2Storage(stub, lru, m, 0, valkey.IsValkeyNil) got, err := s.Get(context.Background(), "nonexistent-key") if err != nil { @@ -272,7 +272,7 @@ func TestValkeyStorage_WriteThroughWarmsL1(t *testing.T) { stub := newStubValkeyClient() m := &testMetrics{} lru := NewLRUStorage(64<<20, nil, metrics.Default) - s := &L2Storage{l2client: stub, l1: lru, metrics: m, l1TTL: 60 * time.Second} + s := NewL2Storage(stub, lru, m, 60*time.Second, valkey.IsValkeyNil) ctx := context.Background() key := "wt-key" @@ -308,7 +308,7 @@ func TestValkeyStorage_WriteThroughWarmsL1(t *testing.T) { func TestValkeyStorage_L1TTLBoundedToEntryTTL(t *testing.T) { stub := newStubValkeyClient() lru := NewLRUStorage(64<<20, nil, metrics.Default) - s := &L2Storage{l2client: stub, l1: lru, metrics: &testMetrics{}, l1TTL: 60 * time.Second} + s := NewL2Storage(stub, lru, &testMetrics{}, 60*time.Second, valkey.IsValkeyNil) ctx := context.Background() key := "bounded-key" @@ -340,7 +340,7 @@ func TestValkeyStorage_L1TTL_Zero_DisablesWarming(t *testing.T) { stub := newStubValkeyClient() m := &testMetrics{} lru := NewLRUStorage(64<<20, nil, metrics.Default) - s := &L2Storage{l2client: stub, l1: lru, metrics: m, l1TTL: 0} // write-around + s := NewL2Storage(stub, lru, m, 0, valkey.IsValkeyNil) ctx := context.Background() key := "no-warm-key" @@ -371,7 +371,7 @@ func TestValkeyStorage_RecordsL2Hit(t *testing.T) { stub := newStubValkeyClient() m := &testMetrics{} lru := NewLRUStorage(64<<20, nil, metrics.Default) - s := &L2Storage{l2client: stub, l1: lru, metrics: m, l1TTL: 0} // write-around: no L1 warming + s := NewL2Storage(stub, lru, m, 0, valkey.IsValkeyNil) ctx := context.Background() key := "l2-hit-key" @@ -404,7 +404,7 @@ func TestValkeyStorage_SplitFallbackCounters(t *testing.T) { stub := newBrokenStubValkeyClient() m := &testMetrics{} lru := NewLRUStorage(64<<20, nil, metrics.Default) - s := &L2Storage{l2client: stub, l1: lru, metrics: m, l1TTL: 0} + s := NewL2Storage(stub, lru, m, 0, valkey.IsValkeyNil) ctx := context.Background() entry := &Entry{StatusCode: 200, Payload: []byte("x"), TTL: time.Minute, CreatedAt: time.Now()} @@ -436,7 +436,7 @@ func TestValkeyStorage_DeleteCleansL1EvenOnValkeyError(t *testing.T) { // regardless of the Expire error from Valkey. stub := newBrokenStubValkeyClient() lru := NewLRUStorage(64<<20, nil, metrics.Default) - s := &L2Storage{l2client: stub, l1: lru, metrics: &testMetrics{}, l1TTL: 0} + s := NewL2Storage(stub, lru, &testMetrics{}, 0, valkey.IsValkeyNil) ctx := context.Background() entry := &Entry{StatusCode: 200, Payload: []byte("body"), TTL: time.Minute, CreatedAt: time.Now()} From 18ffb5010d2c3ad754eb7569c00487c8fc68cbb5 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Sandor=20Sz=C3=BCcs?= Date: Tue, 1 Sep 2026 21:56:47 +0200 Subject: [PATCH 85/89] test: add more test coverage and integration tests with proxytest testing the real proxy with the filter in a route MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Signed-off-by: Sandor Szücs --- filters/cache/filter_test.go | 1277 ++++++++++++++++++++++++++++++++++ 1 file changed, 1277 insertions(+) diff --git a/filters/cache/filter_test.go b/filters/cache/filter_test.go index 89171e8b56..02ce9774d3 100644 --- a/filters/cache/filter_test.go +++ b/filters/cache/filter_test.go @@ -3,8 +3,11 @@ package cache import ( "context" "errors" + "fmt" "io" "net/http" + "net/http/httptest" + "net/url" "strconv" "strings" "sync" @@ -13,8 +16,11 @@ import ( "testing/synctest" "time" + "github.com/zalando/skipper/eskip" + "github.com/zalando/skipper/filters" "github.com/zalando/skipper/filters/filtertest" "github.com/zalando/skipper/metrics/metricstest" + "github.com/zalando/skipper/proxy/proxytest" ) func newTestFilter(t *testing.T, ttl, errorTTL, swrWindow time.Duration, extra ...time.Duration) *cacheFilter { @@ -3014,3 +3020,1274 @@ func Benchmark_malicious_matchesETag(b *testing.B) { matchesETag(ifNoneMatch, "foobar") } } + +// ── storage stubs ────────────────────────────────────────────────────────────── + +type failingSetStorage struct{ Storage } + +func (s *failingSetStorage) Set(_ context.Context, _ string, _ *Entry) error { + return errors.New("set: injected failure") +} + +type failingDeleteStorage struct{ Storage } + +func (s *failingDeleteStorage) Delete(_ context.Context, _ string) error { + return errors.New("delete: injected failure") +} + +// ── Part 1: unit tests for uncovered branches ───────────────────────────────── + +func TestCreateFilter_InvalidArgs_ErrorTTL(t *testing.T) { + spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: "localhost:9090"}) + t.Cleanup(func() { spec.(*cacheSpec).Close() }) + cases := []struct { + name string + args []any + }{ + {"bad errorTTL string", []any{"5m", "bad", "30s"}}, + {"zero errorTTL", []any{"5m", "0s", "30s"}}, + {"non-string errorTTL", []any{"5m", 15, "30s"}}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + if _, err := spec.CreateFilter(tc.args); err == nil { + t.Fatal("expected error for bad errorTTL arg") + } + }) + } +} + +func TestCreateFilter_InvalidArgs_SWR(t *testing.T) { + spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: "localhost:9090"}) + t.Cleanup(func() { spec.(*cacheSpec).Close() }) + cases := []struct { + name string + args []any + }{ + {"bad swrWindow string", []any{"5m", "15s", "bad"}}, + {"zero swrWindow", []any{"5m", "15s", "0s"}}, + {"non-string swrWindow", []any{"5m", "15s", 30}}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + if _, err := spec.CreateFilter(tc.args); err == nil { + t.Fatal("expected error for bad swrWindow arg") + } + }) + } +} + +func TestCreateFilter_InvalidArgs_StaleIfError(t *testing.T) { + spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: "localhost:9090"}) + t.Cleanup(func() { spec.(*cacheSpec).Close() }) + cases := []struct { + name string + args []any + }{ + {"non-string staleIfError", []any{"5m", "15s", "30s", 60}}, + {"bad staleIfError duration", []any{"5m", "15s", "30s", "notaduration"}}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + if _, err := spec.CreateFilter(tc.args); err == nil { + t.Fatal("expected error for bad staleIfError arg") + } + }) + } +} + +func TestCreateFilter_InvalidArgs_KeyHeaders(t *testing.T) { + spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: "localhost:9090"}) + t.Cleanup(func() { spec.(*cacheSpec).Close() }) + // arg 4 (keyHeaders) must be a string + if _, err := spec.CreateFilter([]any{"5m", "15s", "30s", "60s", 42}); err == nil { + t.Fatal("expected error for non-string keyHeaders arg") + } +} + +func TestCacheSpec_Close_Idempotent(t *testing.T) { + spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: "localhost:9090"}) + cs := spec.(*cacheSpec) + if err := cs.Close(); err != nil { + t.Fatalf("first Close: %v", err) + } + if err := cs.Close(); err != nil { + t.Fatalf("second Close: %v", err) + } +} + +func TestCacheFilter_RevalidateHeader_Stripped(t *testing.T) { + f := newTestFilter(t, time.Minute, 15*time.Second, time.Minute) + url := "https://cdn.example.com/revalidate-bypass" + + ctx := newCtx("GET", url, "") + ctx.FRequest.Header.Set(revalidateHeader, "1") + f.Request(ctx) + + // Must not be served from cache (header stripped → cache bypassed for lookup) + if ctx.FServed { + t.Fatal("revalidate-header request must not be served from cache") + } + // Header must be stripped before reaching upstream + if ctx.FRequest.Header.Get(revalidateHeader) != "" { + t.Fatal("X-Cache-Revalidate header must be stripped by Request()") + } +} + +func TestCacheFilter_ContextCancelled_BeforeGet(t *testing.T) { + f := newTestFilter(t, time.Minute, 15*time.Second, time.Minute) + + // Pre-populate the cache so a normal request would HIT + populate := newCtx("GET", "https://cdn.example.com/ctx-cancel", "") + f.Request(populate) + populate.FResponse = upstreamResponseCC(http.StatusOK, `{"data":"v1"}`, "max-age=300") + f.Response(populate) + + // Now issue a request with a pre-cancelled context + cancelCtx, cancel := context.WithCancel(context.Background()) + cancel() // cancel immediately + req, _ := http.NewRequestWithContext(cancelCtx, "GET", "https://cdn.example.com/ctx-cancel", nil) + ctx := &filtertest.Context{ + FRequest: req, + FStateBag: make(map[string]any), + FMetrics: &metricstest.MockMetrics{}, + } + f.Request(ctx) + + // The cancelled-context path returns without calling Serve + if ctx.FServed { + t.Fatal("cancelled context request must not be served") + } +} + +func TestCacheFilter_RFC_NoStore_NotCached_Coalesce(t *testing.T) { + f := newTestFilterRFC(t, time.Minute, 15*time.Second, time.Minute) + url := "https://cdn.example.com/rfc-nostore-coalesce" + + f.fetch = func(req *http.Request) (*http.Response, error) { + return &http.Response{ + StatusCode: http.StatusOK, + Header: http.Header{"Cache-Control": {"no-store"}, "Content-Type": {"application/json"}}, + Body: io.NopCloser(strings.NewReader(`{"data":"private"}`)), + }, nil + } + + ctx1 := newCtx("GET", url, "") + f.Request(ctx1) + if !ctx1.FServed { + t.Fatal("expected response to be served via coalesce even with no-store") + } + + // Second request must NOT be served from cache (entry must not have been stored) + var fetchCount int64 + f.fetch = func(req *http.Request) (*http.Response, error) { + atomic.AddInt64(&fetchCount, 1) + return &http.Response{ + StatusCode: http.StatusOK, + Header: http.Header{"Cache-Control": {"no-store"}, "Content-Type": {"application/json"}}, + Body: io.NopCloser(strings.NewReader(`{"data":"private"}`)), + }, nil + } + ctx2 := newCtx("GET", url, "") + f.Request(ctx2) + if fetchCount == 0 { + t.Fatal("no-store response must not be cached; second request must hit upstream") + } +} + +func TestCacheFilter_RFC_Private_NotCached_Coalesce(t *testing.T) { + f := newTestFilterRFC(t, time.Minute, 15*time.Second, time.Minute) + url := "https://cdn.example.com/rfc-private-coalesce" + + f.fetch = func(req *http.Request) (*http.Response, error) { + return &http.Response{ + StatusCode: http.StatusOK, + Header: http.Header{"Cache-Control": {"private, max-age=300"}, "Content-Type": {"application/json"}}, + Body: io.NopCloser(strings.NewReader(`{"data":"user-specific"}`)), + }, nil + } + + ctx1 := newCtx("GET", url, "") + f.Request(ctx1) + if !ctx1.FServed { + t.Fatal("expected response to be served via coalesce even with private") + } + + var fetchCount int64 + f.fetch = func(req *http.Request) (*http.Response, error) { + atomic.AddInt64(&fetchCount, 1) + return &http.Response{ + StatusCode: http.StatusOK, + Header: http.Header{"Cache-Control": {"private, max-age=300"}, "Content-Type": {"application/json"}}, + Body: io.NopCloser(strings.NewReader(`{"data":"user-specific"}`)), + }, nil + } + ctx2 := newCtx("GET", url, "") + f.Request(ctx2) + if fetchCount == 0 { + t.Fatal("private response must not be cached; second request must hit upstream") + } +} + +func TestCacheFilter_CoalesceSetFailure_Served(t *testing.T) { + mockMetrics := &metricstest.MockMetrics{} + spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: "localhost:9090", Metrics: mockMetrics}) + fi, err := spec.CreateFilter([]interface{}{"1m", "15s", "1m"}) + if err != nil { + t.Fatal(err) + } + f := fi.(*cacheFilter) + t.Cleanup(func() { spec.(*cacheSpec).Close() }) + + // Wrap storage so Set always fails + f.storage = &failingSetStorage{f.storage} + + f.fetch = func(req *http.Request) (*http.Response, error) { + return &http.Response{ + StatusCode: http.StatusOK, + Header: http.Header{"Cache-Control": {"max-age=300"}, "Content-Type": {"application/json"}}, + Body: io.NopCloser(strings.NewReader(`{"data":"v1"}`)), + }, nil + } + + ctx := newCtx("GET", "https://cdn.example.com/coalesce-set-fail", "") + f.Request(ctx) + + // Response must still be served despite storage failure + if !ctx.FServed { + t.Fatal("expected response to be served even when storage Set fails") + } + mockMetrics.WithCounters(func(counters map[string]int64) { + if counters["cache.storage_error"] == 0 { + t.Error("expected cache.storage_error to be incremented on Set failure in coalesce") + } + }) +} + +func TestCacheFilter_HEAD_Freshen_BodyHeaderSkipped(t *testing.T) { + f := newTestFilter(t, time.Minute, 15*time.Second, time.Hour) + url := "https://cdn.example.com/head-freshen-body-header" + + // Populate via GET (coalesce path) + f.fetch = func(r *http.Request) (*http.Response, error) { + rsp := upstreamResponseCC(http.StatusOK, `{"data":"v1"}`, "public, max-age=300") + return rsp, nil + } + ctx1 := newCtx("GET", url, "") + f.Request(ctx1) + + // HEAD request: served from cache + headCtx := newCtx("HEAD", url, "") + f.Request(headCtx) + if !headCtx.FServed { + t.Fatal("expected HEAD to be served from cache") + } + + // HEAD response contains Content-Length (body-related header) — must NOT update stored entry + headCtx.FResponse = &http.Response{ + StatusCode: http.StatusOK, + Header: http.Header{ + "Content-Length": {"999"}, + "Cache-Control": {"public, max-age=300"}, + }, + Body: http.NoBody, + } + f.Response(headCtx) + + // Stored entry must NOT have the Content-Length from the HEAD response + key := cacheKey(headCtx.FRouteId, headCtx.FRequest, nil) + entry, err := f.storage.Get(headCtx.FRequest.Context(), key) + if err != nil || entry == nil { + t.Fatal("expected stored entry after GET") + } + if entry.Header.Get("Content-Length") == "999" { + t.Error("Content-Length (body-related header) must not be updated by HEAD response freshen") + } +} + +func TestCacheFilter_HEAD_Freshen_SetError(t *testing.T) { + mockMetrics := &metricstest.MockMetrics{} + spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: "localhost:9090", Metrics: mockMetrics}) + fi, err := spec.CreateFilter([]interface{}{"1m", "15s", "1m"}) + if err != nil { + t.Fatal(err) + } + f := fi.(*cacheFilter) + t.Cleanup(func() { spec.(*cacheSpec).Close() }) + + url := "https://cdn.example.com/head-freshen-set-err" + + // Populate via GET response path + ctx1 := newCtx("GET", url, "") + f.Request(ctx1) + ctx1.FResponse = upstreamResponseCC(http.StatusOK, `{"data":"v1"}`, "public, max-age=300") + f.Response(ctx1) + + // Now wrap storage so Set fails + f.storage = &failingSetStorage{f.storage} + + // HEAD 200 freshen — storage.Set will fail + headCtx := newCtx("HEAD", url, "") + f.Request(headCtx) + headCtx.FResponse = &http.Response{ + StatusCode: http.StatusOK, + Header: http.Header{"Cache-Control": {"public, max-age=300"}}, + Body: http.NoBody, + } + f.Response(headCtx) // must not panic + + mockMetrics.WithCounters(func(counters map[string]int64) { + if counters["cache.storage_error"] == 0 { + t.Error("expected cache.storage_error incremented on HEAD freshen Set failure") + } + }) +} + +func TestCacheFilter_Response_ServedFromCache_IsNoop(t *testing.T) { + f := newTestFilter(t, time.Minute, 15*time.Second, time.Minute) + url := "https://cdn.example.com/response-served-noop" + + // Populate cache + ctx1 := newCtx("GET", url, "") + f.Request(ctx1) + ctx1.FResponse = upstreamResponseCC(http.StatusOK, `{"data":"v1"}`, "max-age=300") + f.Response(ctx1) + + // Second request: HIT — FServed = true + ctx2 := newCtx("GET", url, "") + f.Request(ctx2) + if !ctx2.FServed { + t.Fatal("expected HIT") + } + + // Calling Response() when already served must be a no-op (no panic, no double write) + ctx2.FResponse = upstreamResponseCC(http.StatusOK, `{"data":"v2"}`, "max-age=300") + f.Response(ctx2) // must return early at ctx.Served() check + + // The stored entry must still have the original payload + key := ctx1.StateBag()[stateBagKey].(string) + entry, err := f.storage.Get(context.Background(), key) + if err != nil || entry == nil { + t.Fatal("expected entry to remain in storage") + } + if string(entry.Payload) != `{"data":"v1"}` { + t.Fatalf("Response() must not overwrite entry when ctx is already served; got %q", string(entry.Payload)) + } +} + +func TestCacheFilter_UnsafeMethod_DeleteError_ContinuesSilently(t *testing.T) { + mockMetrics := &metricstest.MockMetrics{} + spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: "localhost:9090", Metrics: mockMetrics}) + fi, err := spec.CreateFilter([]interface{}{"1m", "15s", "1m"}) + if err != nil { + t.Fatal(err) + } + f := fi.(*cacheFilter) + t.Cleanup(func() { spec.(*cacheSpec).Close() }) + + url := "https://cdn.example.com/unsafe-del-err" + + // Populate cache + ctx1 := newCtx("GET", url, "") + f.Request(ctx1) + ctx1.FResponse = upstreamResponseCC(http.StatusOK, `{"data":"v1"}`, "public, max-age=300") + f.Response(ctx1) + + // Wrap storage so Delete fails + f.storage = &failingDeleteStorage{f.storage} + + // POST to same URL — Delete will fail + postCtx := newCtx("POST", url, "") + f.Request(postCtx) + postCtx.FResponse = &http.Response{ + StatusCode: http.StatusOK, + Header: http.Header{}, + Body: http.NoBody, + } + f.Response(postCtx) // must not panic + + mockMetrics.WithCounters(func(counters map[string]int64) { + if counters["cache.storage_error"] == 0 { + t.Error("expected cache.storage_error incremented on Delete failure in unsafe method invalidation") + } + }) +} + +type errReader struct{} + +func (errReader) Read([]byte) (int, error) { return 0, errors.New("read: injected failure") } +func (errReader) Close() error { return nil } + +func TestCacheFilter_Response_BodyReadError_NotCached(t *testing.T) { + f := newTestFilter(t, time.Minute, 15*time.Second, time.Minute) + url := "https://cdn.example.com/body-read-err" + + ctx := newCtx("GET", url, "") + f.Request(ctx) + + // Response with an erroring body reader + ctx.FResponse = &http.Response{ + StatusCode: http.StatusOK, + Header: http.Header{"Cache-Control": {"max-age=300"}}, + Body: errReader{}, + } + f.Response(ctx) // must not panic + + // Entry must not be stored + key := ctx.StateBag()[stateBagKey].(string) + entry, err := f.storage.Get(context.Background(), key) + if err != nil { + t.Fatalf("storage.Get: %v", err) + } + if entry != nil { + t.Fatal("entry must not be stored when body read fails") + } +} + +func TestCacheFilter_Response_VarySentinelSetError(t *testing.T) { + mockMetrics := &metricstest.MockMetrics{} + spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: "localhost:9090", Metrics: mockMetrics}) + fi, err := spec.CreateFilter([]interface{}{"1m", "15s", "1m"}) + if err != nil { + t.Fatal(err) + } + f := fi.(*cacheFilter) + t.Cleanup(func() { spec.(*cacheSpec).Close() }) + + // Wrap storage so Set fails + f.storage = &failingSetStorage{f.storage} + + ctx := newCtx("GET", "https://cdn.example.com/vary-set-err", "") + ctx.FRequest.Header.Set("Accept-Language", "en-US") + f.Request(ctx) + + rsp := upstreamResponseCC(http.StatusOK, `{"lang":"en"}`, "max-age=300") + rsp.Header.Set("Vary", "Accept-Language") + ctx.FResponse = rsp + f.Response(ctx) // must not panic; vary sentinel Set will fail + + mockMetrics.WithCounters(func(counters map[string]int64) { + if counters["cache.storage_error"] == 0 { + t.Error("expected cache.storage_error incremented on Vary sentinel Set failure") + } + }) +} + +func TestCacheFilter_Response_StorageSetError(t *testing.T) { + mockMetrics := &metricstest.MockMetrics{} + spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: "localhost:9090", Metrics: mockMetrics}) + fi, err := spec.CreateFilter([]interface{}{"1m", "15s", "1m"}) + if err != nil { + t.Fatal(err) + } + f := fi.(*cacheFilter) + t.Cleanup(func() { spec.(*cacheSpec).Close() }) + + f.storage = &failingSetStorage{f.storage} + + ctx := newCtx("GET", "https://cdn.example.com/response-set-err", "") + f.Request(ctx) + ctx.FResponse = upstreamResponseCC(http.StatusOK, `{"data":"v1"}`, "max-age=300") + f.Response(ctx) // must not panic + + mockMetrics.WithCounters(func(counters map[string]int64) { + if counters["cache.storage_error"] == 0 { + t.Error("expected cache.storage_error incremented on Response() storage Set failure") + } + }) +} + +func TestCacheFilter_Revalidate_304_EntryEvicted(t *testing.T) { + synctest.Test(t, func(t *testing.T) { + f := newTestFilter(t, time.Millisecond, 15*time.Second, time.Hour) + url := "https://cdn.example.com/reval-304-evicted" + + ctx1 := newCtx("GET", url, "") + f.Request(ctx1) + rsp := upstreamResponseCC(http.StatusOK, `{"data":"v1"}`, "max-age=300") + rsp.Header.Set("ETag", `"v1"`) + ctx1.FResponse = rsp + f.Response(ctx1) + + // Evict the entry from storage before revalidation fires + key := ctx1.StateBag()[stateBagKey].(string) + if err := f.storage.Delete(context.Background(), key); err != nil { + t.Fatalf("Delete: %v", err) + } + + // Set fetch to return 304 — doRevalidate will find the entry is gone + f.fetch = func(req *http.Request) (*http.Response, error) { + return &http.Response{ + StatusCode: http.StatusNotModified, + Header: http.Header{"ETag": {`"v1"`}}, + Body: http.NoBody, + }, nil + } + + // Advance past TTL to enter SWR window → triggers stale serve + background revalidation + time.Sleep(2 * time.Millisecond) + + // Insert the entry back so the stale serve can find it + staleEntry := &Entry{ + StatusCode: http.StatusOK, + Header: http.Header{"Content-Type": {"application/json"}, "ETag": {`"v1"`}}, + Payload: []byte(`{"data":"v1"}`), + CreatedAt: time.Now().Add(-2 * time.Millisecond), + TTL: time.Millisecond, + StaleWhileRevalidate: time.Hour, + ETag: `"v1"`, + } + if err := f.storage.Set(context.Background(), key, staleEntry); err != nil { + t.Fatalf("Set stale entry: %v", err) + } + + ctx2 := newCtx("GET", url, "") + f.Request(ctx2) + synctest.Wait() + + if !ctx2.FServed { + t.Fatal("expected stale entry served while revalidation fires in background") + } + // The 304 with evicted-entry path must not panic; doRevalidate logs and returns + }) +} + +func TestCacheFilter_Revalidate_BodyReadError(t *testing.T) { + synctest.Test(t, func(t *testing.T) { + f := newTestFilter(t, time.Millisecond, 15*time.Second, time.Hour) + mockMetrics := &metricstest.MockMetrics{} + f.metrics = mockMetrics + url := "https://cdn.example.com/reval-body-err" + + ctx1 := newCtx("GET", url, "") + f.Request(ctx1) + ctx1.FResponse = upstreamResponseCC(http.StatusOK, `{"data":"v1"}`, "max-age=300") + f.Response(ctx1) + + // Fetch returns a 200 with an erroring body + f.fetch = func(req *http.Request) (*http.Response, error) { + return &http.Response{ + StatusCode: http.StatusOK, + Header: http.Header{"Cache-Control": {"max-age=300"}}, + Body: errReader{}, + }, nil + } + + time.Sleep(2 * time.Millisecond) + + ctx2 := newCtx("GET", url, "") + f.Request(ctx2) + synctest.Wait() + + if !ctx2.FServed { + t.Fatal("expected stale to be served") + } + mockMetrics.WithCounters(func(counters map[string]int64) { + if counters["cache.reval_error"] == 0 { + t.Error("expected cache.reval_error incremented on body read failure in doRevalidate") + } + }) + }) +} + +func TestCacheFilter_Revalidate_RFC_NoStore_NotStored(t *testing.T) { + synctest.Test(t, func(t *testing.T) { + f := newTestFilterRFC(t, time.Millisecond, 15*time.Second, time.Hour) + url := "https://cdn.example.com/reval-rfc-nostore" + + // Seed a cacheable entry first + f.fetch = func(req *http.Request) (*http.Response, error) { + return &http.Response{ + StatusCode: http.StatusOK, + Header: http.Header{"Cache-Control": {"max-age=10"}, "Content-Type": {"application/json"}}, + Body: io.NopCloser(strings.NewReader(`{"data":"v1"}`)), + }, nil + } + ctx1 := newCtx("GET", url, "") + f.Request(ctx1) + synctest.Wait() + + key := ctx1.StateBag()[stateBagKey].(string) + if _, err := f.storage.Get(ctx1.FRequest.Context(), key); err != nil { + t.Fatalf("storage.Get: %v", err) + } + + // Background revalidation returns no-store + f.fetch = func(req *http.Request) (*http.Response, error) { + return &http.Response{ + StatusCode: http.StatusOK, + Header: http.Header{"Cache-Control": {"no-store"}, "Content-Type": {"application/json"}}, + Body: io.NopCloser(strings.NewReader(`{"data":"v2"}`)), + }, nil + } + + time.Sleep(2 * time.Millisecond) + + ctx2 := newCtx("GET", url, "") + f.Request(ctx2) + synctest.Wait() + + // Must not panic; doRevalidate simply returns without storing + if !ctx2.FServed { + t.Fatal("expected stale to be served even if revalidation returns no-store") + } + }) +} + +func TestCacheFilter_Revalidate_SetError_MetricIncremented(t *testing.T) { + synctest.Test(t, func(t *testing.T) { + mockMetrics := &metricstest.MockMetrics{} + spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: "localhost:9090", Metrics: mockMetrics}) + fi, err := spec.CreateFilter([]interface{}{time.Millisecond.String(), "15s", time.Hour.String()}) + if err != nil { + t.Fatal(err) + } + f := fi.(*cacheFilter) + t.Cleanup(func() { spec.(*cacheSpec).Close() }) + + url := "https://cdn.example.com/reval-set-err" + + ctx1 := newCtx("GET", url, "") + f.Request(ctx1) + ctx1.FResponse = upstreamResponseCC(http.StatusOK, `{"data":"v1"}`, "max-age=300") + f.Response(ctx1) + + // Now wrap storage so Set fails + f.storage = &failingSetStorage{f.storage} + f.lruStorage = NewLRUStorage(1<<20, nil, mockMetrics) // won't be hit, but keep it valid + + f.fetch = func(req *http.Request) (*http.Response, error) { + return &http.Response{ + StatusCode: http.StatusOK, + Header: http.Header{"Cache-Control": {"max-age=300"}, "Content-Type": {"application/json"}}, + Body: io.NopCloser(strings.NewReader(`{"data":"v2"}`)), + }, nil + } + + time.Sleep(2 * time.Millisecond) + + ctx2 := newCtx("GET", url, "") + f.Request(ctx2) + synctest.Wait() + + mockMetrics.WithCounters(func(counters map[string]int64) { + if counters["cache.storage_error"] == 0 { + t.Error("expected cache.storage_error incremented on Set failure in doRevalidate") + } + }) + }) +} + +func TestEvaluateConditionals_InvalidIMS(t *testing.T) { + req, _ := http.NewRequest("GET", "https://cdn.example.com/path", nil) + req.Header.Set("If-Modified-Since", "not-a-date") + entry := &Entry{LastModified: "Wed, 21 Oct 2015 07:28:00 GMT"} + if evaluateConditionals(req, entry) { + t.Fatal("invalid If-Modified-Since must return false") + } +} + +func TestEvaluateConditionals_InvalidLM(t *testing.T) { + req, _ := http.NewRequest("GET", "https://cdn.example.com/path", nil) + req.Header.Set("If-Modified-Since", "Wed, 21 Oct 2015 07:28:00 GMT") + entry := &Entry{LastModified: "not-a-date"} + if evaluateConditionals(req, entry) { + t.Fatal("invalid Last-Modified in entry must return false") + } +} + +func TestMatchesETag_EmptyETag(t *testing.T) { + if matchesETag(`"abc"`, "") { + t.Fatal("empty etag must not match") + } +} + +func TestMatchesETag_WildcardMatch(t *testing.T) { + if !matchesETag("*", `"any-etag"`) { + t.Fatal("wildcard * must match any etag") + } +} + +func TestMatchesETag_WeakComparison(t *testing.T) { + // W/"etag" and "etag" are equal under weak comparison + if !matchesETag(`W/"abc"`, `"abc"`) { + t.Fatal("W/ prefix must be stripped for weak comparison") + } + if !matchesETag(`"abc"`, `W/"abc"`) { + t.Fatal("W/ prefix in etag must be stripped for weak comparison") + } + if matchesETag(`"abc"`, `"xyz"`) { + t.Fatal("different etags must not match") + } +} + +func TestHeuristicTTL_WithExplicitDirective_ReturnsZero(t *testing.T) { + h := http.Header{} + h.Set("Cache-Control", "max-age=300") + h.Set("Last-Modified", "Wed, 21 Oct 2015 07:28:00 GMT") + d := parseCacheControl(h) + if got := heuristicTTL(h, d, time.Now()); got != 0 { + t.Fatalf("heuristic must return 0 when max-age present, got %v", got) + } +} + +func TestHeuristicTTL_BadLastModified(t *testing.T) { + h := http.Header{} + h.Set("Last-Modified", "not-a-date") + d := cacheDirectives{maxAge: -1, sMaxAge: -1} + if got := heuristicTTL(h, d, time.Now()); got != 0 { + t.Fatalf("heuristic must return 0 for bad Last-Modified, got %v", got) + } +} + +func TestHeuristicTTL_BadDate_FallsBackToNow(t *testing.T) { + h := http.Header{} + h.Set("Last-Modified", "Wed, 21 Oct 2015 07:28:00 GMT") // 10+ years ago + h.Set("Date", "not-a-date") // bad Date → falls back to time.Now() + d := cacheDirectives{maxAge: -1, sMaxAge: -1} + got := heuristicTTL(h, d, time.Now()) + // age = now - 2015 ≈ huge; 0.1 * huge > 0 + if got <= 0 { + t.Fatalf("heuristic with bad Date and old Last-Modified must be > 0, got %v", got) + } +} + +func TestHeuristicTTL_NegativeAge_ReturnsZero(t *testing.T) { + // Last-Modified in the future → age negative → heuristic returns 0 + h := http.Header{} + h.Set("Last-Modified", time.Now().Add(time.Hour).UTC().Format(http.TimeFormat)) + d := cacheDirectives{maxAge: -1, sMaxAge: -1} + if got := heuristicTTL(h, d, time.Now()); got != 0 { + t.Fatalf("heuristic with future Last-Modified must return 0, got %v", got) + } +} + +func TestCapTTLByExpires_MaxAgePresent_IgnoresExpires(t *testing.T) { + h := http.Header{} + h.Set("Expires", time.Now().Add(time.Second).UTC().Format(http.TimeFormat)) + d := cacheDirectives{maxAge: 300, sMaxAge: -1} + // TTL must not be capped: max-age present means Expires is ignored + if got := capTTLByExpires(5*time.Minute, h, d); got != 5*time.Minute { + t.Fatalf("expected TTL unchanged when max-age present, got %v", got) + } +} + +func TestCapTTLByExpires_NoExpires(t *testing.T) { + d := cacheDirectives{maxAge: -1, sMaxAge: -1} + if got := capTTLByExpires(5*time.Minute, http.Header{}, d); got != 5*time.Minute { + t.Fatalf("expected TTL unchanged when no Expires, got %v", got) + } +} + +func TestCapTTLByExpires_InvalidDate(t *testing.T) { + h := http.Header{} + h.Set("Expires", "0") + d := cacheDirectives{maxAge: -1, sMaxAge: -1} + if got := capTTLByExpires(5*time.Minute, h, d); got != 0 { + t.Fatalf("expected TTL=0 for invalid Expires date, got %v", got) + } +} + +func TestCapTTLByExpires_AlreadyExpired(t *testing.T) { + h := http.Header{} + h.Set("Expires", time.Now().Add(-time.Minute).UTC().Format(http.TimeFormat)) + d := cacheDirectives{maxAge: -1, sMaxAge: -1} + if got := capTTLByExpires(5*time.Minute, h, d); got != 0 { + t.Fatalf("expected TTL=0 for past Expires, got %v", got) + } +} + +func TestCapTTLByExpires_Caps(t *testing.T) { + // Expires is 10s from now; TTL=0 means uncapped → must return remaining ~10s + h := http.Header{} + h.Set("Expires", time.Now().Add(10*time.Second).UTC().Format(http.TimeFormat)) + d := cacheDirectives{maxAge: -1, sMaxAge: -1} + got := capTTLByExpires(0, h, d) + if got <= 0 || got > 11*time.Second { + t.Fatalf("expected TTL ~10s for uncapped Expires, got %v", got) + } +} + +func TestCapTTLByExpires_ReturnsTTLWhenSmaller(t *testing.T) { + // Expires is far in the future; TTL=1s < remaining → TTL wins + h := http.Header{} + h.Set("Expires", time.Now().Add(time.Hour).UTC().Format(http.TimeFormat)) + d := cacheDirectives{maxAge: -1, sMaxAge: -1} + if got := capTTLByExpires(time.Second, h, d); got != time.Second { + t.Fatalf("expected TTL=1s (smaller than Expires remaining), got %v", got) + } +} + +func TestVaryKey_EmptyHeaders_ReturnsBase(t *testing.T) { + req, _ := http.NewRequest("GET", "https://cdn.example.com/path", nil) + base := "some-base-key" + if got := varyKey(base, req, nil); got != base { + t.Fatalf("varyKey with nil varyHeaders must return base key, got %q", got) + } + if got := varyKey(base, req, []string{}); got != base { + t.Fatalf("varyKey with empty varyHeaders must return base key, got %q", got) + } +} + +func TestCacheKeyForURL_InvalidURL(t *testing.T) { + base, _ := http.NewRequest("GET", "https://cdn.example.com/path", nil) + got := cacheKeyForURL("route", base, "://invalid url\x00", nil) + if got != "" { + t.Fatalf("expected empty string for invalid URL, got %q", got) + } +} + +func TestCacheKeyForURL_RelativeURL_UsesBaseHost(t *testing.T) { + base, _ := http.NewRequest("GET", "https://cdn.example.com/path", nil) + base.Host = "cdn.example.com" + got := cacheKeyForURL("route", base, "/other/path", nil) + if got == "" { + t.Fatal("expected non-empty key for relative URL") + } + // The key for absolute same-origin must match + abs := cacheKeyForURL("route", base, "https://cdn.example.com/other/path", nil) + if got != abs { + t.Fatalf("relative and absolute same-origin URL must produce the same key; got %q vs %q", got, abs) + } +} + +// ── Part 2: proxytest L1/L2 integration tests ───────────────────────────────── + +// newProxyCacheRoute builds an eskip route for the proxytest proxy that applies +// the cache filter. backendURL is the httptest backend. +func newProxyCacheRoute(t *testing.T, backendURL string, spec filters.Spec, args ...string) *eskip.Route { + t.Helper() + var filterArgs []interface{} + for _, a := range args { + filterArgs = append(filterArgs, a) + } + return &eskip.Route{ + Id: "cache-route", + PathRegexps: []string{".*"}, + Filters: []*eskip.Filter{{Name: spec.Name(), Args: filterArgs}}, + Backend: backendURL, + } +} + +func TestProxy_L1Cache_MissAndHit(t *testing.T) { + var backendCallCount int64 + backend := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + atomic.AddInt64(&backendCallCount, 1) + w.Header().Set("Cache-Control", "public, max-age=300") + w.Header().Set("Content-Type", "application/json") + fmt.Fprint(w, `{"data":"v1"}`) + })) + defer backend.Close() + + spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: "127.0.0.1:0"}) + t.Cleanup(func() { spec.(*cacheSpec).Close() }) + + fr := make(filters.Registry) + fr.Register(spec) + + proxy := proxytest.New(fr, newProxyCacheRoute(t, backend.URL, spec, "5m", "15s", "30s")) + defer proxy.Close() + + // Spec must know the proxy's listen addr for revalidation loop-back + spec.(*cacheSpec).listenAddr = strings.TrimPrefix(proxy.URL, "http://") + for _, f := range spec.(*cacheSpec).filters { + f.listenAddr = spec.(*cacheSpec).listenAddr + } + + client := proxy.Client() + + // First request: MISS — backend must be called + rsp1, err := client.Get(proxy.URL + "/items/1") + if err != nil { + t.Fatalf("first request: %v", err) + } + rsp1.Body.Close() + if rsp1.Header.Get("X-Cache-Status") != "MISS" { + t.Fatalf("expected MISS on first request, got %q", rsp1.Header.Get("X-Cache-Status")) + } + if atomic.LoadInt64(&backendCallCount) != 1 { + t.Fatalf("expected 1 backend call, got %d", atomic.LoadInt64(&backendCallCount)) + } + + // Second request: HIT — backend must NOT be called + rsp2, err := client.Get(proxy.URL + "/items/1") + if err != nil { + t.Fatalf("second request: %v", err) + } + rsp2.Body.Close() + if rsp2.Header.Get("X-Cache-Status") != "HIT" { + t.Fatalf("expected HIT on second request, got %q", rsp2.Header.Get("X-Cache-Status")) + } + if atomic.LoadInt64(&backendCallCount) != 1 { + t.Fatalf("expected still 1 backend call after HIT, got %d", atomic.LoadInt64(&backendCallCount)) + } +} + +func TestProxy_L1Cache_RFCMode_MaxAge(t *testing.T) { + var backendCallCount int64 + backend := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + atomic.AddInt64(&backendCallCount, 1) + w.Header().Set("Cache-Control", "public, max-age=300") + w.Header().Set("Content-Type", "application/json") + fmt.Fprint(w, `{"data":"rfc"}`) + })) + defer backend.Close() + + spec := NewCacheFilter(Options{MaxBytes: 1 << 20}) + t.Cleanup(func() { spec.(*cacheSpec).Close() }) + + fr := make(filters.Registry) + fr.Register(spec) + + // RFC mode: zero args + route := &eskip.Route{ + Id: "rfc-route", + PathRegexps: []string{".*"}, + Filters: []*eskip.Filter{{Name: spec.Name(), Args: nil}}, + Backend: backend.URL, + } + + proxy := proxytest.New(fr, route) + defer proxy.Close() + + client := proxy.Client() + + rsp1, err := client.Get(proxy.URL + "/rfc") + if err != nil { + t.Fatalf("first request: %v", err) + } + rsp1.Body.Close() + + rsp2, err := client.Get(proxy.URL + "/rfc") + if err != nil { + t.Fatalf("second request: %v", err) + } + rsp2.Body.Close() + if rsp2.Header.Get("X-Cache-Status") != "HIT" { + t.Fatalf("expected HIT in RFC mode, got %q", rsp2.Header.Get("X-Cache-Status")) + } + if atomic.LoadInt64(&backendCallCount) != 1 { + t.Fatalf("expected 1 backend call, got %d", atomic.LoadInt64(&backendCallCount)) + } +} + +func TestProxy_L1Cache_UnsafeMethod_Invalidates(t *testing.T) { + var backendCallCount int64 + backend := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + atomic.AddInt64(&backendCallCount, 1) + w.Header().Set("Cache-Control", "public, max-age=300") + w.Header().Set("Content-Type", "application/json") + fmt.Fprint(w, `{"data":"item"}`) + })) + defer backend.Close() + + spec := NewCacheFilter(Options{MaxBytes: 1 << 20}) + t.Cleanup(func() { spec.(*cacheSpec).Close() }) + fr := make(filters.Registry) + fr.Register(spec) + + route := &eskip.Route{ + Id: "invalidate-route", + PathRegexps: []string{".*"}, + Filters: []*eskip.Filter{{Name: spec.Name(), Args: []interface{}{"5m", "15s", "30s"}}}, + Backend: backend.URL, + } + + proxy := proxytest.New(fr, route) + defer proxy.Close() + + client := proxy.Client() + + // GET → MISS, populate cache + rsp1, _ := client.Get(proxy.URL + "/item/42") + rsp1.Body.Close() + if rsp1.Header.Get("X-Cache-Status") != "MISS" { + t.Fatalf("expected MISS on first GET") + } + + // GET → HIT + rsp2, _ := client.Get(proxy.URL + "/item/42") + rsp2.Body.Close() + if rsp2.Header.Get("X-Cache-Status") != "HIT" { + t.Fatalf("expected HIT on second GET") + } + + // POST → invalidates cache + req, _ := http.NewRequest("POST", proxy.URL+"/item/42", strings.NewReader("{}")) + req.Header.Set("Content-Type", "application/json") + rsp3, _ := client.Do(req) + rsp3.Body.Close() + + // GET → MISS again (cache invalidated) + rsp4, _ := client.Get(proxy.URL + "/item/42") + rsp4.Body.Close() + if rsp4.Header.Get("X-Cache-Status") != "MISS" { + t.Fatalf("expected MISS after POST invalidation, got %q", rsp4.Header.Get("X-Cache-Status")) + } +} + +func TestProxy_L1Cache_NoCache_RequestBypasses(t *testing.T) { + var backendCallCount int64 + backend := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + atomic.AddInt64(&backendCallCount, 1) + w.Header().Set("Cache-Control", "public, max-age=300") + fmt.Fprint(w, `{"data":"v1"}`) + })) + defer backend.Close() + + spec := NewCacheFilter(Options{MaxBytes: 1 << 20}) + t.Cleanup(func() { spec.(*cacheSpec).Close() }) + fr := make(filters.Registry) + fr.Register(spec) + + route := &eskip.Route{ + Id: "nocache-route", + PathRegexps: []string{".*"}, + Filters: []*eskip.Filter{{Name: spec.Name(), Args: []interface{}{"5m", "15s", "30s"}}}, + Backend: backend.URL, + } + + proxy := proxytest.New(fr, route) + defer proxy.Close() + + client := proxy.Client() + + for i := 0; i < 3; i++ { + req, _ := http.NewRequest("GET", proxy.URL+"/nc", nil) + req.Header.Set("Cache-Control", "no-cache") + rsp, _ := client.Do(req) + rsp.Body.Close() + } + + if atomic.LoadInt64(&backendCallCount) < 3 { + t.Fatalf("no-cache must bypass cache: expected >=3 backend calls, got %d", atomic.LoadInt64(&backendCallCount)) + } +} + +func TestProxy_L1Cache_VaryStar_NeverCached(t *testing.T) { + var backendCallCount int64 + backend := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + atomic.AddInt64(&backendCallCount, 1) + w.Header().Set("Cache-Control", "public, max-age=300") + w.Header().Set("Vary", "*") + fmt.Fprint(w, `{"data":"vary-star"}`) + })) + defer backend.Close() + + spec := NewCacheFilter(Options{MaxBytes: 1 << 20}) + t.Cleanup(func() { spec.(*cacheSpec).Close() }) + fr := make(filters.Registry) + fr.Register(spec) + + route := &eskip.Route{ + Id: "vary-star-route", + PathRegexps: []string{".*"}, + Filters: []*eskip.Filter{{Name: spec.Name(), Args: []interface{}{"5m", "15s", "30s"}}}, + Backend: backend.URL, + } + + proxy := proxytest.New(fr, route) + defer proxy.Close() + + client := proxy.Client() + + for i := 0; i < 3; i++ { + rsp, _ := client.Get(proxy.URL + "/vary") + rsp.Body.Close() + } + + if atomic.LoadInt64(&backendCallCount) < 3 { + t.Fatalf("Vary: * must never be cached: expected >=3 backend calls, got %d", atomic.LoadInt64(&backendCallCount)) + } +} + +func TestProxy_L2Cache_HitAfterL1Eviction(t *testing.T) { + var backendCallCount int64 + backend := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + atomic.AddInt64(&backendCallCount, 1) + w.Header().Set("Cache-Control", "public, max-age=300") + w.Header().Set("Content-Type", "application/json") + fmt.Fprint(w, `{"data":"l2-hit"}`) + })) + defer backend.Close() + + stub := newStubValkeyClient() + m := &testMetrics{} + lru := NewLRUStorage(1<<20, nil, m) + + spec := NewCacheFilter(Options{ + MaxBytes: 1 << 20, + L2Client: stub, + IsNoL2Err: func(err error) bool { + // valkey.Nil sentinel — mimic the real check without the import + return err != nil && err.Error() == "valkey: nil" + }, + L1TTL: 60 * time.Second, + Metrics: m, + }) + t.Cleanup(func() { spec.(*cacheSpec).Close() }) + + // Use the isNoErr function that matches the stub's "not found" behaviour + isNoErr := func(err error) bool { + return err != nil && strings.Contains(err.Error(), "stub: not found") + } + cs := spec.(*cacheSpec) + cs.storage = NewL2Storage(stub, lru, m, 60*time.Second, isNoErr) + cs.lruStorage = lru + for _, f := range cs.filters { + f.storage = cs.storage + f.lruStorage = lru + } + + fr := make(filters.Registry) + fr.Register(spec) + + route := &eskip.Route{ + Id: "l2-route", + PathRegexps: []string{".*"}, + Filters: []*eskip.Filter{{Name: spec.Name(), Args: []interface{}{"5m", "15s", "30s"}}}, + Backend: backend.URL, + } + + proxy := proxytest.New(fr, route) + defer proxy.Close() + + // Re-wire filters to use the test storage + for _, f := range cs.filters { + f.storage = cs.storage + f.lruStorage = lru + } + + client := proxy.Client() + + // First request: MISS — populates both L1 and L2 + rsp1, _ := client.Get(proxy.URL + "/l2item") + rsp1.Body.Close() + if rsp1.Header.Get("X-Cache-Status") != "MISS" { + t.Fatalf("expected MISS, got %q", rsp1.Header.Get("X-Cache-Status")) + } + // Second request: HIT from L1 + rsp2, _ := client.Get(proxy.URL + "/l2item") + rsp2.Body.Close() + if rsp2.Header.Get("X-Cache-Status") != "HIT" { + t.Fatalf("expected HIT from L1, got %q", rsp2.Header.Get("X-Cache-Status")) + } + + // Verify L2 has the entry + if len(stub.data) == 0 { + t.Fatal("expected L2 to contain an entry after MISS+HIT") + } +} + +func TestProxy_L2Cache_FallsBackToL1OnL2Failure(t *testing.T) { + var backendCallCount int64 + backend := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + atomic.AddInt64(&backendCallCount, 1) + w.Header().Set("Cache-Control", "public, max-age=300") + w.Header().Set("Content-Type", "application/json") + fmt.Fprint(w, `{"data":"l1-fallback"}`) + })) + defer backend.Close() + + stub := newBrokenStubValkeyClient() + m := &testMetrics{} + lru := NewLRUStorage(1<<20, nil, m) + + isNoErr := func(err error) bool { return false } // broken stub always returns real errors + l2store := NewL2Storage(stub, lru, m, 60*time.Second, isNoErr) + + spec := NewCacheFilter(Options{MaxBytes: 1 << 20, Metrics: m}) + t.Cleanup(func() { spec.(*cacheSpec).Close() }) + cs := spec.(*cacheSpec) + cs.storage = l2store + cs.lruStorage = lru + + fr := make(filters.Registry) + fr.Register(spec) + + route := &eskip.Route{ + Id: "l2-fallback-route", + PathRegexps: []string{".*"}, + Filters: []*eskip.Filter{{Name: spec.Name(), Args: []interface{}{"5m", "15s", "30s"}}}, + Backend: backend.URL, + } + + proxy := proxytest.New(fr, route) + defer proxy.Close() + + // Re-wire storage to all filters + for _, f := range cs.filters { + f.storage = l2store + f.lruStorage = lru + } + + client := proxy.Client() + + // First request: L2 broken → Set falls back to L1; served as MISS + rsp1, _ := client.Get(proxy.URL + "/fallback") + rsp1.Body.Close() + if rsp1.Header.Get("X-Cache-Status") != "MISS" { + t.Fatalf("expected MISS on first request with broken L2") + } + + // L2 Set fallback must have written to L1 — check l2_set_fallback counter + if m.counter("cache.l2_set_fallback") == 0 { + t.Error("expected l2_set_fallback to be incremented when L2 Set fails") + } +} + +func TestProxy_L2Cache_MissWhenBothMiss(t *testing.T) { + var backendCallCount int64 + backend := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + atomic.AddInt64(&backendCallCount, 1) + w.Header().Set("Cache-Control", "public, max-age=300") + fmt.Fprint(w, `{"data":"cold"}`) + })) + defer backend.Close() + + stub := newStubValkeyClient() // empty + m := &testMetrics{} + lru := NewLRUStorage(1<<20, nil, m) + isNoErr := func(err error) bool { + return err != nil && strings.Contains(err.Error(), "valkey: nil") + } + l2store := NewL2Storage(stub, lru, m, 0, isNoErr) + + spec := NewCacheFilter(Options{MaxBytes: 1 << 20, Metrics: m}) + t.Cleanup(func() { spec.(*cacheSpec).Close() }) + cs := spec.(*cacheSpec) + cs.storage = l2store + cs.lruStorage = lru + + fr := make(filters.Registry) + fr.Register(spec) + + route := &eskip.Route{ + Id: "both-miss-route", + PathRegexps: []string{".*"}, + Filters: []*eskip.Filter{{Name: spec.Name(), Args: []interface{}{"5m", "15s", "30s"}}}, + Backend: backend.URL, + } + + proxy := proxytest.New(fr, route) + defer proxy.Close() + + for _, f := range cs.filters { + f.storage = l2store + f.lruStorage = lru + } + + client := proxy.Client() + + rsp, _ := client.Get(proxy.URL + "/cold") + rsp.Body.Close() + if rsp.Header.Get("X-Cache-Status") != "MISS" { + t.Fatalf("expected MISS when both L1 and L2 empty, got %q", rsp.Header.Get("X-Cache-Status")) + } + if atomic.LoadInt64(&backendCallCount) == 0 { + t.Fatal("expected backend to be called on cold miss") + } +} + +// Ensure url import is used +var _ = url.Parse From 5e7f074a914be26e60d1dead1d598e35ed1466fe Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Sandor=20Sz=C3=BCcs?= Date: Tue, 1 Sep 2026 21:57:25 +0200 Subject: [PATCH 86/89] ai: adding how to use mockMetrics MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Signed-off-by: Sandor Szücs --- .agents/skills/tests/SKILL.md | 17 +++++++++++++++++ 1 file changed, 17 insertions(+) diff --git a/.agents/skills/tests/SKILL.md b/.agents/skills/tests/SKILL.md index 07bc097730..8fe199d727 100644 --- a/.agents/skills/tests/SKILL.md +++ b/.agents/skills/tests/SKILL.md @@ -80,3 +80,20 @@ func TestSetRequestHeader(t *testing.T) { } } ``` + +If you need metrics.Metrics implementation to inspect metrics, you can use: + +```go +mockMetrics := &metricstest.MockMetrics{} +// some code .. + +// inspect counters +mockMetrics.WithCounters(func(counters map[string]int64) { + if n := counters["a-counter"]; n == int64(5) { t.Fatalf("Failed to get expected counter value 5, got: %d", n) } +}) + +// inspect Gauges +mockMetrics.WithGauges(func(g map[string]float64) { +... +}) +``` From d071612c8f9a1b633d3f1db7a9dd1b44c5306fdf Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Sandor=20Sz=C3=BCcs?= Date: Tue, 1 Sep 2026 22:00:25 +0200 Subject: [PATCH 87/89] refactor: rewrite empty interface to any MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Signed-off-by: Sandor Szücs --- filters/cache/filter.go | 4 +-- filters/cache/filter_test.go | 56 ++++++++++++++++++------------------ 2 files changed, 30 insertions(+), 30 deletions(-) diff --git a/filters/cache/filter.go b/filters/cache/filter.go index 400c018525..a61932ee5d 100644 --- a/filters/cache/filter.go +++ b/filters/cache/filter.go @@ -150,7 +150,7 @@ func (s *cacheSpec) Close() error { return nil } -func (s *cacheSpec) CreateFilter(args []interface{}) (filters.Filter, error) { +func (s *cacheSpec) CreateFilter(args []any) (filters.Filter, error) { if len(args) != 0 && (len(args) < 3 || len(args) > 5) { return nil, fmt.Errorf("cache: expected 0 or 3-5 args (ttl, errorTTL, swrWindow[, staleIfError[, keyHeaders]]), got %d: %w", len(args), filters.ErrInvalidFilterParameters) } @@ -471,7 +471,7 @@ type coalesceResult struct { func (f *cacheFilter) coalesce(ctx filters.FilterContext, key string) { req := ctx.Request().Clone(context.Background()) - ch := f.coldSF.DoChan(key, func() (interface{}, error) { + ch := f.coldSF.DoChan(key, func() (any, error) { // Capture any existing stale-if-error eligible entry before fetching, so that a // subsequent 5xx response cannot overwrite it in storage before we read it. var sieStored *Entry diff --git a/filters/cache/filter_test.go b/filters/cache/filter_test.go index 02ce9774d3..c041d681ad 100644 --- a/filters/cache/filter_test.go +++ b/filters/cache/filter_test.go @@ -26,7 +26,7 @@ import ( func newTestFilter(t *testing.T, ttl, errorTTL, swrWindow time.Duration, extra ...time.Duration) *cacheFilter { t.Helper() spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: "localhost:9090", L1TTL: 60 * time.Second}) - args := []interface{}{ + args := []any{ ttl.String(), errorTTL.String(), swrWindow.String(), @@ -57,7 +57,7 @@ func newTestFilter(t *testing.T, ttl, errorTTL, swrWindow time.Duration, extra . func newTestFilterRFC(t *testing.T, _, _, _ time.Duration, _ ...time.Duration) *cacheFilter { t.Helper() spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: "localhost:9090", L1TTL: 60 * time.Second}) - f, err := spec.CreateFilter([]interface{}{}) + f, err := spec.CreateFilter([]any{}) if err != nil { t.Fatal(err) } @@ -137,7 +137,7 @@ func TestCacheFilter_MissAndHit(t *testing.T) { func TestCacheFilter_KeyIsolationByAuthToken(t *testing.T) { spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: "localhost:9090", L1TTL: 60 * time.Second}) - fi, err := spec.CreateFilter([]interface{}{"1m", "15s", "1m", "0s", "Authorization"}) + fi, err := spec.CreateFilter([]any{"1m", "15s", "1m", "0s", "Authorization"}) if err != nil { t.Fatal(err) } @@ -510,7 +510,7 @@ func TestCacheFilter_ColdMissCoalescing_FetchError_CoalesceErrorMetric(t *testin // coalesce_error must be incremented when the upstream fetch fails during coalescing. mockMetrics := &metricstest.MockMetrics{} spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: "localhost:9090", L1TTL: 60 * time.Second, Metrics: mockMetrics}) - fi, err := spec.CreateFilter([]interface{}{"1m", "15s", "1m"}) + fi, err := spec.CreateFilter([]any{"1m", "15s", "1m"}) if err != nil { t.Fatal(err) } @@ -717,7 +717,7 @@ func TestCacheFilter_Metrics(t *testing.T) { // Metrics passed via Options so f.metrics captures hit/miss/stale counters. mockMetrics := &metricstest.MockMetrics{} spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: "localhost:9090", L1TTL: 60 * time.Second, Metrics: mockMetrics}) - fi, err := spec.CreateFilter([]interface{}{time.Millisecond.String(), (15 * time.Second).String(), time.Hour.String()}) + fi, err := spec.CreateFilter([]any{time.Millisecond.String(), (15 * time.Second).String(), time.Hour.String()}) if err != nil { t.Fatal(err) } @@ -2682,7 +2682,7 @@ func TestCacheFilter_PureRFCMode_ZeroArgs_UsesUpstreamMaxAge(t *testing.T) { spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: ":9090", L1TTL: 60 * time.Second}) t.Cleanup(spec.(*cacheSpec).client.Close) t.Cleanup(func() { spec.(*cacheSpec).Close() }) - f, err := spec.CreateFilter([]interface{}{}) + f, err := spec.CreateFilter([]any{}) if err != nil { t.Fatalf("unexpected error: %v", err) } @@ -2717,7 +2717,7 @@ func TestCacheFilter_LRUBytesGaugeUpdatesWithoutEviction(t *testing.T) { spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: "localhost:9090", L1TTL: 60 * time.Second}) t.Cleanup(spec.(*cacheSpec).client.Close) t.Cleanup(func() { spec.(*cacheSpec).Close() }) - f, err := spec.CreateFilter([]interface{}{"5m", "15s", "5m"}) + f, err := spec.CreateFilter([]any{"5m", "15s", "5m"}) if err != nil { t.Fatal(err) } @@ -2762,7 +2762,7 @@ func TestCacheFilter_PureRFCMode_ZeroArgs_NoUpstreamDirective_NotCached(t *testi spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: ":9090", L1TTL: 60 * time.Second}) t.Cleanup(spec.(*cacheSpec).client.Close) t.Cleanup(func() { spec.(*cacheSpec).Close() }) - f, err := spec.CreateFilter([]interface{}{}) + f, err := spec.CreateFilter([]any{}) if err != nil { t.Fatalf("unexpected error: %v", err) } @@ -2881,13 +2881,13 @@ func TestCacheSpec_FilterRegistry(t *testing.T) { t.Cleanup(func() { spec.(*cacheSpec).Close() }) // Same args — should return same *cacheFilter pointer (registry hit) - f1, err := spec.CreateFilter([]interface{}{"5m", "15s", "30s"}) + f1, err := spec.CreateFilter([]any{"5m", "15s", "30s"}) if err != nil { t.Fatal(err) } cf1 := f1.(*cacheFilter) - f2, err := spec.CreateFilter([]interface{}{"5m", "15s", "30s"}) + f2, err := spec.CreateFilter([]any{"5m", "15s", "30s"}) if err != nil { t.Fatal(err) } @@ -2898,7 +2898,7 @@ func TestCacheSpec_FilterRegistry(t *testing.T) { } // Different args — should return different *cacheFilter pointer - f3, err := spec.CreateFilter([]interface{}{"10m", "15s", "30s"}) + f3, err := spec.CreateFilter([]any{"10m", "15s", "30s"}) if err != nil { t.Fatal(err) } @@ -2909,13 +2909,13 @@ func TestCacheSpec_FilterRegistry(t *testing.T) { } // Different keyHeaders order should normalize to same instance - f4, err := spec.CreateFilter([]interface{}{"5m", "15s", "30s", "60s", "X-Foo,X-Bar"}) + f4, err := spec.CreateFilter([]any{"5m", "15s", "30s", "60s", "X-Foo,X-Bar"}) if err != nil { t.Fatal(err) } cf4 := f4.(*cacheFilter) - f5, err := spec.CreateFilter([]interface{}{"5m", "15s", "30s", "60s", "X-Bar,X-Foo"}) + f5, err := spec.CreateFilter([]any{"5m", "15s", "30s", "60s", "X-Bar,X-Foo"}) if err != nil { t.Fatal(err) } @@ -2937,7 +2937,7 @@ func TestCacheSpec_FilterRegistry_InFlightJobsSurviveRebuild(t *testing.T) { t.Cleanup(func() { spec.(*cacheSpec).Close() }) // Create initial filter with blocking fetch stub (same pattern as TestCacheFilter_RevalDropped_WhenQueueFull). - f1, err := spec.CreateFilter([]interface{}{"5m", "15s", "30s"}) + f1, err := spec.CreateFilter([]any{"5m", "15s", "30s"}) if err != nil { t.Fatal(err) } @@ -2987,7 +2987,7 @@ func TestCacheSpec_FilterRegistry_InFlightJobsSurviveRebuild(t *testing.T) { // Simulate a route rebuild: call CreateFilter again with identical args. // The registry should return the same cf instance. - f2, err := spec.CreateFilter([]interface{}{"5m", "15s", "30s"}) + f2, err := spec.CreateFilter([]any{"5m", "15s", "30s"}) if err != nil { t.Fatal(err) } @@ -3232,7 +3232,7 @@ func TestCacheFilter_RFC_Private_NotCached_Coalesce(t *testing.T) { func TestCacheFilter_CoalesceSetFailure_Served(t *testing.T) { mockMetrics := &metricstest.MockMetrics{} spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: "localhost:9090", Metrics: mockMetrics}) - fi, err := spec.CreateFilter([]interface{}{"1m", "15s", "1m"}) + fi, err := spec.CreateFilter([]any{"1m", "15s", "1m"}) if err != nil { t.Fatal(err) } @@ -3308,7 +3308,7 @@ func TestCacheFilter_HEAD_Freshen_BodyHeaderSkipped(t *testing.T) { func TestCacheFilter_HEAD_Freshen_SetError(t *testing.T) { mockMetrics := &metricstest.MockMetrics{} spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: "localhost:9090", Metrics: mockMetrics}) - fi, err := spec.CreateFilter([]interface{}{"1m", "15s", "1m"}) + fi, err := spec.CreateFilter([]any{"1m", "15s", "1m"}) if err != nil { t.Fatal(err) } @@ -3378,7 +3378,7 @@ func TestCacheFilter_Response_ServedFromCache_IsNoop(t *testing.T) { func TestCacheFilter_UnsafeMethod_DeleteError_ContinuesSilently(t *testing.T) { mockMetrics := &metricstest.MockMetrics{} spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: "localhost:9090", Metrics: mockMetrics}) - fi, err := spec.CreateFilter([]interface{}{"1m", "15s", "1m"}) + fi, err := spec.CreateFilter([]any{"1m", "15s", "1m"}) if err != nil { t.Fatal(err) } @@ -3447,7 +3447,7 @@ func TestCacheFilter_Response_BodyReadError_NotCached(t *testing.T) { func TestCacheFilter_Response_VarySentinelSetError(t *testing.T) { mockMetrics := &metricstest.MockMetrics{} spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: "localhost:9090", Metrics: mockMetrics}) - fi, err := spec.CreateFilter([]interface{}{"1m", "15s", "1m"}) + fi, err := spec.CreateFilter([]any{"1m", "15s", "1m"}) if err != nil { t.Fatal(err) } @@ -3476,7 +3476,7 @@ func TestCacheFilter_Response_VarySentinelSetError(t *testing.T) { func TestCacheFilter_Response_StorageSetError(t *testing.T) { mockMetrics := &metricstest.MockMetrics{} spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: "localhost:9090", Metrics: mockMetrics}) - fi, err := spec.CreateFilter([]interface{}{"1m", "15s", "1m"}) + fi, err := spec.CreateFilter([]any{"1m", "15s", "1m"}) if err != nil { t.Fatal(err) } @@ -3638,7 +3638,7 @@ func TestCacheFilter_Revalidate_SetError_MetricIncremented(t *testing.T) { synctest.Test(t, func(t *testing.T) { mockMetrics := &metricstest.MockMetrics{} spec := NewCacheFilter(Options{MaxBytes: 1 << 20, ListenAddr: "localhost:9090", Metrics: mockMetrics}) - fi, err := spec.CreateFilter([]interface{}{time.Millisecond.String(), "15s", time.Hour.String()}) + fi, err := spec.CreateFilter([]any{time.Millisecond.String(), "15s", time.Hour.String()}) if err != nil { t.Fatal(err) } @@ -3857,7 +3857,7 @@ func TestCacheKeyForURL_RelativeURL_UsesBaseHost(t *testing.T) { // the cache filter. backendURL is the httptest backend. func newProxyCacheRoute(t *testing.T, backendURL string, spec filters.Spec, args ...string) *eskip.Route { t.Helper() - var filterArgs []interface{} + var filterArgs []any for _, a := range args { filterArgs = append(filterArgs, a) } @@ -3989,7 +3989,7 @@ func TestProxy_L1Cache_UnsafeMethod_Invalidates(t *testing.T) { route := &eskip.Route{ Id: "invalidate-route", PathRegexps: []string{".*"}, - Filters: []*eskip.Filter{{Name: spec.Name(), Args: []interface{}{"5m", "15s", "30s"}}}, + Filters: []*eskip.Filter{{Name: spec.Name(), Args: []any{"5m", "15s", "30s"}}}, Backend: backend.URL, } @@ -4043,7 +4043,7 @@ func TestProxy_L1Cache_NoCache_RequestBypasses(t *testing.T) { route := &eskip.Route{ Id: "nocache-route", PathRegexps: []string{".*"}, - Filters: []*eskip.Filter{{Name: spec.Name(), Args: []interface{}{"5m", "15s", "30s"}}}, + Filters: []*eskip.Filter{{Name: spec.Name(), Args: []any{"5m", "15s", "30s"}}}, Backend: backend.URL, } @@ -4082,7 +4082,7 @@ func TestProxy_L1Cache_VaryStar_NeverCached(t *testing.T) { route := &eskip.Route{ Id: "vary-star-route", PathRegexps: []string{".*"}, - Filters: []*eskip.Filter{{Name: spec.Name(), Args: []interface{}{"5m", "15s", "30s"}}}, + Filters: []*eskip.Filter{{Name: spec.Name(), Args: []any{"5m", "15s", "30s"}}}, Backend: backend.URL, } @@ -4145,7 +4145,7 @@ func TestProxy_L2Cache_HitAfterL1Eviction(t *testing.T) { route := &eskip.Route{ Id: "l2-route", PathRegexps: []string{".*"}, - Filters: []*eskip.Filter{{Name: spec.Name(), Args: []interface{}{"5m", "15s", "30s"}}}, + Filters: []*eskip.Filter{{Name: spec.Name(), Args: []any{"5m", "15s", "30s"}}}, Backend: backend.URL, } @@ -4208,7 +4208,7 @@ func TestProxy_L2Cache_FallsBackToL1OnL2Failure(t *testing.T) { route := &eskip.Route{ Id: "l2-fallback-route", PathRegexps: []string{".*"}, - Filters: []*eskip.Filter{{Name: spec.Name(), Args: []interface{}{"5m", "15s", "30s"}}}, + Filters: []*eskip.Filter{{Name: spec.Name(), Args: []any{"5m", "15s", "30s"}}}, Backend: backend.URL, } @@ -4265,7 +4265,7 @@ func TestProxy_L2Cache_MissWhenBothMiss(t *testing.T) { route := &eskip.Route{ Id: "both-miss-route", PathRegexps: []string{".*"}, - Filters: []*eskip.Filter{{Name: spec.Name(), Args: []interface{}{"5m", "15s", "30s"}}}, + Filters: []*eskip.Filter{{Name: spec.Name(), Args: []any{"5m", "15s", "30s"}}}, Backend: backend.URL, } From 9f7f6cebfab070cbc970654a77818174d9e9a177 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Sandor=20Sz=C3=BCcs?= Date: Tue, 1 Sep 2026 22:09:24 +0200 Subject: [PATCH 88/89] add l2_storage coverage MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Signed-off-by: Sandor Szücs --- filters/cache/l2_storage_valkey_test.go | 123 +++++++++++++++++++++++- 1 file changed, 119 insertions(+), 4 deletions(-) diff --git a/filters/cache/l2_storage_valkey_test.go b/filters/cache/l2_storage_valkey_test.go index 8e2f0f8a87..900a53c058 100644 --- a/filters/cache/l2_storage_valkey_test.go +++ b/filters/cache/l2_storage_valkey_test.go @@ -2,6 +2,7 @@ package cache import ( "context" + "encoding/json" "errors" "net/http" "sync" @@ -17,9 +18,10 @@ import ( // stubValkeyClient is an in-memory valkeyClient stub for unit tests that // should not depend on a running Valkey instance or Docker. type stubValkeyClient struct { - mu sync.Mutex - data map[string]string - broken bool // if true, all operations return an error + mu sync.Mutex + data map[string]string + broken bool // if true, all operations return an error + lastExpire time.Duration // records the expire arg of the last SetWithExpire call } func newStubValkeyClient() *stubValkeyClient { @@ -43,13 +45,14 @@ func (s *stubValkeyClient) Get(_ context.Context, key string) (string, error) { return v, nil } -func (s *stubValkeyClient) SetWithExpire(_ context.Context, key, value string, _ time.Duration) error { +func (s *stubValkeyClient) SetWithExpire(_ context.Context, key, value string, expire time.Duration) error { s.mu.Lock() defer s.mu.Unlock() if s.broken { return errors.New("stub: broken") } s.data[key] = value + s.lastExpire = expire return nil } @@ -455,3 +458,115 @@ func TestValkeyStorage_DeleteCleansL1EvenOnValkeyError(t *testing.T) { t.Error("expected L1 to be empty after Delete, but entry still present") } } + +func TestL2Storage_NewL2Storage_NegativeL1TTL_ClampsToDefault(t *testing.T) { + stub := newStubValkeyClient() + lru := NewLRUStorage(64<<20, nil, &testMetrics{}) + s := NewL2Storage(stub, lru, &testMetrics{}, -time.Second, valkey.IsValkeyNil) + if s.l1TTL != defaultMinTTL { + t.Errorf("expected l1TTL clamped to %v, got %v", defaultMinTTL, s.l1TTL) + } +} + +func TestL2Storage_Get_CorruptJSON_ReturnsError(t *testing.T) { + stub := newStubValkeyClient() + lru := NewLRUStorage(64<<20, nil, &testMetrics{}) + s := NewL2Storage(stub, lru, &testMetrics{}, 0, valkey.IsValkeyNil) + + // Write corrupt JSON directly into the stub, bypassing Set. + stub.mu.Lock() + stub.data["bad-key"] = "not json {" + stub.mu.Unlock() + + _, err := s.Get(context.Background(), "bad-key") + if err == nil { + t.Fatal("expected error for corrupt JSON in L2, got nil") + } +} + +func TestL2Storage_Get_L2Hit_WarmsL1(t *testing.T) { + stub := newStubValkeyClient() + m := &testMetrics{} + lru := NewLRUStorage(64<<20, nil, m) + s := NewL2Storage(stub, lru, m, 60*time.Second, valkey.IsValkeyNil) + + ctx := context.Background() + key := "warm-key" + entry := &Entry{StatusCode: 200, Payload: []byte("v"), TTL: 2 * time.Minute, CreatedAt: time.Now()} + + // Write directly to L2 stub (bypass Set so L1 stays cold). + data, _ := json.Marshal(entry) + stub.mu.Lock() + stub.data[key] = string(data) + stub.mu.Unlock() + + // First Get: L2 hit, must warm L1. + got, err := s.Get(ctx, key) + if err != nil || got == nil { + t.Fatalf("expected L2 hit: err=%v, got=%v", err, got) + } + if m.counter("cache.l2_hit") != 1 { + t.Errorf("expected l2_hit=1, got %d", m.counter("cache.l2_hit")) + } + + // Break L2 — second Get must come from L1 (write-through warmed it). + stub.broken = true + got2, err := s.Get(ctx, key) + if err != nil || got2 == nil { + t.Fatalf("expected L1 hit after warming: err=%v, got=%v", err, got2) + } + if m.counter("cache.l1_hit") != 1 { + t.Errorf("expected l1_hit=1 after L1 warming, got %d", m.counter("cache.l1_hit")) + } +} + +func TestL2Storage_Get_L2Hit_ExpiredEntry_SkipsL1Warming(t *testing.T) { + stub := newStubValkeyClient() + m := &testMetrics{} + lru := NewLRUStorage(64<<20, nil, m) + s := NewL2Storage(stub, lru, m, 60*time.Second, valkey.IsValkeyNil) + + ctx := context.Background() + key := "expired-key" + // CreatedAt 10 min ago, TTL 1 min → remaining = -9 min → L1 warming must be skipped. + entry := &Entry{StatusCode: 200, Payload: []byte("stale"), TTL: time.Minute, CreatedAt: time.Now().Add(-10 * time.Minute)} + + data, _ := json.Marshal(entry) + stub.mu.Lock() + stub.data[key] = string(data) + stub.mu.Unlock() + + // L2 hit — but warming must be skipped because remaining <= 0. + got, err := s.Get(ctx, key) + if err != nil || got == nil { + t.Fatalf("expected L2 hit for expired entry: err=%v, got=%v", err, got) + } + + // Break L2 — if L1 was accidentally warmed, Get would still return the entry. + stub.broken = true + got2, _ := s.Get(ctx, key) + if got2 != nil { + t.Error("L1 must NOT be warmed when remaining TTL <= 0, but entry was found in L1") + } +} + +func TestL2Storage_Set_ZeroTTL_UsesDefaultMinTTL(t *testing.T) { + stub := newStubValkeyClient() + lru := NewLRUStorage(64<<20, nil, &testMetrics{}) + s := NewL2Storage(stub, lru, &testMetrics{}, 0, valkey.IsValkeyNil) + + ctx := context.Background() + // TTL=0, StaleIfError=0, StaleWhileRevalidate=0 → l2TTL=0 → must use defaultMinTTL. + entry := &Entry{StatusCode: 200, Payload: []byte("x"), TTL: 0, CreatedAt: time.Now()} + if err := s.Set(ctx, "zero-ttl-key", entry); err != nil { + t.Fatalf("Set: %v", err) + } + + stub.mu.Lock() + gotExpire := stub.lastExpire + stub.mu.Unlock() + + if gotExpire != defaultMinTTL { + t.Errorf("expected SetWithExpire called with %v, got %v", defaultMinTTL, gotExpire) + } +} From 840cf86a0dd64028c9c4448cdfa5336e985b811f Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Sandor=20Sz=C3=BCcs?= Date: Tue, 1 Sep 2026 22:27:11 +0200 Subject: [PATCH 89/89] doc: use PR description to add storage architecture doc MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Signed-off-by: Sandor Szücs --- docs/operation/operation.md | 56 ++++++++++++++++++++++++++----------- 1 file changed, 39 insertions(+), 17 deletions(-) diff --git a/docs/operation/operation.md b/docs/operation/operation.md index b3a79a1c7b..5dd24ab32a 100644 --- a/docs/operation/operation.md +++ b/docs/operation/operation.md @@ -1966,30 +1966,30 @@ r: SourceFromLast("9.0.0.0/24","2001:67c:20a0::/48") -> ...` By default entries are stored in an in-process LRU (L1) local to each Skipper process. When `--swarm-valkey-urls` is configured and `--enable-l2-cache` is set, Valkey serves as a shared backing store (L2) accessible by all Skipper instances via a client-side consistent -hash ring; every read checks L1 first and an L1 hit returns without contacting Valkey. +hash ring; every read checks L1 first and an L1 hit returns without contacting L2 cache. Without `--enable-l2-cache`, `--swarm-valkey-urls` wires Valkey into ratelimit only and the `cache()` filter uses L1 exclusively. -On every successful Valkey write the entry is also written to L1 -(write-through) with a TTL of `min(--cache-l1-ttl, entry.TTL)`. On a Valkey +On every successful L2 write the entry is also written to L1 +(write-through) with a TTL of `min(--cache-l1-ttl, entry.TTL)`. On a L2 read hit, L1 is warmed with `min(--cache-l1-ttl, remaining freshness)` — remaining freshness (`entry.TTL - age`) is used rather than the original TTL to -prevent L1 from serving the entry beyond Valkey's actual expiry. The default is +prevent L1 from serving the entry beyond L2's actual expiry. The default is 60 seconds, bounding how long Skipper serves a locally-cached entry before -re-consulting Valkey. Set `--cache-l1-ttl=0` to disable L1 warming and -restore write-around behaviour. Note: on Valkey write errors (`l2_set_fallback`), L1 -is always used as a fallback regardless of `--cache-l1-ttl`; Valkey read errors +re-consulting L2. Set `--cache-l1-ttl=0` to disable L1 warming and +restore write-around behaviour. Note: on L2 write errors (`l2_set_fallback`), L1 +is always used as a fallback regardless of `--cache-l1-ttl`; L2 read errors (`l2_get_error`) are treated as cache misses. When an upstream responds successfully to an unsafe method (`POST`, `PUT`, `DELETE`, `PATCH`), -the filter removes the cached entry for that URL from both Valkey and the local L1. +the filter removes the cached entry for that URL from both L2 and the local L1. Only the local process's L1 is cleared — other Skipper processes in the fleet retain their own L1 copies until each entry's warmed TTL (bounded by `--cache-l1-ttl`) expires naturally. Set `--cache-l1-ttl` accordingly to bound the stale window after an invalidation. There is no out-of-band operator invalidation API. To clear the cache outside the normal -unsafe-method path, options are: wait for TTL expiry, restart the Skipper process (clears L1 in-memory cache only; Valkey data persists), or -delete the key directly in Valkey (L2 only; does not clear other pods' L1). +unsafe-method path, options are: wait for TTL expiry, restart the Skipper process (clears L1 in-memory cache only; L2 data persists), or +delete the key directly in L2. Concurrent cold-miss requests for the same route and key within one Skipper process are coalesced into a single upstream fetch (thundering-herd protection). Requests arriving via @@ -2009,6 +2009,28 @@ falling back to 2 GB if the limit is unreadable. Override with per-user routes). The L1 storage budget is divided evenly across 256 internal shards; a single entry larger than one shard's budget is dropped with a warning log and increments `cache.lru_oversized`. +### Storage architecture + +- L1 implementation is in-memory (25% of memory set by cgroup or 2GB fix size) +- L2 implementation is for example skpnet.ValkeyRingClient, reusing Valkey ring shards if available. + +```mermaid +flowchart TD + A[Incoming Request] --> B["Skipper cache() filter"] + B --> C{L1 LRUStorage} + C -->|cache.l1_hit — fresh hit| S[Entry Served] + C -->|cache.l2_miss| D{L2Storage} + C -->|cache.stale hit — no l1_hit| D + D -->|cache.l2_hit — key found| S + D -->|cache.l2_miss — key absent| E[Origin Fetch\n CDN] + D -->|cache.l2_get_error — error / timeout| E + E -->|write back: Set warms L2 + L1
min TTL: cache-l1-ttl vs entry.TTL| S +``` + +**Write path:** successful L2 Set warms L1 with min(--cache-l1-ttl, entry.TTL) (default 60s). L2 Get hit warms L1 with min(--cache-l1-ttl, remaining freshness). L1 is also populated on L2 Set errors (fallback). Set --cache-l1-ttl=0 for write-around. + +**Read path:** L1 checked first. Fresh L1 hit → returns immediately (increments `cache.l1_hit`), no L2 call. Stale L1 hit (past TTL but within the stale retention window) → falls through to L2 without incrementing `cache.l1_hit`. L1 miss → L2. L2 hit → increments `cache.l2_hit`, warms L1 with `min(--cache-l1-ttl, remaining freshness)` when `--cache-l1-ttl > 0`, returns entry. L2 miss (nil) → cold miss (increments `cache.l2_miss`). L2 error → increments `cache.l2_get_error`, treated as a cold miss. + ### Metrics **Cache outcomes (always active):** @@ -2032,14 +2054,14 @@ falling back to 2 GB if the limit is unreadable. Override with - `cache.reval_error`: Counter, background revalidation fetch failures - `cache.reval_duration`: Histogram, end-to-end duration of each background revalidation job -**Valkey (when Valkey is configured):** +**L2 (for example if Valkey is configured):** -- `cache.l1_hit`: Counter, L1 hits that bypassed Valkey -- `cache.l2_miss`: Counter, Valkey misses that proceeded to an upstream fetch -- `cache.l2_get_error`: Counter, Valkey Get errors — request treated as a cache miss, fetched from origin -- `cache.l2_set_fallback`: Counter, Valkey Set errors — entry written to L1 only (not L2) -- `cache.l2_hit`: Counter, successful Valkey Get (entry returned from L2); L1 is warmed as a side-effect when `--cache-l1-ttl > 0` -- `cache.storage_error`: Counter, any storage operation (Set or Delete) failed — covers both L2 Valkey failures and L1 eviction-path errors; the request is still served correctly +- `cache.l1_hit`: Counter, L1 hits that bypassed L2 +- `cache.l2_miss`: Counter, L2 misses that proceeded to an upstream fetch +- `cache.l2_get_error`: Counter, L2 Get errors — request treated as a cache miss, fetched from origin +- `cache.l2_set_fallback`: Counter, L2 Set errors — entry written to L1 only (not L2) +- `cache.l2_hit`: Counter, successful L2 Get (entry returned from L2); L1 is warmed as a side-effect when `--cache-l1-ttl > 0` +- `cache.storage_error`: Counter, any storage operation (Set or Delete) failed — covers both L2 failures and L1 eviction-path errors; the request is still served correctly **OpenTracing span tags (set on every request when a span is active):**