From 5627db174d8a388d41abada33ca46a851b01651a Mon Sep 17 00:00:00 2001 From: Jonny Date: Thu, 16 Apr 2026 22:31:07 +0100 Subject: [PATCH 01/40] feat: WebSocket support and transport-agnostic event indexer MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Adds a WebSocket transport for both inbound clients and outbound upstreams, plus a transport-agnostic event indexer that fans head and log notifications from any subscribed ingress out to any number of subscribed clients. Inbound (client-facing): - Server-side WS listener served on the same HTTP port. Each client connection is a long-lived JSON-RPC session that supports both regular RPCs (eth_call, eth_blockNumber, …) and subscriptions (eth_subscribe newHeads / logs / eth_unsubscribe). Directives (UseUpstream, headers etc.) are honoured via query params and upgrade headers. - Subscriptions are dedup'd in the indexer, so N clients subscribing to the same logs filter result in a single upstream subscription. - Failed subscribes do not close the client connection; only the failing eth_subscribe call returns an error. Outbound (upstream-facing): - WsJsonRpcClient parallels HttpJsonRpcClient and is wired into the same upstream selection / failsafe / scoring / rate-limit / metrics pipeline. eth_subscribe is dispatched here; regular RPCs route to whichever client (HTTP or WS) the upstream advertises. - Outbound JSON-RPC ids on the WS wire are rewritten to a unique per-client atomic counter so concurrent requests sharing a caller- supplied id (small ints are common) do not cross-wire each other in the pending response map. The caller's original id is restored on the response before returning. - A NodeGroup-aware "share block-head state across siblings" hook is not included in this commit; can be revisited as a follow-up. Indexer: - Indexer fans events from registered ingresses out to subscribed egresses. Two ingress adapters ship: wsupstream (wraps an upstream WS client) and nullingress (no-op for tests). - newHead dedup is single-CAS on a packed (num, hash) atomic.Pointer so concurrent ingest of the same head doesn't double-deliver. - Reorg handling rolls subscriptions back to the canonical chain on reorg-style notifications. - A `stripSubscribeFromBlockZero` network-level flag drops the meaningless `fromBlock: "0x0"` field from `eth_subscribe("logs")` filters before forwarding, so live-stream subscribes succeed on pruned backends. Non-zero fromBlocks pass through. Documentation, schema generation, and TS bindings are updated accordingly. --- architecture/evm/block_ref.go | 95 +- architecture/evm/block_ref_test.go | 82 + architecture/evm/json_rpc_cache.go | 13 +- clients/registry.go | 14 +- clients/ws_json_rpc_client.go | 618 ++++++ common/config.go | 118 +- common/config_test.go | 62 + common/defaults.go | 31 +- common/errors.go | 64 + common/validation.go | 17 + .../config/database/evm-json-rpc-cache.mdx | 2 + docs/pages/config/projects/networks.mdx | 48 + docs/pages/config/projects/upstreams.mdx | 2 + docs/pages/operation/_meta.js | 3 + docs/pages/operation/production.mdx | 10 + docs/pages/operation/url.mdx | 10 + docs/pages/operation/websocket.mdx | 378 ++++ erpc/http_server.go | 86 +- erpc/http_server_logs_test.go | 12 +- erpc/http_server_test.go | 10 +- erpc/http_timeout.go | 9 + erpc/init.go | 2 +- erpc/networks.go | 157 +- erpc/networks_consensus_test.go | 1 + erpc/networks_registry.go | 16 + erpc/networks_test.go | 308 +++ erpc/subscription_manager.go | 598 ++++++ erpc/ws_server.go | 741 ++++++++ erpc/ws_server_test.go | 1687 +++++++++++++++++ go.mod | 1 + go.sum | 2 + indexer/adapters/nullingress/adapter.go | 119 ++ indexer/adapters/wsclient/adapter.go | 271 +++ indexer/adapters/wsupstream/adapter.go | 467 +++++ indexer/adapters/wsupstream/adapter_test.go | 90 + indexer/constants.go | 22 + indexer/dedup.go | 91 + indexer/dedup_test.go | 84 + indexer/egress.go | 27 + indexer/errors.go | 11 + indexer/event.go | 126 ++ indexer/filters.go | 64 + indexer/filters_test.go | 86 + indexer/indexer.go | 541 ++++++ indexer/indexer_test.go | 584 ++++++ indexer/ingress.go | 89 + indexer/integration_test.go | 150 ++ indexer/log_removed.go | 23 + indexer/reorg.go | 219 +++ indexer/reorg_test.go | 136 ++ upstream/registry.go | 25 + upstream/upstream.go | 14 +- 52 files changed, 8342 insertions(+), 94 deletions(-) create mode 100644 clients/ws_json_rpc_client.go create mode 100644 docs/pages/operation/websocket.mdx create mode 100644 erpc/subscription_manager.go create mode 100644 erpc/ws_server.go create mode 100644 erpc/ws_server_test.go create mode 100644 indexer/adapters/nullingress/adapter.go create mode 100644 indexer/adapters/wsclient/adapter.go create mode 100644 indexer/adapters/wsupstream/adapter.go create mode 100644 indexer/adapters/wsupstream/adapter_test.go create mode 100644 indexer/constants.go create mode 100644 indexer/dedup.go create mode 100644 indexer/dedup_test.go create mode 100644 indexer/egress.go create mode 100644 indexer/errors.go create mode 100644 indexer/event.go create mode 100644 indexer/filters.go create mode 100644 indexer/filters_test.go create mode 100644 indexer/indexer.go create mode 100644 indexer/indexer_test.go create mode 100644 indexer/ingress.go create mode 100644 indexer/integration_test.go create mode 100644 indexer/log_removed.go create mode 100644 indexer/reorg.go create mode 100644 indexer/reorg_test.go diff --git a/architecture/evm/block_ref.go b/architecture/evm/block_ref.go index 694ce26b2..973886e32 100644 --- a/architecture/evm/block_ref.go +++ b/architecture/evm/block_ref.go @@ -75,10 +75,10 @@ func ExtractBlockReferenceFromRequest(ctx context.Context, r *common.NormalizedR // In case of "*" since it means any block, we can still augment it from response ref, because during cache.Get() // we'll be using reverse index (i.e. ignoring ref), but after reorg invalidation is added a specific block ref is useful. // - // TODO An ideal version stores the data for all eth_getBlockByNumber(latest) and eth_getBlockByNumber(blockNumber), - // and eth_getBlockByNumber(blockHash) where blockNumber/blockHash are the actual values returned in the response. - // So that if user gets the latest block, then cache is populated for when they provide that specific block as well. - // When implementing that feature remember that CacheHash() must be calculated separately for each number/hash combo. + // For moving-tag requests ("latest", "finalized", "safe"), the + // cache layer calls ResolveCacheBlockRef instead, which resolves + // the tag to a concrete block number so each tip advance gets + // its own cache key. blockRef = br } if bn > 0 { @@ -114,6 +114,93 @@ func ExtractBlockReferenceFromRequest(ctx context.Context, r *common.NormalizedR return blockRef, blockNumber, nil } +// ResolveCacheBlockRef returns the block reference the cache layer should use +// when keying an eth_getBlockByNumber("latest") response (and other moving +// tags). Regular ExtractBlockReferenceFromRequest preserves the literal tag +// string ("latest") as blockRef so the cache hits on repeat tag queries — +// but that makes every request within the TTL window return the same pinned +// response regardless of chain progression (see the bug fixed alongside this +// helper: stale "latest" responses served from cache until TTL expiry, with +// enforceHighestBlock explicitly skipping cached responses). +// +// This helper substitutes the tag with a concrete block number so each tip +// advance is a distinct cache key: on WRITE we use the response's own block +// number (definitive answer for what the cached payload represents); on READ +// we consult the network's tip tracker (EvmHighestLatestBlockNumber, which +// aggregates max over upstream pollers and the cross-pod shared counter) to +// decide which block we'd be asking for *right now*. Within a single tip +// the key is stable and concurrent "latest" queries coalesce onto one cached +// entry; across tip advances the key changes and the next request forwards +// upstream. +// +// The function does NOT mutate the request's EvmBlockRef — the original +// "latest" tag is preserved on the request so downstream finality computation +// and other tag-aware logic keeps working. +// +// Fallback: if the tag can't be resolved to a concrete block number (no +// response, no network attached to the request, or the tracker hasn't seen +// a block yet), the original tag is returned and the cache key stays +// tag-literal — same as prior behaviour. That path should be rare in +// production since every normal HTTP request has a Network and an upstream +// response by the SET stage. +func ResolveCacheBlockRef(ctx context.Context, req *common.NormalizedRequest, resp *common.NormalizedResponse) (string, int64, error) { + blockRef, blockNumber, err := ExtractBlockReferenceFromRequest(ctx, req) + if err != nil { + return blockRef, blockNumber, err + } + + // Only rewrite moving tip-bound tags. Numeric refs, block-hash refs, "*", + // and slower-moving tags like "earliest" are already correct. + if blockRef != "latest" && blockRef != "finalized" && blockRef != "safe" { + return blockRef, blockNumber, nil + } + + // WRITE path: prefer the response's own block number, which is the + // definitive answer for what payload we're about to cache. + if resp != nil { + if _, respBN, rerr := ExtractBlockReferenceFromResponse(ctx, resp); rerr == nil && respBN > 0 { + hex, herr := common.NormalizeHex(respBN) + if herr == nil { + return hex, respBN, nil + } + } + } + + // READ path (and WRITE fallback): consult the network's aggregated view + // of the tag's current value. Guarded against panics because this helper + // is purely an optimization — if the network state isn't reachable for + // any reason (partially-constructed Network in a test, nil upstream + // registry, transient initialization race), we fall back to the tag- + // literal blockRef and retain the previous behaviour rather than + // aborting a live cache operation. + net := req.Network() + if net == nil { + return blockRef, blockNumber, nil + } + var num int64 + func() { + defer func() { + if r := recover(); r != nil { + num = 0 + } + }() + switch blockRef { + case "latest": + num = net.EvmHighestLatestBlockNumber(ctx) + case "finalized", "safe": + num = net.EvmHighestFinalizedBlockNumber(ctx) + } + }() + if num > 0 { + hex, herr := common.NormalizeHex(num) + if herr == nil { + return hex, num, nil + } + } + + return blockRef, blockNumber, nil +} + func ExtractBlockReferenceFromResponse(ctx context.Context, r *common.NormalizedResponse) (string, int64, error) { ctx, span := common.StartDetailSpan(ctx, "Evm.ExtractBlockReferenceFromResponse") defer span.End() diff --git a/architecture/evm/block_ref_test.go b/architecture/evm/block_ref_test.go index 9aa45cc46..258e5d704 100644 --- a/architecture/evm/block_ref_test.go +++ b/architecture/evm/block_ref_test.go @@ -449,3 +449,85 @@ func TestExtractBlockReference(t *testing.T) { }) } } + +// TestResolveCacheBlockRef covers the cache-specific helper that rewrites +// moving-tag blockRefs ("latest"/"finalized"/"safe") to a concrete block +// number so each tip advance produces a distinct cache key. The previous +// behaviour pinned all "latest" responses under the literal "latest" ref, +// causing stale cache hits for up to TTL after a tip advance. +func TestResolveCacheBlockRef(t *testing.T) { + ctx := context.Background() + + t.Run("numeric ref passes through unchanged (no rewrite for non-tag)", func(t *testing.T) { + rpcReq := &common.JsonRpcRequest{ + Method: "eth_getBlockByNumber", + Params: []interface{}{"0x1234", false}, + } + nrq := common.NewNormalizedRequestFromJsonRpcRequest(rpcReq) + + ref, num, err := ResolveCacheBlockRef(ctx, nrq, nil) + assert.NoError(t, err) + // ExtractBlockReferenceFromRequest normalizes a numeric request ref + // to its decimal string form; the helper forwards whatever that + // returns for non-tag refs. + assert.Equal(t, "4660", ref) + assert.Equal(t, int64(0x1234), num) + }) + + t.Run("latest tag + response with block number rewrites ref to response hex", func(t *testing.T) { + rpcReq := &common.JsonRpcRequest{ + Method: "eth_getBlockByNumber", + Params: []interface{}{"latest", false}, + } + nrq := common.NewNormalizedRequestFromJsonRpcRequest(rpcReq) + rpcResp := common.MustNewJsonRpcResponseFromBytes(nil, []byte(`{"number":"0xabcdef","hash":"0x1","parentHash":"0x0"}`), nil) + nrs := common.NewNormalizedResponse().WithJsonRpcResponse(rpcResp).WithRequest(nrq) + nrq.SetLastValidResponse(ctx, nrs) + + ref, num, err := ResolveCacheBlockRef(ctx, nrq, nrs) + assert.NoError(t, err) + assert.Equal(t, "0xabcdef", ref, "write path must key by response's actual block number, not 'latest'") + assert.Equal(t, int64(0xabcdef), num) + }) + + t.Run("latest tag with no network and no response falls back to tag literal", func(t *testing.T) { + rpcReq := &common.JsonRpcRequest{ + Method: "eth_getBlockByNumber", + Params: []interface{}{"latest", false}, + } + nrq := common.NewNormalizedRequestFromJsonRpcRequest(rpcReq) + + ref, num, err := ResolveCacheBlockRef(ctx, nrq, nil) + assert.NoError(t, err) + assert.Equal(t, "latest", ref, "with no response and no network, helper must fall back to tag so caller can decide to skip caching") + assert.Equal(t, int64(0), num) + }) + + t.Run("finalized tag rewrite on write path uses response block number", func(t *testing.T) { + rpcReq := &common.JsonRpcRequest{ + Method: "eth_getBlockByNumber", + Params: []interface{}{"finalized", false}, + } + nrq := common.NewNormalizedRequestFromJsonRpcRequest(rpcReq) + rpcResp := common.MustNewJsonRpcResponseFromBytes(nil, []byte(`{"number":"0x100","hash":"0x1","parentHash":"0x0"}`), nil) + nrs := common.NewNormalizedResponse().WithJsonRpcResponse(rpcResp).WithRequest(nrq) + nrq.SetLastValidResponse(ctx, nrs) + + ref, num, err := ResolveCacheBlockRef(ctx, nrq, nrs) + assert.NoError(t, err) + assert.Equal(t, "0x100", ref) + assert.Equal(t, int64(0x100), num) + }) + + t.Run("earliest tag not rewritten (not tip-bound, existing semantics preserved)", func(t *testing.T) { + rpcReq := &common.JsonRpcRequest{ + Method: "eth_getBlockByNumber", + Params: []interface{}{"earliest", false}, + } + nrq := common.NewNormalizedRequestFromJsonRpcRequest(rpcReq) + + ref, _, err := ResolveCacheBlockRef(ctx, nrq, nil) + assert.NoError(t, err) + assert.Equal(t, "earliest", ref, "earliest does not move with the tip; must not be rewritten") + }) +} diff --git a/architecture/evm/json_rpc_cache.go b/architecture/evm/json_rpc_cache.go index 7c672a60f..0a8532c54 100644 --- a/architecture/evm/json_rpc_cache.go +++ b/architecture/evm/json_rpc_cache.go @@ -603,7 +603,11 @@ func (c *EvmJsonRpcCache) Set(ctx context.Context, req *common.NormalizedRequest attribute.String("network.id", ntwId), ) - blockRef, blockNumber, err := ExtractBlockReferenceFromRequest(ctx, req) + // For the SET path we resolve moving tags ("latest", "finalized", "safe") + // to the response's concrete block number so each tip advance gets its own + // cache key — see ResolveCacheBlockRef. The request's EvmBlockRef is NOT + // mutated; the original tag is still visible to downstream callers. + blockRef, blockNumber, err := ResolveCacheBlockRef(ctx, req, resp) if err != nil { common.SetTraceSpanError(span, err) return err @@ -969,7 +973,12 @@ func (c *EvmJsonRpcCache) doGet(ctx context.Context, connector data.Connector, r rpcReq.RLockWithTrace(ctx) defer rpcReq.RUnlock() - blockRef, _, err := ExtractBlockReferenceFromRequest(ctx, req) + // For the GET path we resolve moving tags ("latest", "finalized", "safe") + // to the network's currently-known tip block number so the lookup key + // tracks chain progression — see ResolveCacheBlockRef. A burst of + // concurrent "latest" queries landing on the same tip will still coalesce + // onto one cache entry; across tip advances each block gets its own key. + blockRef, _, err := ResolveCacheBlockRef(ctx, req, nil) if err != nil { return nil, err } diff --git a/clients/registry.go b/clients/registry.go index 3b3bba6c8..b52999c02 100644 --- a/clients/registry.go +++ b/clients/registry.go @@ -15,6 +15,7 @@ type ClientType string const ( ClientTypeHttpJsonRpc ClientType = "HttpJsonRpc" ClientTypeGrpcBds ClientType = "GrpcBds" + ClientTypeWsJsonRpc ClientType = "WsJsonRpc" ) type ClientInterface interface { @@ -98,7 +99,18 @@ func (manager *ClientRegistry) CreateClient(appCtx context.Context, ups common.U clientErr = fmt.Errorf("failed to create HTTP client for upstream: %v", cfg.Id) } } else if parsedUrl.Scheme == "ws" || parsedUrl.Scheme == "wss" { - clientErr = fmt.Errorf("websocket client not implemented yet") + newClient, err = NewWsJsonRpcClient( + appCtx, + &lg, + manager.projectId, + ups, + parsedUrl, + cfg.JsonRpc, + manager.evmExtractor, + ) + if err != nil { + clientErr = fmt.Errorf("failed to create WebSocket client for upstream %v: %w", cfg.Id, err) + } } else if parsedUrl.Scheme == "grpc" || parsedUrl.Scheme == "grpc+bds" { newClient, err = NewGrpcBdsClient( appCtx, diff --git a/clients/ws_json_rpc_client.go b/clients/ws_json_rpc_client.go new file mode 100644 index 000000000..db4ff2905 --- /dev/null +++ b/clients/ws_json_rpc_client.go @@ -0,0 +1,618 @@ +package clients + +import ( + "context" + "crypto/tls" + "errors" + "fmt" + "io" + "net/http" + "net/url" + "strconv" + "strings" + "sync" + "sync/atomic" + "time" + + "encoding/json" + + "github.com/erpc/erpc/common" + "github.com/gorilla/websocket" + "github.com/rs/zerolog" + "go.opentelemetry.io/otel/attribute" + "go.opentelemetry.io/otel/trace" +) + +const ( + wsPingInterval = 30 * time.Second + wsWriteWait = 10 * time.Second + wsReconnectMin = 1 * time.Second + wsReconnectMax = 30 * time.Second + wsReconnectFactor = 2.0 +) + +// WsJsonRpcClient implements ClientInterface for WebSocket-based JSON-RPC upstream connections. +type WsJsonRpcClient struct { + Url *url.URL + headers http.Header + + projectId string + upstream common.Upstream + appCtx context.Context + logger *zerolog.Logger + + // Connection state + connMu sync.Mutex + conn *websocket.Conn + + // connWake is pulsed by reconnect() once a new connection is in c.conn, + // so readLoop can wake up without polling. Capacity 1 coalesces bursts. + connWake chan struct{} + + // Write synchronization (gorilla/websocket requires synchronized writes) + writeMu sync.Mutex + + // Pending request tracking: JSON-RPC ID -> response channel. + // Uses RWMutex because the hot path (handleMessage dispatching responses) + // only needs a read lock, while writes (register/deregister) are less frequent. + pendingMu sync.RWMutex + pending map[string]chan *wsPendingResult + + // Signalled when the first connection is established (or app shutdown). + // readLoop blocks on this before entering its main loop. + connReady chan struct{} + connOnce sync.Once + + // Subscription notification callbacks: upstreamSubID -> handler + subHandlersMu sync.RWMutex + subHandlers map[string]func(params []byte) + + // Disconnect/reconnect callbacks are keyed by caller-supplied IDs so + // subscribers can replace (on re-subscribe) and remove (on teardown) + // their hooks, preventing the callback slices from growing unbounded + // over long-lived connections with subscription churn. + onDisconnectMu sync.RWMutex + onDisconnectCbs map[string]func() + + onReconnectMu sync.RWMutex + onReconnectCbs map[string]func() + + // Error extractor for architecture-specific error normalization + errorExtractor common.JsonRpcErrorExtractor + + connected atomic.Bool + + // wireIDCounter generates unique JSON-RPC ids on the WS wire so that + // concurrent SendRequest calls with the same caller-supplied id do not + // collide on the pending response map. The original caller id is + // restored on the response before returning. Seeded at wireIDOffset to + // stay in the same numeric range internal subscribers use when they + // build outbound requests (see indexer/adapters/wsupstream), so callers + // inspecting on-wire ids can tell internal traffic apart from + // small-int client traffic. + wireIDCounter atomic.Uint64 +} + +// wireIDOffset keeps rewritten wire ids in the "internal" id range so they +// don't collide with the small incrementing ints typical of client traffic +// or upstream-side state-poller requests. +const wireIDOffset uint64 = 900_000_000 + +type wsPendingResult struct { + resp *common.NormalizedResponse + err error +} + +// wsMessage is a minimal struct for parsing incoming WS messages to determine if they are +// responses (have "id") or notifications (have "method"). +type wsMessage struct { + JSONRPC string `json:"jsonrpc"` + ID interface{} `json:"id,omitempty"` + Method string `json:"method,omitempty"` + Result json.RawMessage `json:"result,omitempty"` + Error *common.ErrJsonRpcExceptionExternal `json:"error,omitempty"` + Params json.RawMessage `json:"params,omitempty"` +} + +// wsNotificationParams is the structure of subscription notification params. +type wsNotificationParams struct { + Subscription string `json:"subscription"` + Result json.RawMessage `json:"result"` +} + +func NewWsJsonRpcClient( + appCtx context.Context, + logger *zerolog.Logger, + projectId string, + upstream common.Upstream, + parsedUrl *url.URL, + jsonRpcCfg *common.JsonRpcUpstreamConfig, + extractor common.JsonRpcErrorExtractor, +) (ClientInterface, error) { + headers := http.Header{} + if jsonRpcCfg != nil && jsonRpcCfg.Headers != nil { + for k, v := range jsonRpcCfg.Headers { + headers.Set(k, v) + } + } + + client := &WsJsonRpcClient{ + Url: parsedUrl, + headers: headers, + projectId: projectId, + upstream: upstream, + appCtx: appCtx, + logger: logger, + pending: make(map[string]chan *wsPendingResult), + connReady: make(chan struct{}), + connWake: make(chan struct{}, 1), + subHandlers: make(map[string]func(params []byte)), + onDisconnectCbs: make(map[string]func()), + onReconnectCbs: make(map[string]func()), + errorExtractor: extractor, + } + client.wireIDCounter.Store(wireIDOffset) + + if err := client.connect(); err != nil { + // Don't fail on initial connection — start reconnect loop in background. + // The upstream may not be available at startup but will be retried. + logger.Warn().Err(err).Str("url", parsedUrl.String()).Msg("initial websocket connection failed, will retry in background") + go client.reconnect() + } else { + client.connOnce.Do(func() { close(client.connReady) }) + } + + go client.readLoop() + go client.pingLoop() + go func() { + <-appCtx.Done() + client.shutdown() + }() + + return client, nil +} + +func (c *WsJsonRpcClient) GetType() ClientType { + return ClientTypeWsJsonRpc +} + +// IsConnected returns true if the upstream WebSocket connection is currently established. +func (c *WsJsonRpcClient) IsConnected() bool { + return c.connected.Load() +} + +func (c *WsJsonRpcClient) SendRequest(ctx context.Context, req *common.NormalizedRequest) (*common.NormalizedResponse, error) { + ctx, span := common.StartDetailSpan(ctx, "WsJsonRpcClient.SendRequest", + trace.WithAttributes( + attribute.String("upstream.id", c.upstream.Id()), + ), + ) + defer span.End() + + startedAt := time.Now() + + jrReq, err := req.JsonRpcRequest() + if err != nil { + return nil, common.NewErrUpstreamRequest( + err, + c.upstream, + req.NetworkId(), + "", + 0, 0, 0, 0, + ) + } + + // Use a unique outbound wire id so concurrent SendRequest calls with the + // same caller-supplied JSON-RPC id do not collide in c.pending. The + // original id is restored on the response below before returning. + wireID := c.wireIDCounter.Add(1) + idKey := strconv.FormatUint(wireID, 10) + + // Serialize the JSON-RPC request with the rewritten wire id + jrReq.RLock() + originalID := jrReq.ID + requestBody, err := common.SonicCfg.Marshal(map[string]interface{}{ + "jsonrpc": jrReq.JSONRPC, + "id": wireID, + "method": jrReq.Method, + "params": jrReq.Params, + }) + jrReq.RUnlock() + if err != nil { + common.SetTraceSpanError(span, err) + return nil, common.NewErrUpstreamRequest( + err, + c.upstream, + req.NetworkId(), + jrReq.Method, + 0, 0, 0, 0, + ) + } + + // Register a response channel + respCh := make(chan *wsPendingResult, 1) + c.pendingMu.Lock() + c.pending[idKey] = respCh + c.pendingMu.Unlock() + + defer func() { + c.pendingMu.Lock() + delete(c.pending, idKey) + c.pendingMu.Unlock() + }() + + // Write to the WebSocket connection + if err := c.writeMessage(websocket.TextMessage, requestBody); err != nil { + common.SetTraceSpanError(span, err) + return nil, common.NewErrEndpointTransportFailure(c.Url, err) + } + + c.logger.Debug(). + Str("host", c.Url.Host). + RawJSON("request", requestBody). + Msg("sent json rpc websocket request") + + // Wait for response + select { + case result := <-respCh: + if result.err != nil { + common.SetTraceSpanError(span, result.err) + return nil, result.err + } + // Restore the caller's original JSON-RPC id on the response, since + // the on-wire id was rewritten to our unique counter above. + if result.resp != nil { + if jrr, perr := result.resp.JsonRpcResponse(ctx); perr == nil && jrr != nil { + _ = jrr.SetID(originalID) + } + } + return result.resp, nil + case <-ctx.Done(): + err := ctx.Err() + if errors.Is(err, context.DeadlineExceeded) { + err = common.NewErrEndpointRequestTimeout(time.Since(startedAt), err) + } else if errors.Is(err, context.Canceled) { + err = common.NewErrEndpointRequestCanceled(err) + } + common.SetTraceSpanError(span, err) + return nil, err + case <-c.appCtx.Done(): + return nil, common.NewErrEndpointRequestCanceled(c.appCtx.Err()) + } +} + +// RegisterSubscriptionHandler registers a callback for a specific upstream subscription ID. +// When the upstream sends a notification for this subscription, the handler is called with the raw params bytes. +func (c *WsJsonRpcClient) RegisterSubscriptionHandler(upstreamSubID string, handler func(params []byte)) { + c.subHandlersMu.Lock() + c.subHandlers[upstreamSubID] = handler + c.subHandlersMu.Unlock() +} + +// UnregisterSubscriptionHandler removes the callback for a specific upstream subscription ID. +func (c *WsJsonRpcClient) UnregisterSubscriptionHandler(upstreamSubID string) { + c.subHandlersMu.Lock() + delete(c.subHandlers, upstreamSubID) + c.subHandlersMu.Unlock() +} + +// SetOnDisconnect registers (or replaces) the callback keyed by id that fires +// when the upstream WS connection drops. Use RemoveOnDisconnect(id) to +// deregister on subscription teardown so long-lived connections don't +// accumulate dead callbacks. +func (c *WsJsonRpcClient) SetOnDisconnect(id string, callback func()) { + c.onDisconnectMu.Lock() + c.onDisconnectCbs[id] = callback + c.onDisconnectMu.Unlock() +} + +// RemoveOnDisconnect deregisters a disconnect callback previously set with +// SetOnDisconnect. A no-op if id is not registered. +func (c *WsJsonRpcClient) RemoveOnDisconnect(id string) { + c.onDisconnectMu.Lock() + delete(c.onDisconnectCbs, id) + c.onDisconnectMu.Unlock() +} + +// SetOnReconnect registers (or replaces) the callback keyed by id that fires +// after a successful reconnect. +func (c *WsJsonRpcClient) SetOnReconnect(id string, callback func()) { + c.onReconnectMu.Lock() + c.onReconnectCbs[id] = callback + c.onReconnectMu.Unlock() +} + +// RemoveOnReconnect deregisters a reconnect callback previously set with +// SetOnReconnect. A no-op if id is not registered. +func (c *WsJsonRpcClient) RemoveOnReconnect(id string) { + c.onReconnectMu.Lock() + delete(c.onReconnectCbs, id) + c.onReconnectMu.Unlock() +} + +func (c *WsJsonRpcClient) connect() error { + c.connMu.Lock() + defer c.connMu.Unlock() + + dialer := websocket.Dialer{ + HandshakeTimeout: 10 * time.Second, + } + + if c.Url.Scheme == "wss" { + dialer.TLSClientConfig = &tls.Config{ + MinVersion: tls.VersionTLS12, + } + } + + conn, _, err := dialer.DialContext(c.appCtx, c.Url.String(), c.headers) + if err != nil { + return err + } + + c.conn = conn + c.connected.Store(true) + + c.logger.Info().Str("url", c.Url.String()).Msg("websocket connection established") + return nil +} + +func (c *WsJsonRpcClient) readLoop() { + // Wait until the first connection is established (or the app shuts down) + select { + case <-c.connReady: + case <-c.appCtx.Done(): + return + } + + for { + if c.appCtx.Err() != nil { + return + } + + c.connMu.Lock() + conn := c.conn + c.connMu.Unlock() + + if conn == nil { + // Connection is being re-established after a disconnect; block + // until reconnect() pulses connWake (or the app shuts down). + select { + case <-c.connWake: + case <-c.appCtx.Done(): + return + } + continue + } + + _, message, err := conn.ReadMessage() + if err != nil { + if c.appCtx.Err() != nil { + return + } + if websocket.IsCloseError(err, websocket.CloseNormalClosure, websocket.CloseGoingAway) { + c.logger.Info().Msg("websocket connection closed normally") + } else { + c.logger.Warn().Err(err).Msg("websocket read error, will reconnect") + } + c.connected.Store(false) + c.drainPending(common.NewErrEndpointTransportFailure(c.Url, fmt.Errorf("websocket connection lost: %w", err))) + c.fireCallbacks(&c.onDisconnectMu, c.onDisconnectCbs) + + c.reconnect() + continue + } + + c.handleMessage(message) + } +} + +// fireCallbacks snapshots the callback map under rlock and dispatches each +// in its own goroutine. Snapshotting lets callbacks register/deregister +// other callbacks without deadlocking on the map's RWMutex. +func (c *WsJsonRpcClient) fireCallbacks(mu *sync.RWMutex, cbs map[string]func()) { + mu.RLock() + snapshot := make([]func(), 0, len(cbs)) + for _, cb := range cbs { + snapshot = append(snapshot, cb) + } + mu.RUnlock() + for _, cb := range snapshot { + go cb() + } +} + +func (c *WsJsonRpcClient) handleMessage(message []byte) { + var msg wsMessage + if err := common.SonicCfg.Unmarshal(message, &msg); err != nil { + c.logger.Warn().Err(err).Str("raw", string(message)).Msg("failed to parse websocket message") + return + } + + // Subscription notification: has "method" field (typically "eth_subscription") + if msg.Method != "" && msg.ID == nil { + c.handleNotification(msg.Method, msg.Params) + return + } + + // Response to a pending request: has "id" field + if msg.ID != nil { + idKey := normalizeIDKey(msg.ID) + + c.pendingMu.RLock() + ch, ok := c.pending[idKey] + c.pendingMu.RUnlock() + + if !ok { + c.logger.Debug().Str("id", idKey).Msg("received response for unknown request ID") + return + } + + nr := common.NewNormalizedResponse().WithBody(io.NopCloser(strings.NewReader(string(message)))) + + if msg.Error != nil { + ch <- &wsPendingResult{resp: nr, err: msg.Error} + } else { + ch <- &wsPendingResult{resp: nr} + } + return + } + + c.logger.Debug().Str("raw", string(message)).Msg("received unhandled websocket message") +} + +func (c *WsJsonRpcClient) handleNotification(method string, params []byte) { + if method != "eth_subscription" { + c.logger.Debug().Str("method", method).Msg("received non-subscription notification") + return + } + + var notifParams wsNotificationParams + if err := common.SonicCfg.Unmarshal(params, ¬ifParams); err != nil { + c.logger.Warn().Err(err).Msg("failed to parse subscription notification params") + return + } + + c.subHandlersMu.RLock() + handler, ok := c.subHandlers[notifParams.Subscription] + c.subHandlersMu.RUnlock() + + if !ok { + c.logger.Debug().Str("subscriptionId", notifParams.Subscription).Msg("received notification for unknown subscription") + return + } + + handler(params) +} + +func (c *WsJsonRpcClient) reconnect() { + backoff := wsReconnectMin + for { + if c.appCtx.Err() != nil { + return + } + + c.logger.Info().Dur("backoff", backoff).Msg("attempting websocket reconnection") + + if err := c.connect(); err != nil { + c.logger.Warn().Err(err).Dur("backoff", backoff).Msg("websocket reconnection failed") + select { + case <-time.After(backoff): + case <-c.appCtx.Done(): + return + } + backoff = time.Duration(float64(backoff) * wsReconnectFactor) + if backoff > wsReconnectMax { + backoff = wsReconnectMax + } + continue + } + + c.logger.Info().Msg("websocket reconnected successfully") + + // Signal readLoop if this is the first successful connection. + c.connOnce.Do(func() { close(c.connReady) }) + + // Wake readLoop if it's parked waiting for c.conn to be non-nil. + // Buffered channel with cap 1 means we coalesce concurrent pulses. + select { + case c.connWake <- struct{}{}: + default: + } + + c.fireCallbacks(&c.onReconnectMu, c.onReconnectCbs) + + return + } +} + +func (c *WsJsonRpcClient) drainPending(err error) { + c.pendingMu.Lock() + pending := c.pending + c.pending = make(map[string]chan *wsPendingResult) + c.pendingMu.Unlock() + + for _, ch := range pending { + select { + case ch <- &wsPendingResult{err: err}: + default: + } + } +} + +func (c *WsJsonRpcClient) writeMessage(messageType int, data []byte) error { + c.writeMu.Lock() + defer c.writeMu.Unlock() + + c.connMu.Lock() + conn := c.conn + c.connMu.Unlock() + + if conn == nil { + return fmt.Errorf("websocket connection not established") + } + + if err := conn.SetWriteDeadline(time.Now().Add(wsWriteWait)); err != nil { + return err + } + return conn.WriteMessage(messageType, data) +} + +func (c *WsJsonRpcClient) pingLoop() { + ticker := time.NewTicker(wsPingInterval) + defer ticker.Stop() + + for { + select { + case <-ticker.C: + if !c.connected.Load() { + continue + } + if err := c.writeMessage(websocket.PingMessage, nil); err != nil { + c.logger.Debug().Err(err).Msg("websocket ping failed") + } + case <-c.appCtx.Done(): + return + } + } +} + +// normalizeIDKey converts a JSON-RPC ID to a stable string key. +// JSON unmarshalling turns integer IDs into float64, which can produce +// scientific notation with fmt.Sprintf (e.g., "1.51e+09" vs "1510000000"). +// This function normalizes to avoid mismatches. +func normalizeIDKey(id interface{}) string { + switch v := id.(type) { + case float64: + // Format without scientific notation + return fmt.Sprintf("%.0f", v) + case int: + return fmt.Sprintf("%d", v) + case int64: + return fmt.Sprintf("%d", v) + case string: + return v + default: + return fmt.Sprintf("%v", v) + } +} + +func (c *WsJsonRpcClient) shutdown() { + c.connected.Store(false) + + c.connMu.Lock() + conn := c.conn + c.conn = nil + c.connMu.Unlock() + + if conn != nil { + // Send close frame and close + _ = conn.WriteControl( + websocket.CloseMessage, + websocket.FormatCloseMessage(websocket.CloseNormalClosure, ""), + time.Now().Add(wsWriteWait), + ) + _ = conn.Close() + } + + c.drainPending(common.NewErrEndpointRequestCanceled(fmt.Errorf("websocket client shutting down"))) +} diff --git a/common/config.go b/common/config.go index 222f62bab..f392e488d 100644 --- a/common/config.go +++ b/common/config.go @@ -45,6 +45,7 @@ type Config struct { Projects []*ProjectConfig `yaml:"projects,omitempty" json:"projects"` RateLimiters *RateLimiterConfig `yaml:"rateLimiters,omitempty" json:"rateLimiters"` Metrics *MetricsConfig `yaml:"metrics,omitempty" json:"metrics"` + Indexer *IndexerConfig `yaml:"indexer,omitempty" json:"indexer"` ProxyPools []*ProxyPoolConfig `yaml:"proxyPools,omitempty" json:"proxyPools"` Tracing *TracingConfig `yaml:"tracing,omitempty" json:"tracing"` @@ -82,6 +83,21 @@ var LegacyTranslateFn func(*Config) ([]string, error) // emitted by LegacyTranslateFn. If nil, warnings are dropped silently. var LegacyTranslateLogger func(warning string) +// IndexerConfig tunes the transport-neutral event-stream indexer that +// powers `eth_subscribe` fan-out and reorg-aware log invalidation. Most +// deployments can leave this unset. +type IndexerConfig struct { + // CanonicalChainDepth is the per-network ring-buffer size for the + // canonical-chain tracker. It bounds how deep a reorg the indexer + // can fully resolve — reorgs beyond this window get only the + // immediate head evicted. 0 uses the internal default (256). + CanonicalChainDepth int `yaml:"canonicalChainDepth,omitempty" json:"canonicalChainDepth"` + // DedupWindowSize is the per-filter seen-set capacity for log / + // pending-tx fan-out across sibling upstreams. 0 uses the internal + // default (8192). + DedupWindowSize int `yaml:"dedupWindowSize,omitempty" json:"dedupWindowSize"` +} + // LoadConfig loads the configuration from the specified file. // It supports both YAML and TypeScript (.ts) files. func LoadConfig(fs afero.Fs, filename string, opts *DefaultOptions) (*Config, error) { @@ -132,32 +148,33 @@ func LoadConfig(fs afero.Fs, filename string, opts *DefaultOptions) (*Config, er } type ServerConfig struct { - ListenV4 *bool `yaml:"listenV4,omitempty" json:"listenV4"` - HttpHostV4 *string `yaml:"httpHostV4,omitempty" json:"httpHostV4"` - ListenV6 *bool `yaml:"listenV6,omitempty" json:"listenV6"` - HttpHostV6 *string `yaml:"httpHostV6,omitempty" json:"httpHostV6"` - HttpPort *int `yaml:"httpPort,omitempty" json:"httpPort"` // Deprecated: use HttpPortV4 - HttpPortV4 *int `yaml:"httpPortV4,omitempty" json:"httpPortV4"` - HttpPortV6 *int `yaml:"httpPortV6,omitempty" json:"httpPortV6"` - GrpcEnabled *bool `yaml:"grpcEnabled,omitempty" json:"grpcEnabled"` - GrpcHostV4 *string `yaml:"grpcHostV4,omitempty" json:"grpcHostV4"` - GrpcPortV4 *int `yaml:"grpcPortV4,omitempty" json:"grpcPortV4"` - GrpcHostV6 *string `yaml:"grpcHostV6,omitempty" json:"grpcHostV6"` - GrpcPortV6 *int `yaml:"grpcPortV6,omitempty" json:"grpcPortV6"` - GrpcMaxRecvMsgSize *int `yaml:"grpcMaxRecvMsgSize,omitempty" json:"grpcMaxRecvMsgSize"` - GrpcMaxSendMsgSize *int `yaml:"grpcMaxSendMsgSize,omitempty" json:"grpcMaxSendMsgSize"` - MaxTimeout *Duration `yaml:"maxTimeout,omitempty" json:"maxTimeout" tstype:"Duration"` - ReadTimeout *Duration `yaml:"readTimeout,omitempty" json:"readTimeout" tstype:"Duration"` - WriteTimeout *Duration `yaml:"writeTimeout,omitempty" json:"writeTimeout" tstype:"Duration"` - EnableGzip *bool `yaml:"enableGzip,omitempty" json:"enableGzip"` - TLS *TLSConfig `yaml:"tls,omitempty" json:"tls"` - Aliasing *AliasingConfig `yaml:"aliasing" json:"aliasing"` - WaitBeforeShutdown *Duration `yaml:"waitBeforeShutdown,omitempty" json:"waitBeforeShutdown" tstype:"Duration"` - WaitAfterShutdown *Duration `yaml:"waitAfterShutdown,omitempty" json:"waitAfterShutdown" tstype:"Duration"` - IncludeErrorDetails *bool `yaml:"includeErrorDetails,omitempty" json:"includeErrorDetails"` - TrustedIPForwarders []string `yaml:"trustedIPForwarders,omitempty" json:"trustedIPForwarders"` - TrustedIPHeaders []string `yaml:"trustedIPHeaders,omitempty" json:"trustedIPHeaders"` - ResponseHeaders map[string]string `yaml:"responseHeaders,omitempty" json:"responseHeaders"` + ListenV4 *bool `yaml:"listenV4,omitempty" json:"listenV4"` + HttpHostV4 *string `yaml:"httpHostV4,omitempty" json:"httpHostV4"` + ListenV6 *bool `yaml:"listenV6,omitempty" json:"listenV6"` + HttpHostV6 *string `yaml:"httpHostV6,omitempty" json:"httpHostV6"` + HttpPort *int `yaml:"httpPort,omitempty" json:"httpPort"` // Deprecated: use HttpPortV4 + HttpPortV4 *int `yaml:"httpPortV4,omitempty" json:"httpPortV4"` + HttpPortV6 *int `yaml:"httpPortV6,omitempty" json:"httpPortV6"` + GrpcEnabled *bool `yaml:"grpcEnabled,omitempty" json:"grpcEnabled"` + GrpcHostV4 *string `yaml:"grpcHostV4,omitempty" json:"grpcHostV4"` + GrpcPortV4 *int `yaml:"grpcPortV4,omitempty" json:"grpcPortV4"` + GrpcHostV6 *string `yaml:"grpcHostV6,omitempty" json:"grpcHostV6"` + GrpcPortV6 *int `yaml:"grpcPortV6,omitempty" json:"grpcPortV6"` + GrpcMaxRecvMsgSize *int `yaml:"grpcMaxRecvMsgSize,omitempty" json:"grpcMaxRecvMsgSize"` + GrpcMaxSendMsgSize *int `yaml:"grpcMaxSendMsgSize,omitempty" json:"grpcMaxSendMsgSize"` + MaxTimeout *Duration `yaml:"maxTimeout,omitempty" json:"maxTimeout" tstype:"Duration"` + ReadTimeout *Duration `yaml:"readTimeout,omitempty" json:"readTimeout" tstype:"Duration"` + WriteTimeout *Duration `yaml:"writeTimeout,omitempty" json:"writeTimeout" tstype:"Duration"` + EnableGzip *bool `yaml:"enableGzip,omitempty" json:"enableGzip"` + TLS *TLSConfig `yaml:"tls,omitempty" json:"tls"` + Aliasing *AliasingConfig `yaml:"aliasing" json:"aliasing"` + WaitBeforeShutdown *Duration `yaml:"waitBeforeShutdown,omitempty" json:"waitBeforeShutdown" tstype:"Duration"` + WaitAfterShutdown *Duration `yaml:"waitAfterShutdown,omitempty" json:"waitAfterShutdown" tstype:"Duration"` + IncludeErrorDetails *bool `yaml:"includeErrorDetails,omitempty" json:"includeErrorDetails"` + TrustedIPForwarders []string `yaml:"trustedIPForwarders,omitempty" json:"trustedIPForwarders"` + TrustedIPHeaders []string `yaml:"trustedIPHeaders,omitempty" json:"trustedIPHeaders"` + ResponseHeaders map[string]string `yaml:"responseHeaders,omitempty" json:"responseHeaders"` + WebSocket *WebSocketServerConfig `yaml:"webSocket,omitempty" json:"webSocket"` // ExecutionHeaders controls the per-request diagnostic headers // (X-ERPC-Attempts, X-ERPC-Upstreams-Tried, etc.) that expose how @@ -167,6 +184,14 @@ type ServerConfig struct { ExecutionHeaders *ExecutionHeadersMode `yaml:"executionHeaders,omitempty" json:"executionHeaders" tstype:"ExecutionHeadersMode"` } +type WebSocketServerConfig struct { + ReadBufferSize int `yaml:"readBufferSize,omitempty" json:"readBufferSize"` + WriteBufferSize int `yaml:"writeBufferSize,omitempty" json:"writeBufferSize"` + MaxMessageSize int64 `yaml:"maxMessageSize,omitempty" json:"maxMessageSize"` + PingInterval *Duration `yaml:"pingInterval,omitempty" json:"pingInterval" tstype:"Duration"` + MaxSubscriptionsPerConnection int `yaml:"maxSubscriptionsPerConnection,omitempty" json:"maxSubscriptionsPerConnection"` +} + // ExecutionHeadersMode controls how much per-request execution detail is // exposed in HTTP response headers. type ExecutionHeadersMode string @@ -597,6 +622,27 @@ type NetworkDefaults struct { DirectiveDefaults *DirectiveDefaultsConfig `yaml:"directiveDefaults,omitempty" json:"directiveDefaults"` Evm *EvmNetworkConfig `yaml:"evm,omitempty" json:"evm" tstype:"TsEvmNetworkConfigForDefaults"` Multiplexing *bool `yaml:"multiplexing,omitempty" json:"multiplexing"` + Failover *FailoverConfig `yaml:"failover,omitempty" json:"failover"` +} + +// FailoverConfig controls within-request escalation between upstream groups. +// Independent of SelectionPolicy (which evaluates group membership +// periodically across requests) — Failover operates per-request only. +type FailoverConfig struct { + // OnDefaultsExhausted, when true, causes the network request loop to + // try upstreams with group "default" (or unset) first and only advance + // to group "fallback" if every default upstream returned a retryable + // error within the same request. Deterministic client errors still + // short-circuit without advancing. + OnDefaultsExhausted *bool `yaml:"onDefaultsExhausted,omitempty" json:"onDefaultsExhausted"` +} + +// Enabled reports whether any failover behaviour is configured. Nil-safe. +func (f *FailoverConfig) Enabled() bool { + if f == nil { + return false + } + return f.OnDefaultsExhausted != nil && *f.OnDefaultsExhausted } // UnmarshalYAML provides backward compatibility for old single failsafe object format @@ -699,6 +745,13 @@ func (p *ProviderConfig) MarshalYAML() (interface{}, error) { }, nil } +// TagTierFallback marks an upstream as part of the fallback tier (via the +// `tier:fallback` tag convention): used only when all non-fallback upstreams +// are unavailable. Referenced by default selection policies and by +// network-level block-number aggregation so that a more-advanced fallback +// doesn't drag the shared counter ahead of what primaries can actually serve. +const TagTierFallback = "tier:fallback" + type UpstreamConfig struct { Id string `yaml:"id,omitempty" json:"id"` Type UpstreamType `yaml:"type,omitempty" json:"type" tstype:"TsUpstreamType"` @@ -1966,6 +2019,7 @@ type NetworkConfig struct { Methods *MethodsConfig `yaml:"methods,omitempty" json:"methods"` Multiplexing *bool `yaml:"multiplexing,omitempty" json:"multiplexing"` StaticResponses []*StaticResponseConfig `yaml:"staticResponses,omitempty" json:"staticResponses,omitempty"` + Failover *FailoverConfig `yaml:"failover,omitempty" json:"failover"` } // StaticResponseConfig declares a canned JSON-RPC response for a specific @@ -2173,6 +2227,18 @@ type EvmNetworkConfig struct { // Default includes common point-lookup methods like eth_getBlockByNumber, eth_getTransactionByHash, etc. MarkEmptyAsErrorMethods []string `yaml:"markEmptyAsErrorMethods,omitempty" json:"markEmptyAsErrorMethods,omitempty"` + // StripSubscribeFromBlockZero, when true, removes `fromBlock: "0x0"` from + // eth_subscribe logs filters before forwarding to upstream WebSockets. + // Some clients include `fromBlock: "0x0"` in the filter as a + // "from genesis" marker. eth_subscribe is a live-stream RPC — fromBlock + // has no standardised meaning there — and on backends that prune + // historical data the subscription fails outright. Enabling this flag + // for such networks drops the field so the live stream succeeds; + // historical logs remain retrievable via eth_getLogs. Only the exact + // value "0x0" or "0" is stripped — non-zero fromBlocks pass through + // unchanged. DEFAULT: false. + StripSubscribeFromBlockZero *bool `yaml:"stripSubscribeFromBlockZero,omitempty" json:"stripSubscribeFromBlockZero,omitempty"` + // DynamicBlockTimeDebounceMultiplier scales the EMA-estimated block time to derive // the debounce interval for block polling. A value of 0.7 means debounce = 70% of // the estimated block time, preferring fresher data at the cost of slightly more diff --git a/common/config_test.go b/common/config_test.go index 7d5b0bf6f..dc3c0c868 100644 --- a/common/config_test.go +++ b/common/config_test.go @@ -1211,3 +1211,65 @@ projects: assert.Equal(t, 5, network.Failsafe[1].Retry.MaxAttempts) }) } + +// TestNetworkConfig_SetDefaults_FailoverSkipsAutoSelectionPolicy verifies +// that enabling failover.onDefaultsExhausted=true suppresses the auto-applied +// SelectionPolicy that would otherwise filter fallback-group upstreams out +// of the eligible set. The per-request loop needs fallbacks to remain +// visible so it can escalate to them on demand. +func TestNetworkConfig_SetDefaults_FailoverSkipsAutoSelectionPolicy(t *testing.T) { + upstreams := []*UpstreamConfig{ + {Id: "a", Endpoint: "http://a", Type: UpstreamTypeEvm}, + {Id: "b", Endpoint: "http://b", Type: UpstreamTypeEvm, Group: UpstreamGroupFallback}, + } + + t.Run("without failover the auto policy is applied", func(t *testing.T) { + n := &NetworkConfig{Architecture: ArchitectureEvm, Evm: &EvmNetworkConfig{ChainId: 1}} + err := n.SetDefaults(upstreams, nil) + assert.NoError(t, err) + assert.NotNil(t, n.SelectionPolicy, "auto SelectionPolicy should be applied when fallback upstreams exist") + }) + + t.Run("with failover enabled the auto policy is suppressed", func(t *testing.T) { + enabled := true + n := &NetworkConfig{ + Architecture: ArchitectureEvm, + Evm: &EvmNetworkConfig{ChainId: 1}, + Failover: &FailoverConfig{OnDefaultsExhausted: &enabled}, + } + err := n.SetDefaults(upstreams, nil) + assert.NoError(t, err) + assert.Nil(t, n.SelectionPolicy, "auto SelectionPolicy should NOT be applied when failover handles escalation") + }) + + t.Run("user-supplied SelectionPolicy is preserved regardless of failover", func(t *testing.T) { + enabled := true + userPolicy := &SelectionPolicyConfig{EvalInterval: Duration(time.Minute)} + n := &NetworkConfig{ + Architecture: ArchitectureEvm, + Evm: &EvmNetworkConfig{ChainId: 1}, + Failover: &FailoverConfig{OnDefaultsExhausted: &enabled}, + SelectionPolicy: userPolicy, + } + err := n.SetDefaults(upstreams, nil) + assert.NoError(t, err) + assert.NotNil(t, n.SelectionPolicy) + assert.Equal(t, Duration(time.Minute), n.SelectionPolicy.EvalInterval) + }) +} + +// TestNetworkConfig_SetDefaults_FailoverInheritsFromDefaults verifies that a +// failover flag set at the NetworkDefaults level is propagated to networks +// that don't override it. +func TestNetworkConfig_SetDefaults_FailoverInheritsFromDefaults(t *testing.T) { + enabled := true + defaults := &NetworkDefaults{ + Failover: &FailoverConfig{OnDefaultsExhausted: &enabled}, + } + n := &NetworkConfig{Architecture: ArchitectureEvm, Evm: &EvmNetworkConfig{ChainId: 1}} + err := n.SetDefaults(nil, defaults) + assert.NoError(t, err) + assert.NotNil(t, n.Failover) + assert.True(t, n.Failover.Enabled()) +} + diff --git a/common/defaults.go b/common/defaults.go index 54d34289d..ed67efa8a 100644 --- a/common/defaults.go +++ b/common/defaults.go @@ -722,6 +722,25 @@ func (s *ServerConfig) SetDefaults() error { s.ExecutionHeaders = &m } + if s.WebSocket == nil { + s.WebSocket = &WebSocketServerConfig{} + } + if s.WebSocket.ReadBufferSize == 0 { + s.WebSocket.ReadBufferSize = 4096 + } + if s.WebSocket.WriteBufferSize == 0 { + s.WebSocket.WriteBufferSize = 4096 + } + if s.WebSocket.MaxMessageSize == 0 { + s.WebSocket.MaxMessageSize = 1 * 1024 * 1024 // 1MB + } + if s.WebSocket.PingInterval == nil { + d := Duration(30 * time.Second) + s.WebSocket.PingInterval = &d + } + if s.WebSocket.MaxSubscriptionsPerConnection == 0 { + s.WebSocket.MaxSubscriptionsPerConnection = 100 + } // Safe defaults for client IP resolution if len(s.TrustedIPForwarders) == 0 { // Only loopback by default; do not trust private subnets unless explicitly configured @@ -1193,6 +1212,8 @@ func (p *ProjectConfig) SetDefaults(opts *DefaultOptions) error { func convertUpstreamToProvider(upstream *UpstreamConfig) (*ProviderConfig, error) { if strings.HasPrefix(upstream.Endpoint, "http://") || strings.HasPrefix(upstream.Endpoint, "https://") || + strings.HasPrefix(upstream.Endpoint, "ws://") || + strings.HasPrefix(upstream.Endpoint, "wss://") || strings.HasPrefix(upstream.Endpoint, "grpc://") || strings.HasPrefix(upstream.Endpoint, "grpc+bds://") { return nil, nil @@ -1839,6 +1860,14 @@ func (n *NetworkConfig) SetDefaults(upstreams []*UpstreamConfig, defaults *Netwo v := *defaults.Multiplexing n.Multiplexing = &v } + if n.Failover == nil && defaults.Failover != nil { + cp := *defaults.Failover + if defaults.Failover.OnDefaultsExhausted != nil { + v := *defaults.Failover.OnDefaultsExhausted + cp.OnDefaultsExhausted = &v + } + n.Failover = &cp + } if n.Evm != nil && defaults.Evm != nil { if n.Evm.Integrity == nil && defaults.Evm.Integrity != nil { n.Evm.Integrity = &EvmIntegrityConfig{} @@ -1921,7 +1950,7 @@ func (n *NetworkConfig) SetDefaults(upstreams []*UpstreamConfig, defaults *Netwo if len(upstreams) > 0 { anyUpstreamInFallbackTier := slices.ContainsFunc(upstreams, func(u *UpstreamConfig) bool { - return u.HasTag("tier:fallback") + return u.HasTag(TagTierFallback) }) if anyUpstreamInFallbackTier && n.SelectionPolicy == nil { defCfg := NewDefaultNetworkConfig(upstreams) diff --git a/common/errors.go b/common/errors.go index 9634ff4d6..3073feb31 100644 --- a/common/errors.go +++ b/common/errors.go @@ -2779,3 +2779,67 @@ var NewErrEndpointNonceException = func(cause error, reason NonceExceptionReason func (e *ErrEndpointNonceException) ErrorStatusCode() int { return http.StatusOK } + +// +// WebSocket / Subscription errors +// + +type ErrSubscriptionNotFound struct{ BaseError } + +const ErrCodeSubscriptionNotFound ErrorCode = "ErrSubscriptionNotFound" + +var NewErrSubscriptionNotFound = func(subId string) error { + return &ErrSubscriptionNotFound{ + BaseError{ + Code: ErrCodeSubscriptionNotFound, + Message: "subscription not found", + Details: map[string]interface{}{ + "subscriptionId": subId, + }, + }, + } +} + +func (e *ErrSubscriptionNotFound) ErrorStatusCode() int { + return http.StatusNotFound +} + +type ErrNoWsUpstreamAvailable struct{ BaseError } + +const ErrCodeNoWsUpstreamAvailable ErrorCode = "ErrNoWsUpstreamAvailable" + +var NewErrNoWsUpstreamAvailable = func(networkId string) error { + return &ErrNoWsUpstreamAvailable{ + BaseError{ + Code: ErrCodeNoWsUpstreamAvailable, + Message: fmt.Sprintf("eth_subscribe requires a WebSocket-capable upstream, none configured for network %s", networkId), + Details: map[string]interface{}{ + "networkId": networkId, + }, + }, + } +} + +func (e *ErrNoWsUpstreamAvailable) ErrorStatusCode() int { + return http.StatusBadRequest +} + +type ErrSubscriptionLimitExceeded struct{ BaseError } + +const ErrCodeSubscriptionLimitExceeded ErrorCode = "ErrSubscriptionLimitExceeded" + +var NewErrSubscriptionLimitExceeded = func(max int) error { + return &ErrSubscriptionLimitExceeded{ + BaseError{ + Code: ErrCodeSubscriptionLimitExceeded, + Message: "maximum subscriptions per connection exceeded", + Details: map[string]interface{}{ + "maxPerConnection": max, + }, + }, + } +} + +func (e *ErrSubscriptionLimitExceeded) ErrorStatusCode() int { + return http.StatusTooManyRequests +} diff --git a/common/validation.go b/common/validation.go index 569878ee1..2143b03ec 100644 --- a/common/validation.go +++ b/common/validation.go @@ -63,6 +63,23 @@ func (c *Config) Validate() error { } } } + if c.Indexer != nil { + if err := c.Indexer.Validate(); err != nil { + return err + } + } + return nil +} + +// Validate rejects nonsensical values; zeros pass through and resolve to +// internal defaults inside the indexer. +func (i *IndexerConfig) Validate() error { + if i.CanonicalChainDepth < 0 { + return fmt.Errorf("indexer.canonicalChainDepth must be >= 0 (0 uses the default)") + } + if i.DedupWindowSize < 0 { + return fmt.Errorf("indexer.dedupWindowSize must be >= 0 (0 uses the default)") + } return nil } diff --git a/docs/pages/config/database/evm-json-rpc-cache.mdx b/docs/pages/config/database/evm-json-rpc-cache.mdx index 22448fd3f..ca540cade 100644 --- a/docs/pages/config/database/evm-json-rpc-cache.mdx +++ b/docs/pages/config/database/evm-json-rpc-cache.mdx @@ -297,6 +297,8 @@ Compression is **on by default** with Zstandard. Tune via `compression.*`: `threshold` (default `1024` bytes) — values smaller than this are stored uncompressed because the codec overhead outweighs savings. Bumping this up reduces CPU at the cost of disk; bumping down does the opposite. +For `eth_getBlockByNumber` requests that use a moving tag (`"latest"`, `"finalized"`, `"safe"`), eRPC keys the cache entry by the concrete block number — the response's own number on write, the network's tracked tip on read. That means a burst of concurrent tag queries landing on the same tip coalesces onto one cached entry, but once the tip advances the key changes and the next request forwards upstream. You don't have to worry about a `"latest"` query returning a block that's multiple tips stale just because the TTL hasn't expired. + Compression is particularly effective for: - Large block responses (`eth_getBlockByNumber` with full transactions) diff --git a/docs/pages/config/projects/networks.mdx b/docs/pages/config/projects/networks.mdx index 2da448df4..701430a85 100644 --- a/docs/pages/config/projects/networks.mdx +++ b/docs/pages/config/projects/networks.mdx @@ -168,6 +168,54 @@ export default createConfig({ `multiplexing` is also settable under `networkDefaults` and behaves like any other scalar field: a per-network `multiplexing` value wins; if absent, the default applies to every network in the project. +## `eth_subscribe` + +Some JSON-RPC clients include `fromBlock: "0x0"` in the `eth_subscribe("logs", {…})` filter as a "from genesis" marker. `eth_subscribe` is a live-stream RPC — `fromBlock` has no standardised meaning there — and on backends that prune historical data the subscribe call fails outright. Enable `stripSubscribeFromBlockZero` on the network to drop the field before forwarding so the live stream succeeds; historical logs stay retrievable via `eth_getLogs` in the normal way. + + + +```yaml filename="erpc.yaml" +projects: + - id: main + networks: + - architecture: evm + evm: + chainId: 1 + # Drops fromBlock: "0x0" / "0" from eth_subscribe logs filters + # before forwarding to upstream WebSockets. Only the exact + # zero value is stripped — non-zero fromBlocks pass through. + stripSubscribeFromBlockZero: true # default: false +``` + + +```ts filename="erpc.ts" +import { createConfig } from "@erpc-cloud/config"; + +export default createConfig({ + projects: [ + { + id: "main", + networks: [ + { + architecture: "evm", + evm: { + chainId: 1, + // Drops fromBlock: "0x0" / "0" from eth_subscribe logs filters + // before forwarding to upstream WebSockets. Only the exact + // zero value is stripped — non-zero fromBlocks pass through. + stripSubscribeFromBlockZero: true, // default: false + }, + }, + ], + }, + ], +}); +``` + + + +Leave unset for chains where upstreams handle `fromBlock: "0x0"` in an `eth_subscribe` filter correctly (most do — the flag is a targeted workaround, not a universal default). + ## Name aliasing Use a friendly alias instead of the `architecture/chainId` URL segment. Aliases are only available for statically-defined networks (lazy-loaded networks don't have one). diff --git a/docs/pages/config/projects/upstreams.mdx b/docs/pages/config/projects/upstreams.mdx index da90770f7..ae95f0673 100644 --- a/docs/pages/config/projects/upstreams.mdx +++ b/docs/pages/config/projects/upstreams.mdx @@ -40,6 +40,8 @@ The smallest workable upstream is just an endpoint. Everything else has sensible yaml={`projects: - id: main upstreams: + # Endpoint URL supports http(s) and ws(s) schemes. Use ws:// or wss:// for + # WebSocket upstreams (required for eth_subscribe support). - endpoint: https://eth-mainnet.g.alchemy.com/v2/YOUR_KEY # eRPC auto-detects the chain ID and applies default failsafe (15s timeout, # 2 retries, circuit breaker at 80% failure rate).`} diff --git a/docs/pages/operation/_meta.js b/docs/pages/operation/_meta.js index 7165dd26d..cc92eac95 100644 --- a/docs/pages/operation/_meta.js +++ b/docs/pages/operation/_meta.js @@ -2,6 +2,9 @@ module.exports = { "url": { title: "URL", }, + "websocket": { + title: "WebSocket", + }, "healthcheck": { title: "Healthcheck", }, diff --git a/docs/pages/operation/production.mdx b/docs/pages/operation/production.mdx index 18595371c..8dc122fa7 100644 --- a/docs/pages/operation/production.mdx +++ b/docs/pages/operation/production.mdx @@ -104,6 +104,16 @@ Auto-detected chain IDs add one upstream call per network at startup and slow ro - `networks.*.evm.chainId` — under [Networks](/config/projects/networks) - `upstreams.*.evm.chainId` — under [Upstreams](/config/projects/upstreams) +## WebSocket subscriptions + +If you serve [`eth_subscribe`](/operation/websocket) traffic, a few additional notes: + +- **One upstream subscription per (network, filter)** — eRPC de-duplicates across clients. 1,000 clients subscribed to the same `logs` filter consume one upstream subscription, not 1,000. +- **Per-network memory for reorg tracking** — eRPC keeps a bounded ring of the last N canonical block headers per network (default 256, tune via [`indexer.canonicalChainDepth`](/operation/websocket#indexer-configuration)) plus an index of delivered logs keyed by `blockHash`. Upper bound is roughly `ring-depth × logs-per-block × payload-size`. On logs-heavy chains (~50 logs/block, ~400 B each) that is well under 10 MB per network — dominated by log payload, not ring metadata. +- **Pod-local reorg view** — client WS connections are sticky to one pod. The `removed: true` re-emission that pod sends reflects *that pod's* local observation of the canonical chain. Cross-pod exactly-once reorg delivery is not provided; if required, rely on upstream consensus policies. +- **Slow-client policy** — each subscription gets a bounded notification buffer. A slow client drops its *own* oldest queued notifications rather than blocking upstream delivery or other clients. No tuning knob; scale connections horizontally if drops become meaningful. +- **`waitBeforeShutdown` still applies** — WS connections are drained during the grace window along with HTTP requests. A 30s `waitBeforeShutdown` is a safe default when a load balancer sits in front of eRPC. + ## Healthcheck and zero-downtime rollout Configure a [Healthcheck](/operation/healthcheck) readiness probe so your orchestrator stops routing to a pod before it shuts down. diff --git a/docs/pages/operation/url.mdx b/docs/pages/operation/url.mdx index da67cba02..5a0da3f67 100644 --- a/docs/pages/operation/url.mdx +++ b/docs/pages/operation/url.mdx @@ -285,3 +285,13 @@ Network aliases are short names configured inside a project's `networks[]` block Append `.llms.txt` to this URL (or use the **AI** link above) to fetch the entire expanded reference as plain markdown for an AI assistant. + +## WebSocket + +You can also connect via WebSocket using the same URL path with a `ws://` or `wss://` scheme: + +``` +ws://localhost:4000/main/evm/1 +``` + +WebSocket connections support all JSON-RPC methods, batch requests, and real-time subscriptions (`eth_subscribe`). Read more in the [WebSocket](/operation/websocket) page. diff --git a/docs/pages/operation/websocket.mdx b/docs/pages/operation/websocket.mdx new file mode 100644 index 000000000..a6a4767ce --- /dev/null +++ b/docs/pages/operation/websocket.mdx @@ -0,0 +1,378 @@ +--- +description: WebSocket support in eRPC for persistent connections and real-time subscriptions... +--- + +import { Callout, Tabs, Tab } from "nextra/components"; + +# WebSocket + +eRPC supports WebSocket connections on the same HTTP port. Clients can connect via `ws://` or `wss://` for persistent JSON-RPC connections and real-time subscriptions (`eth_subscribe`). + +## Connecting + +Connect to eRPC using the same URL path as HTTP, but with the `ws://` or `wss://` scheme: + +``` +ws:///// +``` + +```javascript +// Browser / Node.js example +const ws = new WebSocket("ws://localhost:4000/main/evm/1"); + +ws.onopen = () => { + ws.send(JSON.stringify({ + jsonrpc: "2.0", + id: 1, + method: "eth_blockNumber", + params: [] + })); +}; + +ws.onmessage = (event) => { + console.log(JSON.parse(event.data)); + // { "jsonrpc": "2.0", "id": 1, "result": "0x1234567" } +}; +``` + + + WebSocket connections use the same port as HTTP — no additional configuration needed. Domain aliasing rules also apply to WebSocket connections. + + +## Regular JSON-RPC over WebSocket + +All JSON-RPC methods supported over HTTP also work over WebSocket. The connection is persistent, so you can send multiple requests without reconnecting: + +```javascript +// Send multiple requests on the same connection +ws.send(JSON.stringify({ jsonrpc: "2.0", id: 1, method: "eth_blockNumber", params: [] })); +ws.send(JSON.stringify({ jsonrpc: "2.0", id: 2, method: "eth_chainId", params: [] })); +ws.send(JSON.stringify({ jsonrpc: "2.0", id: 3, method: "eth_getBlockByNumber", params: ["latest", false] })); +``` + +Batch requests (JSON arrays) are also supported: + +```javascript +ws.send(JSON.stringify([ + { jsonrpc: "2.0", id: 1, method: "eth_blockNumber", params: [] }, + { jsonrpc: "2.0", id: 2, method: "eth_chainId", params: [] } +])); +// Response is a JSON array +``` + +All eRPC features work identically for WebSocket RPC requests: caching, upstream scoring, failsafe policies (retry, hedge, circuit breaker), rate limiting, multiplexing (request deduplication), EVM integrity checks, and metrics. + +## Subscriptions + +eRPC supports `eth_subscribe` and `eth_unsubscribe` for real-time event streaming. Subscriptions require at least one upstream configured with a `ws://` or `wss://` endpoint. + +### eth_subscribe + +```javascript +const ws = new WebSocket("ws://localhost:4000/main/evm/1"); + +ws.onopen = () => { + // Subscribe to new block headers + ws.send(JSON.stringify({ + jsonrpc: "2.0", + id: 1, + method: "eth_subscribe", + params: ["newHeads"] + })); +}; + +ws.onmessage = (event) => { + const msg = JSON.parse(event.data); + + if (msg.id) { + // Subscription confirmation + console.log("Subscription ID:", msg.result); + // "0x2a151383161618ebffddfed9de4be5db" + } + + if (msg.method === "eth_subscription") { + // New block notification + console.log("New block:", msg.params.result.number); + } +}; +``` + +Supported subscription types (depends on upstream node capabilities): +- `newHeads` — new block headers +- `logs` — log entries matching a filter +- `newPendingTransactions` — pending transaction hashes + +#### Subscribing to logs with a filter + +The second params entry is a filter object with optional `address` (string or array) and `topics` (array of topic hex strings or arrays for OR-matching). The filter is hashed by eRPC so multiple clients requesting the same filter share a single underlying upstream subscription. + +```javascript +ws.send(JSON.stringify({ + jsonrpc: "2.0", + id: 1, + method: "eth_subscribe", + params: [ + "logs", + { + address: "0xA0b86991c6218b36c1d19D4a2e9Eb0cE3606eB48", // USDC + topics: [ + "0xddf252ad1be2c89b69c2b068fc378daa952ba7f163c4a11628f55a4df523b3ef" // Transfer + ] + } + ] +})); +``` + +### eth_unsubscribe + +```javascript +ws.send(JSON.stringify({ + jsonrpc: "2.0", + id: 2, + method: "eth_unsubscribe", + params: ["0x2a151383161618ebffddfed9de4be5db"] +})); +// Response: { "jsonrpc": "2.0", "id": 2, "result": true } +``` + +### Subscription features + +- **Resilient delivery** — if you have multiple WebSocket upstreams configured for a network, eRPC subscribes on all of them and delivers blocks/logs/pending txs from whichever is fastest. If one upstream slows down or disconnects, you keep receiving events from the others without any gap. +- **No surprise disconnects** — eRPC never closes your WebSocket connection because of upstream issues. Your subscriptions stay alive through upstream reconnects, scoring changes, and brief outages. +- **Deduplication** — even though multiple upstreams are subscribed in parallel, you only receive each block (or log, or pending tx) once. +- **Shared subscriptions** — if multiple clients subscribe to the same event on the same network, they share a single underlying upstream subscription. Each client gets its own subscription ID and receives every notification independently. +- **Slow-client safety** — each client has its own notification buffer. A client that's slow to read won't slow down other clients or block upstream delivery; if the buffer overflows, the oldest queued notifications are dropped in favor of the newest. +- **Transparent IDs** — eRPC issues its own subscription IDs to clients so you can switch between eRPC instances or upstreams without your subscription IDs changing. +- **Rate limiting** — subscribe/unsubscribe calls count against the same project and network rate limits as regular RPC. + +### Reorg handling for `logs` subscriptions + +eRPC tracks a per-network window of recent block headers (parentHash continuity) and uses it to re-emit invalidated logs when a reorg happens. If you're subscribed to `logs` and a block containing one of your matched logs is reorged out, eRPC re-sends that log with `"removed": true` before announcing the new canonical head — matching the semantics Ethereum clients use over a direct WebSocket connection to a node. + +```javascript +ws.onmessage = (event) => { + const msg = JSON.parse(event.data); + if (msg.method === "eth_subscription" && msg.params.result.removed) { + // This log was in a block that was reorged out. + // Roll back any state you derived from it. + } +}; +``` + +Two `removed` signals can reach clients: + +- **Upstream-asserted** — the upstream itself emitted the log with `removed: true`. eRPC passes it through unchanged. +- **Indexer-detected** — eRPC's own canonical-chain tracker observed the block was evicted. eRPC re-emits the log with `removed: true` from the cached copy it kept when the log first arrived. + +Scope limits: + +- The tracker keeps a ring of the last 256 canonical heads per network by default. Reorgs deeper than the ring window are partially handled — only the immediate head is evicted. Tunable via `indexer.canonicalChainDepth` (see [Indexer configuration](#indexer-configuration) below). +- Logs emitted with `removed: true` originate from eRPC's local view of the canonical chain at the time the reorg was observed. If your operational guarantees require cluster-wide exactly-once reorg notification across multiple eRPC pods, fall back to upstream-level consensus policies. + +### When eth_subscribe fails + +`eth_subscribe` only returns an error when the network has no WebSocket upstreams configured at all. Brief upstream disconnects don't fail your subscribe — eRPC accepts the subscription and starts delivering events as soon as any upstream comes back. + +A failed `eth_subscribe` returns the JSON-RPC error and leaves your WebSocket connection open. You can retry on the same connection. + +## WebSocket upstreams + +To support subscriptions, configure at least one upstream with a `ws://` or `wss://` endpoint: + + + +```yaml filename="erpc.yaml" +projects: + - id: main + upstreams: + # Regular HTTP upstream (for standard RPC) + - id: eth-mainnet-http + endpoint: https://eth-mainnet.example.com + evm: + chainId: 1 + + # WebSocket upstream (enables subscriptions) + - id: eth-mainnet-ws + endpoint: ws://eth-mainnet.example.com:8546 + evm: + chainId: 1 + failsafe: + retry: + maxAttempts: 2 + delay: 100ms + timeout: + duration: 30s +``` + + +```ts filename="erpc.ts" +import { createConfig } from "@erpc-cloud/config"; + +export default createConfig({ + projects: [ + { + id: "main", + upstreams: [ + // Regular HTTP upstream + { + id: "eth-mainnet-http", + endpoint: "https://eth-mainnet.example.com", + evm: { chainId: 1 }, + }, + // WebSocket upstream (enables subscriptions) + { + id: "eth-mainnet-ws", + endpoint: "ws://eth-mainnet.example.com:8546", + evm: { chainId: 1 }, + failsafe: { + retry: { maxAttempts: 2, delay: "100ms" }, + timeout: { duration: "30s" }, + }, + }, + ], + }, + ], +}); +``` + + + + + WebSocket upstreams also serve regular JSON-RPC requests (not just subscriptions). eRPC's scoring system picks the best upstream regardless of transport — HTTP or WebSocket. + + +The WebSocket upstream client features: +- **Persistent connection** with automatic reconnection (exponential backoff, 1s to 30s) +- **Ping/pong keepalive** (30s interval) to detect dead connections +- **Lazy connection** — if the upstream is unavailable at startup, eRPC retries in the background without blocking + +## Server configuration + +WebSocket server settings can be tuned via the `webSocket` section in server config. Defaults work well for most deployments: + +```yaml filename="erpc.yaml" +server: + httpPortV4: 4000 + webSocket: + # Read buffer size in bytes (default: 4096) + readBufferSize: 4096 + # Write buffer size in bytes (default: 4096) + writeBufferSize: 4096 + # Maximum incoming message size in bytes (default: 1048576 = 1MB) + maxMessageSize: 1048576 + # Ping interval for keepalive (default: 30s) + pingInterval: 30s + # Maximum active subscriptions per WebSocket connection (default: 100) + maxSubscriptionsPerConnection: 100 +``` + +## Per-frame wire tracing + +For targeted debugging of client subscription behaviour (early unsubscribes, missing notifications, exact frame shape), eRPC can log every inbound/outbound WS frame on a selected network to `info` level. Leave the list empty in production — enabling it turns the WS write path off gorilla's zero-copy `NextWriter` streaming onto buffered `WriteMessage` so the bytes can be captured, and each frame becomes a log entry. + + + +```yaml filename="erpc.yaml" +diagnostics: + # Per-frame logging for client WS connections whose networkId matches. + # Empty = off. Payload is capped at 4 KiB per event. + wsFrameTraceNetworkIds: + - "evm:1" +``` + + +```ts filename="erpc.ts" +import { createConfig } from "@erpc-cloud/config"; + +export default createConfig({ + diagnostics: { + wsFrameTraceNetworkIds: ["evm:1"], + }, +}); +``` + + + + + The `ERPC_WS_TRACE_NETWORK` environment variable is honored as a legacy single-entry backdoor when `wsFrameTraceNetworkIds` is empty — it seeds the list with the env value. The config field takes precedence when non-empty. + + +A handful of other WS-subscription-related diagnostic events emit at `trace` level and surface when you drop `logLevel` to `trace` (globally or on a specific network logger via zerolog's context-scoped level). These cover per-upstream WS newHead non-monotonicity, and the `eth_getBlockByNumber("latest")` response-vs-tip lag check that fires when the final HTTP "latest" response is below the network's WS-tracked tip. + +## Indexer configuration + +`eth_subscribe` fan-out and reorg-aware `removed: true` emission are handled by an internal event indexer. Most deployments can leave this unset; the tunables exist for chains with unusual reorg windows or very high log throughput. + + + +```yaml filename="erpc.yaml" +indexer: + # Per-network ring-buffer of recent canonical heads used to detect + # reorgs. Reorgs deeper than this are only partially handled — the + # immediate head is evicted and a warning is logged. 0 = default (256). + canonicalChainDepth: 256 + # Per-filter seen-set used to drop duplicate log / pending-tx + # notifications delivered by sibling upstreams. 0 = default (8192). + dedupWindowSize: 8192 +``` + + +```ts filename="erpc.ts" +import { createConfig } from "@erpc-cloud/config"; + +export default createConfig({ + indexer: { + canonicalChainDepth: 256, + dedupWindowSize: 8192, + }, +}); +``` + + + + + Increase `canonicalChainDepth` on chains with historically deep reorgs (e.g. some BSC incidents) or on nodes that batch-announce heads. Memory impact is dominated by the per-block log payload index, not the ring metadata. + + +## Subscription compatibility knobs + +### Stripping `fromBlock: "0x0"` from `eth_subscribe` logs filters + +Some JSON-RPC clients include `fromBlock: "0x0"` in the `eth_subscribe("logs", {…})` filter as a "from genesis" marker. `eth_subscribe` is a live-stream RPC — `fromBlock` has no standardised meaning there — and on backends that prune historical data the subscribe call fails outright. eRPC can strip the field per-network so the live stream succeeds; historical logs stay retrievable via `eth_getLogs` in the normal way. + +```yaml filename="erpc.yaml" +projects: + - id: main + networks: + - architecture: evm + evm: + chainId: 196 + # Drops fromBlock: "0x0" / "0" from eth_subscribe logs filters + # before forwarding to upstream. Non-zero fromBlocks pass through. + stripSubscribeFromBlockZero: true +``` + +Only the exact value `"0x0"` / `"0"` is stripped — non-zero `fromBlock` values pass through unchanged. Leave unset for chains where upstreams handle the field correctly. + +### HTTP / WebSocket tip-sync considerations for strict consumers + +Some RPC consumers compare the block number returned by HTTP `eth_getBlockByNumber("latest")` against a tip they've observed via a WebSocket `newHeads` subscription and flag any lag as a health-check failure. On fast-block chains (sub-second to a few seconds) the HTTP round-trip can trivially exceed the inter-block interval, so the HTTP response carries a block that was current when it was generated but older than a `newHeads` notification that arrived in parallel. That behaviour is inherent to HTTP-vs-push timing and is visible with or without a proxy in front. + +eRPC minimises the gap in three ways: + +- **Per-block cache keying for moving tags.** The realtime cache entry for `eth_getBlockByNumber("latest")` is keyed by the response's actual block number (and looked up by the network's currently-known tip), so each tip advance is a distinct key rather than pinning every call within the TTL to the same cached payload. +- **`enforceHighestBlock` response rewrite.** When the upstream returns a block number below the network's tracked tip, eRPC's [integrity check](/config/failsafe/integrity) fetches the specific higher block from another upstream and returns that instead. +- **WS newHead advances the tip tracker synchronously.** When an upstream WebSocket delivers a `newHeads` notification, eRPC updates its per-upstream tip tracker before the indexer fans the notification out to clients, so the next HTTP request on the same pod sees the fresher tip. + +Even so, a transient single-block lag can reach a client when the chain produces a block faster than an in-flight HTTP round-trip completes. If your consumer treats single-block HTTP-vs-WS lag as a hard failure, widen its tolerance to match the chain's block time rather than expect eRPC (or any proxy) to eliminate it. + +## Directives + +[Request directives](/operation/directives) can be set via headers or query parameters on the initial WebSocket upgrade request. They apply to all regular RPC requests on that connection: + +```javascript +// Pass directives via query params on the WebSocket URL +const ws = new WebSocket("ws://localhost:4000/main/evm/1?useUpstream=my-preferred-upstream"); +``` + +Subscriptions (`eth_subscribe`) ignore the `useUpstream` directive — they're delivered by every WebSocket upstream configured for the network so notifications keep flowing if any upstream slows down. diff --git a/erpc/http_server.go b/erpc/http_server.go index e003c520f..2616118eb 100644 --- a/erpc/http_server.go +++ b/erpc/http_server.go @@ -23,8 +23,10 @@ import ( "github.com/bytedance/sonic" "github.com/erpc/erpc/auth" "github.com/erpc/erpc/common" + "github.com/erpc/erpc/indexer" "github.com/erpc/erpc/telemetry" "github.com/erpc/erpc/util" + "github.com/gorilla/websocket" "github.com/rs/zerolog" "go.opentelemetry.io/otel/attribute" "go.opentelemetry.io/otel/trace" @@ -52,6 +54,8 @@ type HttpServer struct { trustedForwarderIPs map[string]struct{} trustedIPHeaders []string resolvedResponseHeaders map[string]string + subscriptionManager *SubscriptionManager + activeWsConns sync.Map // connId -> *WsConnection } func NewHttpServer( @@ -60,6 +64,7 @@ func NewHttpServer( cfg *common.ServerConfig, healthCheckCfg *common.HealthCheckConfig, adminCfg *common.AdminConfig, + indexerCfg *common.IndexerConfig, erpc *ERPC, ) (*HttpServer, error) { reqMaxTimeout := 150 * time.Second @@ -84,15 +89,25 @@ func NewHttpServer( gzipPool := util.NewGzipReaderPool() + subMgrLogger := logger.With().Str("component", "subscriptions").Logger() + indexerLogger := logger.With().Str("component", "indexer").Logger() + indexerOpts := indexer.Options{} + if indexerCfg != nil { + indexerOpts.CanonicalChainDepth = indexerCfg.CanonicalChainDepth + indexerOpts.DedupWindowSize = indexerCfg.DedupWindowSize + } + idx := indexer.New(&indexerLogger, indexerOpts) + srv := &HttpServer{ - logger: logger, - appCtx: ctx, - serverCfg: cfg, - healthCheckCfg: healthCheckCfg, - adminCfg: adminCfg, - erpc: erpc, - draining: &draining, - gzipPool: gzipPool, + logger: logger, + appCtx: ctx, + serverCfg: cfg, + healthCheckCfg: healthCheckCfg, + adminCfg: adminCfg, + erpc: erpc, + draining: &draining, + gzipPool: gzipPool, + subscriptionManager: NewSubscriptionManager(&subMgrLogger, idx), } if cfg != nil { @@ -346,6 +361,12 @@ func (s *HttpServer) createRequestHandler() http.Handler { } } + // WebSocket upgrade: handle before body reading since WS upgrades don't have a JSON body + if websocket.IsWebSocketUpgrade(r) { + s.handleWebSocket(httpCtx, w, r, &lg, project, architecture, chainId) + return + } + // Handle gzipped request bodies var bodyReader io.Reader = r.Body if r.Header.Get("Content-Encoding") == "gzip" { @@ -998,7 +1019,7 @@ func (s *HttpServer) parseUrlPath( return "", "", "", false, false, common.NewErrInvalidUrlPath("architecture is not valid (must be 'evm')", ps) } - if !isPost && !isOptions { + if !isPost && !isOptions && r.Header.Get("Upgrade") != "websocket" { isHealthCheck = true } @@ -1639,6 +1660,11 @@ func (s *HttpServer) createTLSConfig() (*tls.Config, error) { func (s *HttpServer) Shutdown(logger *zerolog.Logger) error { logger.Info().Msg("stopping http servers...") + // Close all active WebSocket connections first with GoingAway status. + // This sends a close frame to clients so they know to reconnect, + // and cleans up all subscriptions before the HTTP server stops. + s.shutdownWebSockets(logger) + ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second) defer cancel() @@ -1682,6 +1708,48 @@ func (s *HttpServer) Shutdown(logger *zerolog.Logger) error { return lastErr } +// shutdownWebSockets closes all active WebSocket connections with a GoingAway +// close frame and waits for subscription cleanup to complete. +func (s *HttpServer) shutdownWebSockets(logger *zerolog.Logger) { + count := 0 + s.activeWsConns.Range(func(key, value interface{}) bool { + count++ + return true + }) + + if count == 0 { + return + } + + logger.Info().Int("connections", count).Msg("closing active WebSocket connections...") + + var wg sync.WaitGroup + s.activeWsConns.Range(func(key, value interface{}) bool { + wsc := value.(*WsConnection) + wg.Add(1) + go func() { + defer wg.Done() + wsc.CloseWithGoingAway() + }() + s.activeWsConns.Delete(key) + return true + }) + + // Wait for all connections to close with a timeout + done := make(chan struct{}) + go func() { + wg.Wait() + close(done) + }() + + select { + case <-done: + logger.Info().Int("connections", count).Msg("all WebSocket connections closed") + case <-time.After(10 * time.Second): + logger.Warn().Int("connections", count).Msg("timed out waiting for WebSocket connections to close") + } +} + // conditionalGzipWriter wraps ResponseWriter and decides whether to compress // based on the first write size. This avoids buffering while still allowing // us to skip compression for small responses. diff --git a/erpc/http_server_logs_test.go b/erpc/http_server_logs_test.go index e811e63eb..5076c1499 100644 --- a/erpc/http_server_logs_test.go +++ b/erpc/http_server_logs_test.go @@ -118,7 +118,7 @@ func TestHttp_EvmGetLogs_SplitOnError_MergedResponse(t *testing.T) { erpcInstance.Bootstrap(ctx) require.NoError(t, err) - httpServer, err := NewHttpServer(ctx, &logger, cfg.Server, cfg.HealthCheck, cfg.Admin, erpcInstance) + httpServer, err := NewHttpServer(ctx, &logger, cfg.Server, cfg.HealthCheck, cfg.Admin, cfg.Indexer, erpcInstance) require.NoError(t, err) listener, err := net.Listen("tcp", "127.0.0.1:0") @@ -263,7 +263,7 @@ func TestHttp_EvmGetLogs_ProactiveRangeSplit_MergedResponse(t *testing.T) { erpcInstance.Bootstrap(ctx) require.NoError(t, err) - httpServer, err := NewHttpServer(ctx, &logger, cfg.Server, cfg.HealthCheck, cfg.Admin, erpcInstance) + httpServer, err := NewHttpServer(ctx, &logger, cfg.Server, cfg.HealthCheck, cfg.Admin, cfg.Indexer, erpcInstance) require.NoError(t, err) listener, err := net.Listen("tcp", "127.0.0.1:0") @@ -388,7 +388,7 @@ func TestHttp_EvmGetLogs_SplitOnError_ByAddresses_MergedResponse(t *testing.T) { require.NoError(t, err) erpcInstance.Bootstrap(ctx) require.NoError(t, err) - httpServer, err := NewHttpServer(ctx, &logger, cfg.Server, cfg.HealthCheck, cfg.Admin, erpcInstance) + httpServer, err := NewHttpServer(ctx, &logger, cfg.Server, cfg.HealthCheck, cfg.Admin, cfg.Indexer, erpcInstance) require.NoError(t, err) listener, err := net.Listen("tcp", "127.0.0.1:0") require.NoError(t, err) @@ -486,7 +486,7 @@ func TestHttp_EvmGetLogs_SplitOnError_ByTopic0ORList_MergedResponse(t *testing.T require.NoError(t, err) erpcInstance.Bootstrap(ctx) require.NoError(t, err) - httpServer, err := NewHttpServer(ctx, &logger, cfg.Server, cfg.HealthCheck, cfg.Admin, erpcInstance) + httpServer, err := NewHttpServer(ctx, &logger, cfg.Server, cfg.HealthCheck, cfg.Admin, cfg.Indexer, erpcInstance) require.NoError(t, err) listener, err := net.Listen("tcp", "127.0.0.1:0") require.NoError(t, err) @@ -572,7 +572,7 @@ func TestHttp_EvmGetLogs_SplitOnError_EmptyAndNonEmptyMergedSkipsEmpty(t *testin require.NoError(t, err) erpcInstance.Bootstrap(ctx) require.NoError(t, err) - httpServer, err := NewHttpServer(ctx, &logger, cfg.Server, cfg.HealthCheck, cfg.Admin, erpcInstance) + httpServer, err := NewHttpServer(ctx, &logger, cfg.Server, cfg.HealthCheck, cfg.Admin, cfg.Indexer, erpcInstance) require.NoError(t, err) listener, err := net.Listen("tcp", "127.0.0.1:0") require.NoError(t, err) @@ -640,7 +640,7 @@ func TestHttp_ConcurrentIdenticalRequests_NoEmptyBodyParse(t *testing.T) { require.NoError(t, err) erpcInstance.Bootstrap(ctx) require.NoError(t, err) - httpServer, err := NewHttpServer(ctx, &logger, cfg.Server, cfg.HealthCheck, cfg.Admin, erpcInstance) + httpServer, err := NewHttpServer(ctx, &logger, cfg.Server, cfg.HealthCheck, cfg.Admin, cfg.Indexer, erpcInstance) require.NoError(t, err) listener, err := net.Listen("tcp", "127.0.0.1:0") diff --git a/erpc/http_server_test.go b/erpc/http_server_test.go index dc9e6e71f..a61e5ca61 100644 --- a/erpc/http_server_test.go +++ b/erpc/http_server_test.go @@ -111,7 +111,7 @@ func TestHttpServer_RaceTimeouts(t *testing.T) { erpcInstance.Bootstrap(ctx) require.NoError(t, err) - httpServer, err := NewHttpServer(ctx, &logger, cfg.Server, cfg.HealthCheck, cfg.Admin, erpcInstance) + httpServer, err := NewHttpServer(ctx, &logger, cfg.Server, cfg.HealthCheck, cfg.Admin, cfg.Indexer, erpcInstance) require.NoError(t, err) // Start the server on a random port @@ -253,7 +253,7 @@ func TestHttpServer_RaceTimeouts(t *testing.T) { erpcInstance.Bootstrap(ctx) require.NoError(t, err) - httpServer, err := NewHttpServer(ctx, &logger, cfg.Server, cfg.HealthCheck, cfg.Admin, erpcInstance) + httpServer, err := NewHttpServer(ctx, &logger, cfg.Server, cfg.HealthCheck, cfg.Admin, cfg.Indexer, erpcInstance) require.NoError(t, err) // Start the server on a random port @@ -399,7 +399,7 @@ func TestHttpServer_RaceTimeouts(t *testing.T) { erpcInstance.Bootstrap(ctx) require.NoError(t, err) - httpServer, err := NewHttpServer(ctx, &logger, cfg.Server, cfg.HealthCheck, cfg.Admin, erpcInstance) + httpServer, err := NewHttpServer(ctx, &logger, cfg.Server, cfg.HealthCheck, cfg.Admin, cfg.Indexer, erpcInstance) require.NoError(t, err) // Start the server on a random port @@ -7788,7 +7788,7 @@ func createServerTestFixtures(cfg *common.Config, t *testing.T) ( erpcInstance.Bootstrap(ctx) require.NoError(t, err) - httpServer, err := NewHttpServer(ctx, &logger, cfg.Server, cfg.HealthCheck, cfg.Admin, erpcInstance) + httpServer, err := NewHttpServer(ctx, &logger, cfg.Server, cfg.HealthCheck, cfg.Admin, cfg.Indexer, erpcInstance) require.NoError(t, err) listener, err := net.Listen("tcp", "127.0.0.1:0") @@ -8080,7 +8080,7 @@ func TestHttpServer_Evm_GetLogs_MemoryProfile(t *testing.T) { // Give state poller more time to initialize and update shared state time.Sleep(1 * time.Second) - httpServer, err := NewHttpServer(ctx, &logger, cfg.Server, cfg.HealthCheck, cfg.Admin, erpcInstance) + httpServer, err := NewHttpServer(ctx, &logger, cfg.Server, cfg.HealthCheck, cfg.Admin, cfg.Indexer, erpcInstance) require.NoError(t, err) listener, err := net.Listen("tcp", "127.0.0.1:0") diff --git a/erpc/http_timeout.go b/erpc/http_timeout.go index eaee7f918..a330f2dfb 100644 --- a/erpc/http_timeout.go +++ b/erpc/http_timeout.go @@ -34,6 +34,15 @@ type timeoutHandler struct { } func (h *timeoutHandler) ServeHTTP(w http.ResponseWriter, r *http.Request) { + // WebSocket upgrades need direct access to the underlying connection via Hijack(). + // The timeout writer buffers responses and doesn't implement http.Hijacker, + // so we bypass the timeout handler entirely for WS connections. + // WS connections are long-lived and use their own ping/pong for liveness. + if r.Header.Get("Upgrade") == "websocket" { + h.handler.ServeHTTP(w, r) + return + } + ctx, cancelCtx := context.WithTimeoutCause(r.Context(), h.dt, ErrHandlerTimeout) defer func() { cancelCtx() diff --git a/erpc/init.go b/erpc/init.go index 324e853f6..f63be83f9 100644 --- a/erpc/init.go +++ b/erpc/init.go @@ -89,7 +89,7 @@ func Init( // logger.Info().Msg("initializing transports") if cfg.Server != nil { - httpServer, err := NewHttpServer(appCtx, &logger, cfg.Server, cfg.HealthCheck, cfg.Admin, erpcInstance) + httpServer, err := NewHttpServer(appCtx, &logger, cfg.Server, cfg.HealthCheck, cfg.Admin, cfg.Indexer, erpcInstance) if err != nil { return err } diff --git a/erpc/networks.go b/erpc/networks.go index de8faf2c0..9c84b138a 100644 --- a/erpc/networks.go +++ b/erpc/networks.go @@ -13,6 +13,7 @@ import ( "github.com/erpc/erpc/architecture/evm" "github.com/erpc/erpc/common" + "github.com/erpc/erpc/data" "github.com/erpc/erpc/health" "github.com/erpc/erpc/internal/policy" "github.com/erpc/erpc/telemetry" @@ -45,6 +46,15 @@ type Network struct { // permit-acquisition is no longer needed. policyEngine *policy.Engine initializer *util.Initializer + + // latestBlockShared / finalizedBlockShared make the network's block tags + // cross-pod monotonic. Without them, different eRPC instances could return + // regressing values (e.g. 100 then 97) when their local upstream state + // pollers are briefly out of step — a client polling through a load + // balancer would see the regression and treat it as a data-integrity + // failure. May be nil in tests or when shared state is unavailable. + latestBlockShared data.CounterInt64SharedVariable + finalizedBlockShared data.CounterInt64SharedVariable } // Bootstrap registers this network with the policy engine. The engine kicks @@ -254,29 +264,14 @@ func (n *Network) EvmHighestLatestBlockNumber(ctx context.Context) int64 { ctx, span := common.StartDetailSpan(ctx, "Network.EvmHighestLatestBlockNumber") defer span.End() - upstreams := n.upstreamsRegistry.GetNetworkUpstreams(ctx, n.networkId) - var maxBlock int64 = 0 - for _, u := range upstreams { - statePoller := u.EvmStatePoller() - if statePoller == nil { - continue - } - - // Check if the node is syncing - skip syncing nodes as their block numbers may be unreliable - if u.EvmSyncingState() == common.EvmSyncingStateSyncing { - n.logger.Debug().Str("upstreamId", u.Id()).Msg("skipping syncing upstream for highest latest block calculation") - continue - } - - // Use effective latest block which considers blockAvailability.upper config - // (e.g., if upstream has latestBlockMinus: 5, use latest-5 instead of latest) - upBlock := u.EvmEffectiveLatestBlock() - if upBlock > maxBlock { - maxBlock = upBlock - } - } - span.SetAttributes(attribute.Int64("highest_latest_block", maxBlock)) - return maxBlock + result := n.evmHighestBlockNumber( + ctx, + "eth_blockNumber", + (*upstream.Upstream).EvmEffectiveLatestBlock, + n.latestBlockShared, + ) + span.SetAttributes(attribute.Int64("highest_latest_block", result)) + return result } func (n *Network) EvmHighestFinalizedBlockNumber(ctx context.Context) int64 { @@ -285,28 +280,75 @@ func (n *Network) EvmHighestFinalizedBlockNumber(ctx context.Context) int64 { )) defer span.End() - upstreams := n.upstreamsRegistry.GetNetworkUpstreams(ctx, n.networkId) - var maxBlock int64 = 0 - for _, u := range upstreams { - statePoller := u.EvmStatePoller() - if statePoller == nil { + result := n.evmHighestBlockNumber( + ctx, + "eth_getBlockByNumber", + (*upstream.Upstream).EvmEffectiveFinalizedBlock, + n.finalizedBlockShared, + ) + span.SetAttributes(attribute.Int64("highest_finalized_block", result)) + return result +} + +// evmHighestBlockNumber aggregates a per-upstream block number across a +// network and reconciles it with the cross-pod shared counter. +// +// Primary vs. fallback: fallback-group upstreams may run ahead of primaries +// (e.g. 3rd-party providers vs. our own nodes). Feeding their values into +// the shared counter causes tag translation to ask for blocks primaries +// cannot yet serve. We therefore only use the fallback max when no primary +// is up — "up" meaning circuit breaker closed, i.e. not a heuristic. +// +// Shared counter: we only ever publish forward progress. If our local max +// exceeds the high-water mark we advance it; otherwise we return the shared +// value unchanged. Publishing a lower value is not safe even though the +// counter has a rollback tolerance — a large-enough drop crosses that +// threshold and would wrongly clobber the high-water mark, producing the +// exact cross-pod regression we want to prevent. +func (n *Network) evmHighestBlockNumber( + ctx context.Context, + selectionMethod string, + blockOf func(*upstream.Upstream) int64, + shared data.CounterInt64SharedVariable, +) int64 { + var primaryMax, fallbackMax int64 + anyPrimaryUp := false + for _, u := range n.upstreamsRegistry.GetNetworkUpstreams(ctx, n.networkId) { + if u.EvmStatePoller() == nil { continue } - - // Check if the node is syncing - skip syncing nodes as their block numbers may be unreliable if u.EvmSyncingState() == common.EvmSyncingStateSyncing { - n.logger.Debug().Str("upstreamId", u.Id()).Msg("skipping syncing upstream for highest finalized block calculation") + n.logger.Debug().Str("upstreamId", u.Id()).Msg("skipping syncing upstream for highest block calculation") continue } - - // Use effective finalized block which considers blockAvailability.upper config - upBlock := u.EvmEffectiveFinalizedBlock() - if upBlock > maxBlock { - maxBlock = upBlock + upBlock := blockOf(u) + if u.Config().HasTag(common.TagTierFallback) { + if upBlock > fallbackMax { + fallbackMax = upBlock + } + continue + } + if !u.IsDown() { + anyPrimaryUp = true } + if upBlock > primaryMax { + primaryMax = upBlock + } + } + + localMax := primaryMax + if fallbackMax > 0 && !anyPrimaryUp { + localMax = fallbackMax } - span.SetAttributes(attribute.Int64("highest_finalized_block", maxBlock)) - return maxBlock + + if shared == nil { + return localMax + } + sharedVal := shared.GetValue() + if localMax > sharedVal { + return shared.TryUpdate(ctx, localMax) + } + return sharedVal } func (n *Network) EvmLowestFinalizedBlockNumber(ctx context.Context) int64 { @@ -507,6 +549,15 @@ func (n *Network) Forward(ctx context.Context, req *common.NormalizedRequest) (* return nil, err } + // Failover tiering: when enabled, order default-group upstreams ahead of + // fallback-group ones while preserving score order within each tier. The + // network request loop then naturally tries defaults first and only + // advances to fallbacks once every default has returned a retryable + // error within this request. + if n.cfg.Failover.Enabled() { + upsList = tierUpstreamsByGroup(upsList) + } + // Set upstreams on the request req.SetUpstreams(upsList) @@ -1746,3 +1797,33 @@ func (n *Network) acquireRateLimitPermit(ctx context.Context, req *common.Normal return nil } + +// tierUpstreamsByGroup returns a copy of ups with primary-tier upstreams +// (those NOT tagged `tier:fallback`) ordered before fallback-tier upstreams. +// Within each tier, input order is preserved (the caller's score-based +// sort). If the input contains no fallback-tier upstreams, the original +// slice is returned unchanged. +func tierUpstreamsByGroup(ups []common.Upstream) []common.Upstream { + hasFallback := false + for _, u := range ups { + if u.Config() != nil && u.Config().HasTag(common.TagTierFallback) { + hasFallback = true + break + } + } + if !hasFallback { + return ups + } + tiered := make([]common.Upstream, 0, len(ups)) + for _, u := range ups { + if u.Config() == nil || !u.Config().HasTag(common.TagTierFallback) { + tiered = append(tiered, u) + } + } + for _, u := range ups { + if u.Config() != nil && u.Config().HasTag(common.TagTierFallback) { + tiered = append(tiered, u) + } + } + return tiered +} diff --git a/erpc/networks_consensus_test.go b/erpc/networks_consensus_test.go index 977729eba..502afc2d2 100644 --- a/erpc/networks_consensus_test.go +++ b/erpc/networks_consensus_test.go @@ -697,6 +697,7 @@ func TestConsensusPolicy(t *testing.T) { }, retryPolicy: &common.RetryPolicyConfig{ MaxAttempts: 2, + Delay: common.Duration(50 * time.Millisecond), }, mockResponses: []mockResponse{ {status: 200, body: jsonRpcError(-32603, "internal server error")}, diff --git a/erpc/networks_registry.go b/erpc/networks_registry.go index b7652a513..4183ec944 100644 --- a/erpc/networks_registry.go +++ b/erpc/networks_registry.go @@ -170,6 +170,22 @@ func NewNetwork( nwCfg.Architecture = common.ArchitectureEvm } + // Cross-pod monotonic block counters. Tolerates up to 1024-block rollbacks + // (same threshold as per-upstream state pollers) so a rare deep reorg can + // still correct the value, but routine per-upstream jitter cannot. + if upstreamsRegistry != nil { + if ssr := upstreamsRegistry.SharedStateRegistry(); ssr != nil { + network.latestBlockShared = ssr.GetCounterInt64( + fmt.Sprintf("network/%s/latestBlock", netId), + evm.DefaultToleratedBlockHeadRollback, + ) + network.finalizedBlockShared = ssr.GetCounterInt64( + fmt.Sprintf("network/%s/finalizedBlock", netId), + evm.DefaultToleratedBlockHeadRollback, + ) + } + } + return network, nil } diff --git a/erpc/networks_test.go b/erpc/networks_test.go index 380497e86..b4b250db4 100644 --- a/erpc/networks_test.go +++ b/erpc/networks_test.go @@ -12221,6 +12221,214 @@ func TestNetwork_HighestFinalizedBlockNumber(t *testing.T) { assert.Equal(t, int64(1800), highest, "Should exclude syncing nodes even when they have upper bounds configured") }) + + t.Run("EvmHighestFinalizedBlockNumber_MonotonicAcrossPods", func(t *testing.T) { + // Regression guard for the "finalized tag regresses across load-balanced + // instances" failure mode. Without the shared counter, a caller polling + // through a load balancer can see a lower value than it saw on a prior + // poll — clients that enforce repeatable-read semantics treat that as a + // data-integrity violation and mark eRPC unhealthy. + util.ResetGock() + defer util.ResetGock() + util.SetupMocksForEvmStatePoller() + + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + + up := &common.UpstreamConfig{ + Type: common.UpstreamTypeEvm, + Id: "only-node", + Endpoint: "http://only.localhost", + Evm: &common.EvmUpstreamConfig{ChainId: 123}, + } + + gock.New("http://only.localhost"). + Post(""). + Persist(). + Filter(func(r *http.Request) bool { + return strings.Contains(util.SafeReadBody(r), `eth_chainId`) + }). + Reply(200). + JSON([]byte(`{"result":"0x7b"}`)) + + rateLimitersRegistry, _ := upstream.NewRateLimitersRegistry(context.Background(), &common.RateLimiterConfig{}, &log.Logger) + metricsTracker := health.NewTracker(&log.Logger, "test", time.Minute) + + vr := thirdparty.NewVendorsRegistry() + pr, err := thirdparty.NewProvidersRegistry(&log.Logger, vr, []*common.ProviderConfig{}, nil) + require.NoError(t, err) + + ssr, err := data.NewSharedStateRegistry(ctx, &log.Logger, &common.SharedStateConfig{ + Connector: &common.ConnectorConfig{ + Driver: "memory", + Memory: &common.MemoryConnectorConfig{MaxItems: 100_000, MaxTotalSize: "1GB"}, + }, + }) + require.NoError(t, err) + + upstreamsRegistry := upstream.NewUpstreamsRegistry( + ctx, &log.Logger, "test", + []*common.UpstreamConfig{up}, ssr, rateLimitersRegistry, vr, pr, nil, + metricsTracker, 1*time.Second, nil, nil, + ) + + networkConfig := &common.NetworkConfig{ + Architecture: common.ArchitectureEvm, + Evm: &common.EvmNetworkConfig{ChainId: 123}, + } + network, err := NewNetwork(ctx, &log.Logger, "test", networkConfig, + rateLimitersRegistry, upstreamsRegistry, metricsTracker) + require.NoError(t, err) + + upstreamsRegistry.Bootstrap(ctx) + time.Sleep(200 * time.Millisecond) + require.NoError(t, upstreamsRegistry.GetInitializer().WaitForTasks(ctx)) + require.NoError(t, network.Bootstrap(ctx)) + time.Sleep(250 * time.Millisecond) + + upsList := upstreamsRegistry.GetNetworkUpstreams(ctx, util.EvmNetworkId(123)) + require.Len(t, upsList, 1) + u := upsList[0] + + // Initial finalized: 1000 + u.EvmStatePoller().SuggestLatestBlock(1000) + u.EvmStatePoller().SuggestFinalizedBlock(1000) + time.Sleep(50 * time.Millisecond) + + got := network.EvmHighestFinalizedBlockNumber(ctx) + assert.Equal(t, int64(1000), got, "first observation should be 1000") + + // Simulate a peer instance that has already observed a higher finalized + // value (1050) by pushing directly into the network-level shared + // counter. This is how cross-pod propagation lands on this instance. + require.NotNil(t, network.finalizedBlockShared, "network-level shared counter should be initialized") + network.finalizedBlockShared.TryUpdate(ctx, 1050) + + // Even though THIS instance's local upstream state still says 1000, + // we must return at least 1050 — clients relying on monotonic finalized + // progression must never observe a regression. + got = network.EvmHighestFinalizedBlockNumber(ctx) + assert.GreaterOrEqual(t, got, int64(1050), + "must not regress below the shared (peer-observed) value") + + // Local advances past the shared value — we return the new max. + u.EvmStatePoller().SuggestFinalizedBlock(1100) + time.Sleep(50 * time.Millisecond) + got = network.EvmHighestFinalizedBlockNumber(ctx) + assert.Equal(t, int64(1100), got, "should advance when local exceeds shared") + + // Now simulate a local regression: the upstream goes syncing, so + // EvmHighestFinalizedBlockNumber's pre-shared max is 0. The returned + // value must still be ≥ 1100 (the high-water mark stored in shared). + u.EvmStatePoller().SetSyncingState(common.EvmSyncingStateSyncing) + got = network.EvmHighestFinalizedBlockNumber(ctx) + assert.GreaterOrEqual(t, got, int64(1100), + "must not regress when local view temporarily drops (upstream syncing / excluded)") + }) + + t.Run("EvmHighestFinalizedBlockNumber_FallbackExcludedWhenPrimariesUp", func(t *testing.T) { + util.ResetGock() + defer util.ResetGock() + util.SetupMocksForEvmStatePoller() + + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + + primary := &common.UpstreamConfig{ + Type: common.UpstreamTypeEvm, + Id: "primary-node", + Endpoint: "http://primary.localhost", + Evm: &common.EvmUpstreamConfig{ChainId: 123}, + } + fallback := &common.UpstreamConfig{ + Type: common.UpstreamTypeEvm, + Id: "fallback-node", + Endpoint: "http://fallback.localhost", + Group: "fallback", + Evm: &common.EvmUpstreamConfig{ChainId: 123}, + } + + gock.New("http://primary.localhost").Post("").Persist(). + Filter(func(r *http.Request) bool { return strings.Contains(util.SafeReadBody(r), `eth_chainId`) }). + Reply(200).JSON([]byte(`{"result":"0x7b"}`)) + gock.New("http://fallback.localhost").Post("").Persist(). + Filter(func(r *http.Request) bool { return strings.Contains(util.SafeReadBody(r), `eth_chainId`) }). + Reply(200).JSON([]byte(`{"result":"0x7b"}`)) + + rateLimitersRegistry, _ := upstream.NewRateLimitersRegistry(context.Background(), &common.RateLimiterConfig{}, &log.Logger) + metricsTracker := health.NewTracker(&log.Logger, "test", time.Minute) + + vr := thirdparty.NewVendorsRegistry() + pr, err := thirdparty.NewProvidersRegistry(&log.Logger, vr, []*common.ProviderConfig{}, nil) + require.NoError(t, err) + + ssr, err := data.NewSharedStateRegistry(ctx, &log.Logger, &common.SharedStateConfig{ + Connector: &common.ConnectorConfig{ + Driver: "memory", + Memory: &common.MemoryConnectorConfig{MaxItems: 100_000, MaxTotalSize: "1GB"}, + }, + }) + require.NoError(t, err) + + upstreamsRegistry := upstream.NewUpstreamsRegistry( + ctx, &log.Logger, "test", + []*common.UpstreamConfig{primary, fallback}, ssr, rateLimitersRegistry, vr, pr, nil, + metricsTracker, 1*time.Second, nil, nil, + ) + + networkConfig := &common.NetworkConfig{ + Architecture: common.ArchitectureEvm, + Evm: &common.EvmNetworkConfig{ChainId: 123}, + } + network, err := NewNetwork(ctx, &log.Logger, "test", networkConfig, + rateLimitersRegistry, upstreamsRegistry, metricsTracker) + require.NoError(t, err) + + upstreamsRegistry.Bootstrap(ctx) + time.Sleep(200 * time.Millisecond) + require.NoError(t, upstreamsRegistry.GetInitializer().WaitForTasks(ctx)) + require.NoError(t, network.Bootstrap(ctx)) + time.Sleep(250 * time.Millisecond) + + upsList := upstreamsRegistry.GetNetworkUpstreams(ctx, util.EvmNetworkId(123)) + require.Len(t, upsList, 2) + + var primaryUp, fallbackUp *upstream.Upstream + for _, ups := range upsList { + if ups.Id() == "primary-node" { + primaryUp = ups + } else if ups.Id() == "fallback-node" { + fallbackUp = ups + } + } + require.NotNil(t, primaryUp) + require.NotNil(t, fallbackUp) + + // Primary at 1000, fallback at 1050 (ahead). + primaryUp.EvmStatePoller().SuggestLatestBlock(1000) + primaryUp.EvmStatePoller().SuggestFinalizedBlock(1000) + fallbackUp.EvmStatePoller().SuggestLatestBlock(1050) + fallbackUp.EvmStatePoller().SuggestFinalizedBlock(1050) + + // Wait until the seeded values are observable via the upstream's + // block accessors. The poller is still running in the background and + // may overwrite until it gives up on the unmocked endpoints, so we + // poll with a bounded deadline rather than sleeping blindly. + require.Eventually(t, func() bool { + return primaryUp.EvmEffectiveFinalizedBlock() == 1000 && + fallbackUp.EvmEffectiveFinalizedBlock() == 1050 + }, 2*time.Second, 10*time.Millisecond, + "primary/fallback block values should settle after Suggest*Block") + + // With primary up: should use primary value (1000), not fallback (1050) + got := network.EvmHighestFinalizedBlockNumber(ctx) + assert.Equal(t, int64(1000), got, + "should use primary value when primaries are up, even if fallback is higher") + + got = network.EvmHighestLatestBlockNumber(ctx) + assert.Equal(t, int64(1000), got, + "should use primary latest when primaries are up, even if fallback is higher") + }) } func TestNetwork_CacheEmptyBehavior(t *testing.T) { @@ -12304,3 +12512,103 @@ func TestNetwork_CacheEmptyBehavior(t *testing.T) { cache.AssertExpectations(t) }) } + +// minimalUpstream is a stub implementation of common.Upstream used only to +// test tierUpstreamsByGroup, which relies on Id() and Config(). All other +// methods are unused in this test and left as panics to surface accidental +// coupling. +type minimalUpstream struct { + common.Upstream // embed to inherit nil methods; direct calls will panic + id string + cfg *common.UpstreamConfig +} + +func (m *minimalUpstream) Id() string { return m.id } +func (m *minimalUpstream) Config() *common.UpstreamConfig { return m.cfg } + +func newStubUpstream(id, group string) common.Upstream { + return &minimalUpstream{ + id: id, + cfg: &common.UpstreamConfig{Id: id, Group: group}, + } +} + +func TestTierUpstreamsByGroup(t *testing.T) { + t.Run("no fallback returns input slice unchanged", func(t *testing.T) { + in := []common.Upstream{ + newStubUpstream("a", ""), + newStubUpstream("b", ""), + } + out := tierUpstreamsByGroup(in) + // Same backing array (helper returns input unchanged). + require.Len(t, out, 2) + assert.Equal(t, "a", out[0].Id()) + assert.Equal(t, "b", out[1].Id()) + }) + + t.Run("defaults placed before fallbacks preserving order within tier", func(t *testing.T) { + // Input order mixes default / fallback — simulates the score-sorted + // list the request loop receives before tiering. + in := []common.Upstream{ + newStubUpstream("fallback-hi", common.UpstreamGroupFallback), + newStubUpstream("default-lo", ""), + newStubUpstream("default-hi", ""), + newStubUpstream("fallback-lo", common.UpstreamGroupFallback), + } + out := tierUpstreamsByGroup(in) + require.Len(t, out, 4) + // Defaults first, in their original relative order. + assert.Equal(t, "default-lo", out[0].Id()) + assert.Equal(t, "default-hi", out[1].Id()) + // Fallbacks last, in their original relative order. + assert.Equal(t, "fallback-hi", out[2].Id()) + assert.Equal(t, "fallback-lo", out[3].Id()) + }) + + t.Run("only fallbacks still orders them after empty default tier", func(t *testing.T) { + in := []common.Upstream{ + newStubUpstream("fb-1", common.UpstreamGroupFallback), + newStubUpstream("fb-2", common.UpstreamGroupFallback), + } + out := tierUpstreamsByGroup(in) + require.Len(t, out, 2) + assert.Equal(t, "fb-1", out[0].Id()) + assert.Equal(t, "fb-2", out[1].Id()) + }) + + t.Run("unknown group treated as default", func(t *testing.T) { + in := []common.Upstream{ + newStubUpstream("fb", common.UpstreamGroupFallback), + newStubUpstream("custom", "experimental"), + newStubUpstream("def", ""), + } + out := tierUpstreamsByGroup(in) + require.Len(t, out, 3) + // Only "fallback" is recognised as fallback tier; everything else + // (including custom group names) stays in the primary tier. + assert.Equal(t, "custom", out[0].Id()) + assert.Equal(t, "def", out[1].Id()) + assert.Equal(t, "fb", out[2].Id()) + }) +} + +func TestFailoverConfig_Enabled(t *testing.T) { + t.Run("nil is disabled", func(t *testing.T) { + var f *common.FailoverConfig + assert.False(t, f.Enabled()) + }) + t.Run("empty is disabled", func(t *testing.T) { + f := &common.FailoverConfig{} + assert.False(t, f.Enabled()) + }) + t.Run("onDefaultsExhausted=false is disabled", func(t *testing.T) { + v := false + f := &common.FailoverConfig{OnDefaultsExhausted: &v} + assert.False(t, f.Enabled()) + }) + t.Run("onDefaultsExhausted=true is enabled", func(t *testing.T) { + v := true + f := &common.FailoverConfig{OnDefaultsExhausted: &v} + assert.True(t, f.Enabled()) + }) +} diff --git a/erpc/subscription_manager.go b/erpc/subscription_manager.go new file mode 100644 index 000000000..54b78f152 --- /dev/null +++ b/erpc/subscription_manager.go @@ -0,0 +1,598 @@ +package erpc + +import ( + "context" + "fmt" + "net/url" + "strings" + "sync" + "sync/atomic" + "time" + + "github.com/erpc/erpc/common" + "github.com/erpc/erpc/indexer" + "github.com/erpc/erpc/indexer/adapters/wsclient" + "github.com/erpc/erpc/indexer/adapters/wsupstream" + "github.com/erpc/erpc/telemetry" + "github.com/erpc/erpc/upstream" + "github.com/rs/zerolog" +) + +// JSON-RPC subscription methods. Kept in erpc/ rather than the indexer +// because they are JSON-RPC-specific — a Kafka or gRPC egress deals in +// SubType only, never in RPC method strings. +const ( + MethodEthSubscribe = "eth_subscribe" + MethodEthUnsubscribe = "eth_unsubscribe" +) + +// Subscription type aliases for convenient use in the erpc package. +// The canonical definitions live in the indexer package. +const ( + SubTypeNewHeads = indexer.SubTypeNewHeads + SubTypeLogs = indexer.SubTypeLogs + SubTypeNewPendingTransactions = indexer.SubTypeNewPendingTransactions +) + +const ( + // unsubscribeTimeout is the deadline for best-effort upstream + // unsubscribe calls during connection cleanup. + unsubscribeTimeout = 5 * time.Second +) + +// SubscriptionManager is the client-facing egress layer. It owns +// per-connection *wsclient.Adapter instances, lazily registers networks + +// ingresses with the indexer the first time a client subscribes on a +// given network, and translates the public eth_subscribe / eth_unsubscribe +// surface into indexer calls. +type SubscriptionManager struct { + logger *zerolog.Logger + idx *indexer.Indexer + + // conns maps connId -> *connEntry. One egress adapter per live WS + // connection. + conns sync.Map + + // bySubID maps a client-facing subscription ID to the record needed + // to route Unsubscribe/Cleanup without walking every connection. + bySubID sync.Map // clientSubId -> *subRecord + + // networks tracks which networkIds have been bootstrapped with + // ingresses so we don't double-register on every Subscribe call. + networks sync.Map // networkId -> struct{} + + // bootstrapMu serialises bootstrapNetwork; the indexer's + // RegisterNetwork is idempotent but ingress creation (WS connects on + // upstreams) is not, so we avoid duplicate adapters. + bootstrapMu sync.Mutex +} + +// connEntry is the per-connection bookkeeping: the egress adapter and +// the indexer detach handle. +type connEntry struct { + adapter *wsclient.Adapter + detach func() +} + +// subRecord is the per-subscription record kept for Unsubscribe / +// CleanupConnection routing. We don't persist these in the adapter +// because the adapter can't know the original subType (it holds +// EventKind, which is a lossy projection of subType for filters). +type subRecord struct { + clientSubID string + connID string + networkID string + subType string + kind indexer.EventKind + filterHash string +} + +// NewSubscriptionManager creates a client-facing SubscriptionManager +// backed by the given indexer. +func NewSubscriptionManager(logger *zerolog.Logger, idx *indexer.Indexer) *SubscriptionManager { + return &SubscriptionManager{ + logger: logger, + idx: idx, + } +} + +// IsSubscriptionMethod returns true when the JSON-RPC method targets the +// subscription surface (eth_subscribe or eth_unsubscribe). +func IsSubscriptionMethod(method string) bool { + return method == MethodEthSubscribe || method == MethodEthUnsubscribe +} + +// IsSubscribeMethod returns true when the method is eth_subscribe. +func IsSubscribeMethod(method string) bool { + return method == MethodEthSubscribe +} + +// Subscribe handles an eth_subscribe request from a client WS connection. +// Generates a client-facing subscription ID, ensures the corresponding +// upstream subscription exists (via the indexer's EnsureFilter), and +// registers the client with the connection's egress adapter. +func (sm *SubscriptionManager) Subscribe( + ctx context.Context, + wsc *WsConnection, + nq *common.NormalizedRequest, + project *PreparedProject, + networkId string, +) (*common.NormalizedResponse, error) { + start := time.Now() + method := MethodEthSubscribe + lg := sm.logger.With().Str("connId", wsc.id).Str("networkId", networkId).Logger() + + nw, err := project.GetNetwork(ctx, networkId) + if err != nil { + return nil, err + } + nq.SetNetwork(nw) + + // Bootstrap the network if this is the first touch. Returns + // ErrNoWsUpstreamAvailable if the network has no WS-capable + // upstreams configured. + if err := sm.bootstrapNetwork(ctx, nw); err != nil { + return nil, err + } + + // Per-connection subscription limit. Get/create the egress adapter + // up-front — we need it for the limit check and for the subsequent + // AddSubscription call anyway. + conn := sm.getOrCreateConn(wsc) + maxSubs := wsc.server.serverCfg.WebSocket.MaxSubscriptionsPerConnection + if conn.adapter.Count() >= maxSubs { + return nil, common.NewErrSubscriptionLimitExceeded(maxSubs) + } + + if err := sm.acquireRateLimits(ctx, project, nw, nq); err != nil { + return nil, err + } + + reqFinality := nq.Finality(ctx) + telemetry.CounterHandle(telemetry.MetricNetworkRequestsReceived, + project.Config.Id, nw.Label(), method, + reqFinality.String(), nq.UserId(), nq.AgentName(), + ).Inc() + + jrReq, err := nq.JsonRpcRequest() + if err != nil { + sm.recordFailureMetrics(project, nw, method, reqFinality, start, nq, err) + return nil, err + } + + subType := indexer.ExtractSubscriptionType(jrReq.Params) + clientSubID, err := indexer.GenerateClientSubID() + if err != nil { + sm.recordFailureMetrics(project, nw, method, reqFinality, start, nq, fmt.Errorf("failed to generate subscription ID: %w", err)) + return nil, fmt.Errorf("failed to generate subscription ID: %w", err) + } + + kind, filterHash, err := sm.resolveSubscription(ctx, networkId, subType, jrReq.Params) + if err != nil { + sm.recordFailureMetrics(project, nw, method, reqFinality, start, nq, err) + return nil, err + } + + conn.adapter.AddSubscription(clientSubID, networkId, kind, filterHash) + sm.bySubID.Store(clientSubID, &subRecord{ + clientSubID: clientSubID, + connID: wsc.id, + networkID: networkId, + subType: subType, + kind: kind, + filterHash: filterHash, + }) + + lg.Info(). + Str("clientSubId", clientSubID). + Str("subType", subType). + Msg("subscription established") + + telemetry.CounterHandle(telemetry.MetricNetworkSuccessfulRequests, + project.Config.Id, nw.Label(), "proxy", "proxy", + method, "1", reqFinality.String(), "false", nq.UserId(), nq.AgentName(), + ).Inc() + telemetry.ObserverHandle(telemetry.MetricNetworkRequestDuration, + project.Config.Id, nw.Label(), "proxy", "proxy", + method, reqFinality.String(), nq.UserId(), + ).Observe(time.Since(start).Seconds()) + + return sm.buildSubscribeResponse(nq, jrReq, clientSubID), nil +} + +// Unsubscribe handles an eth_unsubscribe request. +func (sm *SubscriptionManager) Unsubscribe( + ctx context.Context, + wsc *WsConnection, + nq *common.NormalizedRequest, + project *PreparedProject, + networkId string, +) (*common.NormalizedResponse, error) { + start := time.Now() + method := MethodEthUnsubscribe + lg := sm.logger.With().Str("connId", wsc.id).Str("networkId", networkId).Logger() + + nw, err := project.GetNetwork(ctx, networkId) + if err != nil { + return nil, err + } + nq.SetNetwork(nw) + + if err := sm.acquireRateLimits(ctx, project, nw, nq); err != nil { + return nil, err + } + + reqFinality := nq.Finality(ctx) + telemetry.CounterHandle(telemetry.MetricNetworkRequestsReceived, + project.Config.Id, nw.Label(), method, + reqFinality.String(), nq.UserId(), nq.AgentName(), + ).Inc() + + jrReq, err := nq.JsonRpcRequest() + if err != nil { + sm.recordFailureMetrics(project, nw, method, reqFinality, start, nq, err) + return nil, err + } + + clientSubID, err := indexer.ExtractClientSubID(jrReq.Params) + if err != nil { + sm.recordFailureMetrics(project, nw, method, reqFinality, start, nq, err) + return nil, err + } + + recRaw, ok := sm.bySubID.LoadAndDelete(clientSubID) + if !ok { + err := common.NewErrSubscriptionNotFound(clientSubID) + sm.recordFailureMetrics(project, nw, method, reqFinality, start, nq, err) + return nil, err + } + rec := recRaw.(*subRecord) + + if connRaw, ok := sm.conns.Load(rec.connID); ok { + connRaw.(*connEntry).adapter.RemoveSubscription(clientSubID) + } + if rec.kind != indexer.KindNewHead && rec.filterHash != "" { + sm.idx.ReleaseFilter(ctx, rec.networkID, rec.subType, rec.filterHash) + } + + lg.Info().Str("clientSubId", clientSubID).Str("subType", rec.subType).Msg("subscription removed") + + telemetry.CounterHandle(telemetry.MetricNetworkSuccessfulRequests, + project.Config.Id, nw.Label(), "proxy", "proxy", + method, "1", reqFinality.String(), "false", nq.UserId(), nq.AgentName(), + ).Inc() + telemetry.ObserverHandle(telemetry.MetricNetworkRequestDuration, + project.Config.Id, nw.Label(), "proxy", "proxy", + method, reqFinality.String(), nq.UserId(), + ).Observe(time.Since(start).Seconds()) + + resp := common.NewNormalizedResponse().WithRequest(nq) + jrr := &common.JsonRpcResponse{} + _ = jrr.SetID(jrReq.ID) + jrr.SetResult([]byte("true")) + resp.WithJsonRpcResponse(jrr) + return resp, nil +} + +// CleanupConnection is invoked on WS disconnect. It walks the adapter's +// active subscriptions, releases each filter refcount in the indexer, +// drains the adapter, and detaches it from the indexer's egress set. +func (sm *SubscriptionManager) CleanupConnection(wsc *WsConnection, _ *PreparedProject) { + lg := sm.logger.With().Str("connId", wsc.id).Logger() + + connRaw, ok := sm.conns.LoadAndDelete(wsc.id) + if !ok { + return + } + conn := connRaw.(*connEntry) + + for _, sub := range conn.adapter.Subscriptions() { + sm.bySubID.Delete(sub.ClientSubID) + if sub.Kind != indexer.KindNewHead && sub.FilterHash != "" { + ctx, cancel := context.WithTimeout(context.Background(), unsubscribeTimeout) + sm.idx.ReleaseFilter(ctx, sub.NetworkID, subTypeFor(sub.Kind), sub.FilterHash) + cancel() + } + } + conn.adapter.Drain() + conn.detach() + + lg.Debug().Msg("cleaned up all subscriptions for connection") +} + +// --- internals -------------------------------------------------------- + +// buildWsAdapterOptions resolves network-level toggles that the wsupstream +// adapter needs into a flat Options struct. Returns nil when no override +// applies so the adapter keeps its defaults. +func buildWsAdapterOptions(cfg *common.NetworkConfig) *wsupstream.Options { + if cfg == nil || cfg.Evm == nil { + return nil + } + opts := &wsupstream.Options{} + set := false + if cfg.Evm.StripSubscribeFromBlockZero != nil && *cfg.Evm.StripSubscribeFromBlockZero { + opts.StripSubscribeFromBlockZero = true + set = true + } + if !set { + return nil + } + return opts +} + +// bootstrapNetwork registers the network with the indexer and attaches a +// wsupstream.Adapter for each WS upstream on the network. Idempotent per +// networkId — subsequent calls are no-ops. +func (sm *SubscriptionManager) bootstrapNetwork(ctx context.Context, nw *Network) error { + networkID := nw.networkId + if _, ok := sm.networks.Load(networkID); ok { + return nil + } + sm.bootstrapMu.Lock() + defer sm.bootstrapMu.Unlock() + if _, ok := sm.networks.Load(networkID); ok { + return nil + } + + wsUpstreams := nw.upstreamsRegistry.GetWsUpstreams(ctx, networkID) + if len(wsUpstreams) == 0 { + return common.NewErrNoWsUpstreamAvailable(networkID) + } + + sm.idx.RegisterNetwork(&networkHandle{nw: nw}) + sm.idx.RegisterNetworkSelector(networkID, &subIngressSelector{nw: nw, networkID: networkID}) + adapterOpts := buildWsAdapterOptions(nw.cfg) + for _, up := range wsUpstreams { + adapter := wsupstream.New(up, networkID, sm.logger, adapterOpts) + if adapter == nil { + continue + } + if err := sm.idx.AddIngress(ctx, networkID, adapter); err != nil { + sm.logger.Warn().Err(err).Str("upstreamId", up.Id()). + Msg("failed to register upstream ingress with indexer") + } + } + sm.networks.Store(networkID, struct{}{}) + return nil +} + +// subIngressSelector routes filter subscribes through the same upstream +// selector used by the HTTP request path: score-ordered, circuit-breaker +// aware, and group-tiered when failover.onDefaultsExhausted is set. The +// output is a list of EventIngress names (matching wsupstream.Adapter.Name() +// == "ws:" + upstreamId). +type subIngressSelector struct { + nw *Network + networkID string +} + +// Select returns (defaults, fallbacks) for a filter subscribe. Both tiers +// are ordered by the upstream registry's score for eth_subscribe; upstreams +// whose circuit breaker is open are skipped. Fallback-group upstreams only +// populate the fallback tier when network-level failover.onDefaultsExhausted +// is enabled — otherwise they mix into the defaults, matching the HTTP +// selector's behaviour for non-failover networks. +func (s *subIngressSelector) Select(_networkId, _subType string, _params []interface{}) (defaults, fallbacks []string) { + if s == nil || s.nw == nil || s.nw.upstreamsRegistry == nil { + return nil, nil + } + // A single method key keeps subscribe-path scoring warm across subTypes + // and mirrors how the HTTP path warms method-scoped sort lists. + ups, err := s.nw.upstreamsRegistry.GetSortedUpstreams(context.Background(), s.networkID, MethodEthSubscribe) + if err != nil { + return nil, nil + } + + failoverOn := s.nw.cfg != nil && s.nw.cfg.Failover.Enabled() + for _, u := range ups { + up, ok := u.(*upstream.Upstream) + if !ok { + continue + } + cfg := up.Config() + if cfg == nil { + continue + } + if !isWsEndpoint(cfg.Endpoint) { + continue + } + if up.IsDown() { + continue + } + name := "ws:" + up.Id() + if failoverOn && cfg.HasTag(common.TagTierFallback) { + fallbacks = append(fallbacks, name) + continue + } + defaults = append(defaults, name) + } + return defaults, fallbacks +} + +func isWsEndpoint(endpoint string) bool { + parsed, err := url.Parse(endpoint) + if err != nil { + return false + } + return parsed.Scheme == "ws" || parsed.Scheme == "wss" +} + +// resolveSubscription validates the subType and, for filter subs, +// translates params into a filterHash via the indexer. Returns +// (kind, filterHash) for subsequent AddSubscription. +func (sm *SubscriptionManager) resolveSubscription(ctx context.Context, networkID, subType string, params []interface{}) (indexer.EventKind, string, error) { + switch subType { + case indexer.SubTypeNewHeads: + return indexer.KindNewHead, "", nil + case indexer.SubTypeLogs, indexer.SubTypeNewPendingTransactions: + hash, err := sm.idx.EnsureFilter(ctx, networkID, subType, params) + if err != nil { + return 0, "", err + } + kind := indexer.KindLog + if subType == indexer.SubTypeNewPendingTransactions { + kind = indexer.KindPendingTx + } + return kind, hash, nil + default: + return 0, "", fmt.Errorf("unsupported subscription type: %s", subType) + } +} + +// getOrCreateConn returns the egress adapter for a WsConnection, +// attaching a new one to the indexer if this is the first subscription +// on that connection. +func (sm *SubscriptionManager) getOrCreateConn(wsc *WsConnection) *connEntry { + if existing, ok := sm.conns.Load(wsc.id); ok { + return existing.(*connEntry) + } + adapter := wsclient.New(wsc.id, wsc, sm.logger) + detach := sm.idx.Attach(adapter) + entry := &connEntry{adapter: adapter, detach: detach} + if existing, loaded := sm.conns.LoadOrStore(wsc.id, entry); loaded { + // Lost the race with another Subscribe — drop our adapter. + detach() + adapter.Drain() + return existing.(*connEntry) + } + return entry +} + +// acquireRateLimits acquires rate-limit permits at both project and +// network level. +func (sm *SubscriptionManager) acquireRateLimits( + ctx context.Context, + project *PreparedProject, + nw *Network, + nq *common.NormalizedRequest, +) error { + if err := project.AcquireRateLimitPermit(ctx, nq); err != nil { + return err + } + return nw.acquireRateLimitPermit(ctx, nq) +} + +// buildSubscribeResponse constructs the JSON-RPC response carrying the +// client-facing subscription ID. +func (sm *SubscriptionManager) buildSubscribeResponse( + nq *common.NormalizedRequest, + jrReq *common.JsonRpcRequest, + clientSubID string, +) *common.NormalizedResponse { + resp := common.NewNormalizedResponse().WithRequest(nq) + jrr := &common.JsonRpcResponse{} + _ = jrr.SetID(jrReq.ID) + jrr.SetResult([]byte(fmt.Sprintf(`"%s"`, clientSubID))) + resp.WithJsonRpcResponse(jrr) + return resp +} + +// recordFailureMetrics emits the failure-path metrics when a subscribe +// or unsubscribe request fails before reaching the indexer. +func (sm *SubscriptionManager) recordFailureMetrics( + project *PreparedProject, + nw *Network, + method string, + finality common.DataFinalityState, + start time.Time, + nq *common.NormalizedRequest, + err error, +) { + telemetry.CounterHandle(telemetry.MetricNetworkFailedRequests, + project.Config.Id, nw.Label(), method, + "0", // no upstream attempts for client-facing subscription failures + common.ErrorFingerprint(err), + string(common.ClassifySeverity(err)), + finality.String(), + nq.UserId(), + nq.AgentName(), + ).Inc() + telemetry.ObserverHandle(telemetry.MetricNetworkRequestDuration, + project.Config.Id, nw.Label(), "", "", + method, finality.String(), nq.UserId(), + ).Observe(time.Since(start).Seconds()) +} + +// subTypeFor is the inverse of the kind-to-subType mapping done in +// resolveSubscription. Used by CleanupConnection where we've only got +// the adapter's Kind in hand. +func subTypeFor(kind indexer.EventKind) string { + switch kind { + case indexer.KindLog: + return indexer.SubTypeLogs + case indexer.KindPendingTx: + return indexer.SubTypeNewPendingTransactions + } + return "" +} + +// internalReqIdCounter generates unique JSON-RPC IDs for internal +// requests (subscribe/unsubscribe). Kept here so erpc-level callers +// still have access to a uniform counter; the wsupstream adapter +// maintains its own counter since it can't import erpc. +var internalReqIdCounter atomic.Int64 + +// internalReqIdOffset keeps internal IDs out of the range clients +// typically use (small incrementing integers). +const internalReqIdOffset = 900_000_000 + +// buildJsonRpcBody marshals a JSON-RPC request with the given method and +// params. Retained for non-subscription JSON-RPC send paths that still +// live inside erpc/. +func buildJsonRpcBody(method string, params interface{}) ([]byte, error) { + id := internalReqIdCounter.Add(1) + internalReqIdOffset + return common.SonicCfg.Marshal(map[string]interface{}{ + "jsonrpc": "2.0", + "id": id, + "method": method, + "params": params, + }) +} + +// --- NetworkHandle ---------------------------------------------------- + +// networkHandle adapts *Network to indexer.NetworkHandle. Lives here +// (rather than in indexer/) because it touches Network internals; the +// indexer package deliberately doesn't know about erpc. +type networkHandle struct { + nw *Network +} + +func (h *networkHandle) Id() string { return h.nw.networkId } + +func (h *networkHandle) FinalityDepth() int64 { + if h.nw.cfg != nil && h.nw.cfg.Evm != nil { + return h.nw.cfg.Evm.FallbackFinalityDepth + } + return 0 +} + +// SuggestLatestBlock routes a per-source block observation to the +// upstream's state poller. sourceId is the ingress adapter's Name(), +// which for wsupstream.Adapter is "ws:". +func (h *networkHandle) SuggestLatestBlock(sourceId string, blockNumber int64) { + const prefix = "ws:" + if !strings.HasPrefix(sourceId, prefix) { + return + } + upstreamID := sourceId[len(prefix):] + for _, u := range h.nw.upstreamsRegistry.GetNetworkUpstreams(context.Background(), h.nw.networkId) { + if u.Id() != upstreamID { + continue + } + poller := u.EvmStatePoller() + if poller == nil || poller.IsObjectNull() { + return + } + poller.SuggestLatestBlock(blockNumber) + return + } +} + +// Interface checks: fail the build if either contract drifts. +var ( + _ wsclient.NotificationWriter = (*WsConnection)(nil) + _ indexer.NetworkHandle = (*networkHandle)(nil) +) + diff --git a/erpc/ws_server.go b/erpc/ws_server.go new file mode 100644 index 000000000..e3c53354e --- /dev/null +++ b/erpc/ws_server.go @@ -0,0 +1,741 @@ +package erpc + +import ( + "context" + "encoding/json" + "fmt" + "net/http" + "runtime/debug" + "sync" + "sync/atomic" + "time" + + "github.com/erpc/erpc/auth" + "github.com/erpc/erpc/common" + "github.com/erpc/erpc/telemetry" + "github.com/gorilla/websocket" + "github.com/rs/zerolog" +) + +// wsConnCounter is an atomic counter for generating unique WebSocket connection IDs. +var wsConnCounter int64 + +// WsConnection represents a single client WebSocket connection to the proxy. +// Each connection tracks its own subscriptions and enforces per-connection limits. +type WsConnection struct { + id string + conn *websocket.Conn + appCtx context.Context + cancel context.CancelFunc + logger *zerolog.Logger + + server *HttpServer + project *PreparedProject + subscriptionManager *SubscriptionManager + architecture string + chainId string + networkId string + httpReq *http.Request // original upgrade request for auth/headers + + // Write synchronization (gorilla/websocket requires synchronized writes) + writeMu sync.Mutex + + // Subscription state lives on the per-connection wsclient.Adapter + // owned by the SubscriptionManager — see indexer/adapters/wsclient. + // WsConnection no longer tracks subscriptions directly. + + closed atomic.Bool +} + +// handleWebSocket upgrades an HTTP connection to WebSocket and runs the +// read/write loops for the lifetime of the connection. +func (s *HttpServer) handleWebSocket( + httpCtx context.Context, + w http.ResponseWriter, + r *http.Request, + lg *zerolog.Logger, + project *PreparedProject, + architecture string, + chainId string, +) { + wsCfg := s.serverCfg.WebSocket + + upgrader := websocket.Upgrader{ + ReadBufferSize: wsCfg.ReadBufferSize, + WriteBufferSize: wsCfg.WriteBufferSize, + CheckOrigin: func(r *http.Request) bool { + return checkWsOrigin(r, project) + }, + } + + wsConn, err := upgrader.Upgrade(w, r, nil) + if err != nil { + lg.Error().Err(err).Msg("websocket upgrade failed") + return + } + + networkId := fmt.Sprintf("%s:%s", architecture, chainId) + + connCtx, connCancel := context.WithCancel(s.appCtx) + wsc := &WsConnection{ + id: fmt.Sprintf("ws-%d", atomic.AddInt64(&wsConnCounter, 1)), + conn: wsConn, + appCtx: connCtx, + cancel: connCancel, + logger: lg, + server: s, + project: project, + subscriptionManager: s.subscriptionManager, + architecture: architecture, + chainId: chainId, + networkId: networkId, + httpReq: r, + } + + lg.Info().Str("connId", wsc.id).Str("remoteAddr", r.RemoteAddr).Msg("websocket connection established") + + // Track active connection for graceful shutdown + s.activeWsConns.Store(wsc.id, wsc) + + wsConn.SetReadLimit(wsCfg.MaxMessageSize) + + // Set up pong handler to extend read deadline on pong receipt + pingInterval := wsCfg.PingInterval.Duration() + wsConn.SetPongHandler(func(string) error { + lg.Trace().Str("connId", wsc.id).Msg("websocket pong received") + return wsConn.SetReadDeadline(time.Now().Add(pingInterval * 2)) + }) + + // Record the close code/reason the peer sent before the gorilla library + // swallows it. This is the only place we can observe how the remote side + // is terminating the connection (vs. network-level drops). + wsConn.SetCloseHandler(func(code int, text string) error { + lg.Info(). + Str("connId", wsc.id). + Int("closeCode", code). + Str("closeReason", text). + Msg("websocket close frame received from peer") + // Mirror gorilla's default behavior: send a close frame back. + message := websocket.FormatCloseMessage(code, "") + _ = wsConn.WriteControl(websocket.CloseMessage, message, time.Now().Add(wsWriteDeadline)) + return nil + }) + + go wsc.pingLoop(pingInterval) + + // Run read loop (blocks until connection closes) + wsc.readLoop() + + // Cleanup + s.activeWsConns.Delete(wsc.id) + wsc.Close() +} + +// checkWsOrigin validates the WebSocket upgrade request origin against +// the project's CORS policy. If no CORS config is set, all origins are allowed. +func checkWsOrigin(r *http.Request, project *PreparedProject) bool { + if project == nil || project.Config.CORS == nil { + return true + } + + origin := r.Header.Get("Origin") + if origin == "" { + return true + } + + for _, allowedOrigin := range project.Config.CORS.AllowedOrigins { + match, err := common.WildcardMatch(allowedOrigin, origin) + if err != nil { + continue + } + if match { + return true + } + } + return false +} + +// +// --- Read loop and message dispatch --- +// + +func (wsc *WsConnection) readLoop() { + for { + if wsc.appCtx.Err() != nil { + return + } + + _, message, err := wsc.conn.ReadMessage() + if err != nil { + if wsc.appCtx.Err() != nil { + return + } + // Log the specific error class — this is the only signal we get + // for whether the peer closed cleanly, the TCP connection dropped, + // or our read deadline expired (missed pongs). Previously all of + // these were flattened into one "connection closed by client" line. + ev := wsc.logger.Info().Str("connId", wsc.id).Err(err). + Str("errType", fmt.Sprintf("%T", err)) + if ce, ok := err.(*websocket.CloseError); ok { + ev = ev.Int("closeCode", ce.Code).Str("closeReason", ce.Text) + } + if websocket.IsUnexpectedCloseError(err, websocket.CloseNormalClosure, websocket.CloseGoingAway) { + ev.Msg("websocket read ended: unexpected close") + } else { + ev.Msg("websocket read ended") + } + return + } + + go wsc.handleMessage(message) + } +} + +func (wsc *WsConnection) handleMessage(raw []byte) { + defer func() { + if rec := recover(); rec != nil { + telemetry.MetricUnexpectedPanicTotal.WithLabelValues( + "ws-request-handler", + fmt.Sprintf("project:%s network:%s", wsc.project.Config.Id, wsc.networkId), + common.ErrorFingerprint(rec), + ).Inc() + wsc.logger.Error(). + Interface("panic", rec). + Str("stack", string(debug.Stack())). + Str("connId", wsc.id). + Msg("unexpected panic in websocket request handler") + } + }() + + startedAt := time.Now() + + isBatch := len(raw) > 0 && raw[0] == '[' + if isBatch { + wsc.handleBatch(raw, &startedAt) + return + } + + wsc.handleSingleRequest(raw, &startedAt) +} + +// +// --- Single request handling --- +// + +func (wsc *WsConnection) handleSingleRequest(raw []byte, startedAt *time.Time) { + nq := common.NewNormalizedRequest(raw) + nq.ForwardHeaders = make(http.Header) + + requestCtx := common.StartRequestSpan(wsc.appCtx, nq) + + clientIP := wsc.server.resolveRealClientIP(wsc.httpReq) + nq.SetClientIP(clientIP) + + if err := nq.Validate(); err != nil { + wsc.writeErrorResponse(nq, err, startedAt, &common.TRUE) + common.EndRequestSpan(requestCtx, nil, err) + return + } + + wsc.applyForwardHeaders(nq) + + method, _ := nq.Method() + + if !wsc.isMethodAllowed(method) { + wsc.writeMethodNotSupportedError(nq, method) + common.EndRequestSpan(requestCtx, nil, nil) + return + } + + if err := wsc.authenticate(requestCtx, nq, method); err != nil { + wsc.writeErrorResponse(nq, err, startedAt, wsc.server.serverCfg.IncludeErrorDetails) + common.EndRequestSpan(requestCtx, nil, err) + return + } + + nw, err := wsc.project.GetNetwork(wsc.appCtx, wsc.networkId) + if err != nil { + wsc.writeErrorResponse(nq, err, startedAt, wsc.server.serverCfg.IncludeErrorDetails) + common.EndRequestSpan(requestCtx, nil, err) + return + } + nq.SetNetwork(nw) + + nq.ApplyDirectiveDefaults(nw.Config().DirectiveDefaults) + uaMode := common.UserAgentTrackingModeSimplified + if wsc.project != nil && wsc.project.Config.UserAgentMode != "" { + uaMode = wsc.project.Config.UserAgentMode + } + nq.EnrichFromHttp(wsc.httpReq.Header, wsc.httpReq.URL.Query(), uaMode) + + // Subscription methods have their own dedicated handling path + if IsSubscriptionMethod(method) { + wsc.handleSubscriptionMethod(requestCtx, nq, method, startedAt) + return + } + + // Forward the request through the normal chain + resp, err := wsc.project.Forward(requestCtx, wsc.networkId, nq) + if err != nil { + if resp != nil { + go resp.Release() + } + wsc.writeErrorResponse(nq, err, startedAt, wsc.server.serverCfg.IncludeErrorDetails) + common.EndRequestSpan(requestCtx, nil, err) + return + } + + wsc.writeNormalizedResponse(resp) + common.EndRequestSpan(requestCtx, resp, nil) +} + +// handleSubscriptionMethod routes eth_subscribe and eth_unsubscribe to the +// subscription manager. +func (wsc *WsConnection) handleSubscriptionMethod(requestCtx context.Context, nq *common.NormalizedRequest, method string, startedAt *time.Time) { + var resp *common.NormalizedResponse + var err error + + if IsSubscribeMethod(method) { + resp, err = wsc.subscriptionManager.Subscribe(requestCtx, wsc, nq, wsc.project, wsc.networkId) + } else { + resp, err = wsc.subscriptionManager.Unsubscribe(requestCtx, wsc, nq, wsc.project, wsc.networkId) + } + + if err != nil { + if resp != nil { + go resp.Release() + } + wsc.writeErrorResponse(nq, err, startedAt, wsc.server.serverCfg.IncludeErrorDetails) + common.EndRequestSpan(requestCtx, nil, err) + return + } + + wsc.writeNormalizedResponse(resp) + common.EndRequestSpan(requestCtx, resp, nil) +} + +// applyForwardHeaders copies matching headers from the original upgrade +// request to the normalized request per the project's ForwardHeaders config. +func (wsc *WsConnection) applyForwardHeaders(nq *common.NormalizedRequest) { + if wsc.project == nil { + return + } + for _, matchKey := range wsc.project.Config.ForwardHeaders { + for key, values := range wsc.httpReq.Header { + matches, err := common.WildcardMatch(matchKey, key) + if err != nil { + continue + } + if matches { + for _, value := range values { + nq.ForwardHeaders.Add(matchKey, value) + } + } + } + } +} + +// authenticate validates the request against the project's auth config. +// Returns nil if authentication succeeds or no auth is configured. +func (wsc *WsConnection) authenticate(requestCtx context.Context, nq *common.NormalizedRequest, method string) error { + if wsc.project == nil { + return nil + } + + ap, err := auth.NewPayloadFromHttp(method, wsc.httpReq.RemoteAddr, wsc.httpReq.Header, wsc.httpReq.URL.Query()) + if err != nil { + return err + } + user, err := wsc.project.AuthenticateConsumer(requestCtx, nq, method, ap) + if err != nil { + return err + } + nq.SetUser(user) + return nil +} + +// writeMethodNotSupportedError writes a JSON-RPC error response for +// methods that are blocked by the project's allowlist/denylist. +func (wsc *WsConnection) writeMethodNotSupportedError(nq *common.NormalizedRequest, method string) { + jsonrpcVersion := "2.0" + var reqId interface{} + if jrr, err := nq.JsonRpcRequest(); err == nil { + jsonrpcVersion = jrr.JSONRPC + reqId = jrr.ID + } + resp := map[string]interface{}{ + "jsonrpc": jsonrpcVersion, + "id": reqId, + "error": map[string]interface{}{ + "code": int(common.JsonRpcErrorUnsupportedException), + "message": fmt.Sprintf("method not supported: %s", method), + }, + } + _ = wsc.writeJSON(resp) +} + +// +// --- Batch request handling --- +// + +func (wsc *WsConnection) handleBatch(raw []byte, startedAt *time.Time) { + var requests []json.RawMessage + if err := common.SonicCfg.Unmarshal(raw, &requests); err != nil { + errResp := map[string]interface{}{ + "jsonrpc": "2.0", + "id": nil, + "error": map[string]interface{}{ + "code": -32700, + "message": "parse error", + }, + } + _ = wsc.writeJSON(errResp) + return + } + + responses := make([]interface{}, len(requests)) + var wg sync.WaitGroup + + for i, reqBody := range requests { + wg.Add(1) + go func(index int, reqRaw json.RawMessage) { + defer wg.Done() + defer func() { + if rec := recover(); rec != nil { + wsc.logger.Error().Interface("panic", rec).Msg("panic in batch websocket request") + responses[index] = processErrorBody(wsc.logger, startedAt, nil, fmt.Errorf("%v", rec), wsc.server.serverCfg.IncludeErrorDetails) + } + }() + + wsc.handleBatchItem(index, reqRaw, startedAt, responses) + }(i, reqBody) + } + + wg.Wait() + + wsc.writeBatchResponse(responses) + + for _, resp := range responses { + if r, ok := resp.(*common.NormalizedResponse); ok { + go r.Release() + } + } +} + +// handleBatchItem processes a single request within a batch. Subscription +// methods are rejected in batch requests since they require a persistent +// connection context. +func (wsc *WsConnection) handleBatchItem(index int, reqRaw json.RawMessage, startedAt *time.Time, responses []interface{}) { + nq := common.NewNormalizedRequest(reqRaw) + nq.ForwardHeaders = make(http.Header) + requestCtx := common.StartRequestSpan(wsc.appCtx, nq) + + clientIP := wsc.server.resolveRealClientIP(wsc.httpReq) + nq.SetClientIP(clientIP) + + if err := nq.Validate(); err != nil { + responses[index] = processErrorBody(wsc.logger, startedAt, nq, err, &common.TRUE) + common.EndRequestSpan(requestCtx, nil, err) + return + } + + method, _ := nq.Method() + + // Subscription methods are not supported in batch requests + if IsSubscriptionMethod(method) { + responses[index] = wsc.buildUnsupportedMethodResponse(nq, "subscription methods (eth_subscribe, eth_unsubscribe) are not supported in batch requests") + common.EndRequestSpan(requestCtx, nil, nil) + return + } + + if !wsc.isMethodAllowed(method) { + responses[index] = wsc.buildUnsupportedMethodResponse(nq, fmt.Sprintf("method not supported: %s", method)) + common.EndRequestSpan(requestCtx, nil, nil) + return + } + + if wsc.project != nil { + ap, err := auth.NewPayloadFromHttp(method, wsc.httpReq.RemoteAddr, wsc.httpReq.Header, wsc.httpReq.URL.Query()) + if err != nil { + responses[index] = processErrorBody(wsc.logger, startedAt, nq, err, &common.TRUE) + common.EndRequestSpan(requestCtx, nil, err) + return + } + user, err := wsc.project.AuthenticateConsumer(requestCtx, nq, method, ap) + if err != nil { + responses[index] = processErrorBody(wsc.logger, startedAt, nq, err, wsc.server.serverCfg.IncludeErrorDetails) + common.EndRequestSpan(requestCtx, nil, err) + return + } + nq.SetUser(user) + } + + nw, err := wsc.project.GetNetwork(wsc.appCtx, wsc.networkId) + if err != nil { + responses[index] = processErrorBody(wsc.logger, startedAt, nq, err, wsc.server.serverCfg.IncludeErrorDetails) + common.EndRequestSpan(requestCtx, nil, err) + return + } + nq.SetNetwork(nw) + nq.ApplyDirectiveDefaults(nw.Config().DirectiveDefaults) + + resp, err := wsc.project.Forward(requestCtx, wsc.networkId, nq) + if err != nil { + if resp != nil { + go resp.Release() + } + responses[index] = processErrorBody(wsc.logger, startedAt, nq, err, wsc.server.serverCfg.IncludeErrorDetails) + common.EndRequestSpan(requestCtx, nil, err) + return + } + + responses[index] = resp + common.EndRequestSpan(requestCtx, resp, nil) +} + +// buildUnsupportedMethodResponse constructs a JSON-RPC error response for +// methods that cannot be used in the current context. +func (wsc *WsConnection) buildUnsupportedMethodResponse(nq *common.NormalizedRequest, message string) *HttpJsonRpcErrorResponse { + jsonrpcVersion := "2.0" + var reqId interface{} + if jrr, err := nq.JsonRpcRequest(); err == nil { + jsonrpcVersion = jrr.JSONRPC + reqId = jrr.ID + } + return &HttpJsonRpcErrorResponse{ + Jsonrpc: jsonrpcVersion, + Id: reqId, + Error: map[string]interface{}{ + "code": int(common.JsonRpcErrorUnsupportedException), + "message": message, + }, + } +} + +// +// --- Method filtering --- +// + +func (wsc *WsConnection) isMethodAllowed(method string) bool { + if wsc.project == nil { + return true + } + + shouldHandle := true + + if wsc.project.Config.IgnoreMethods != nil { + for _, m := range wsc.project.Config.IgnoreMethods { + match, _ := common.WildcardMatch(m, method) + if match { + shouldHandle = false + break + } + } + } + + if wsc.project.Config.AllowMethods != nil { + for _, m := range wsc.project.Config.AllowMethods { + match, _ := common.WildcardMatch(m, method) + if match { + shouldHandle = true + break + } + } + } + + return shouldHandle +} + +// +// --- Write helpers --- +// + +// wsWriteDeadline is the timeout applied to all WebSocket write operations. +const wsWriteDeadline = 10 * time.Second + +func (wsc *WsConnection) writeJSON(v interface{}) error { + wsc.writeMu.Lock() + defer wsc.writeMu.Unlock() + + if wsc.closed.Load() { + return fmt.Errorf("connection closed") + } + + if err := wsc.conn.SetWriteDeadline(time.Now().Add(wsWriteDeadline)); err != nil { + return err + } + defer wsc.conn.SetWriteDeadline(time.Time{}) + + return wsc.conn.WriteJSON(v) +} + +func (wsc *WsConnection) writeMessage(messageType int, data []byte) error { + wsc.writeMu.Lock() + defer wsc.writeMu.Unlock() + + if wsc.closed.Load() { + return fmt.Errorf("connection closed") + } + + return wsc.conn.WriteMessage(messageType, data) +} + +func (wsc *WsConnection) writeNormalizedResponse(resp *common.NormalizedResponse) { + wsc.writeMu.Lock() + defer wsc.writeMu.Unlock() + + if wsc.closed.Load() { + return + } + + // Bound every write under writeMu by the same deadline writeJSON uses. + // Without this, a slow or half-dead client (full TCP buffer, missing + // keepalive, network partition without a clean RST) can hold writeMu + // forever inside NextWriter/WriteTo/Close, which then starves every + // other writer on this connection — including the per-subscription + // runWriters delivering newHead notifications — and presents to + // clients as a stalled subscription. + if err := wsc.conn.SetWriteDeadline(time.Now().Add(wsWriteDeadline)); err != nil { + wsc.logger.Debug().Err(err).Str("connId", wsc.id).Msg("failed to set websocket write deadline") + return + } + defer wsc.conn.SetWriteDeadline(time.Time{}) + + w, err := wsc.conn.NextWriter(websocket.TextMessage) + if err != nil { + wsc.logger.Debug().Err(err).Str("connId", wsc.id).Msg("failed to get websocket writer") + return + } + + _, err = resp.WriteTo(w) + if closeErr := w.Close(); closeErr != nil && err == nil { + err = closeErr + } + if err != nil { + wsc.logger.Debug().Err(err).Str("connId", wsc.id).Msg("failed to write websocket response") + } + + go resp.Release() +} + +func (wsc *WsConnection) writeBatchResponse(responses []interface{}) { + wsc.writeMu.Lock() + defer wsc.writeMu.Unlock() + + if wsc.closed.Load() { + return + } + + // See writeNormalizedResponse — same deadline-or-starve-the-mutex + // reasoning applies to batch responses. + if err := wsc.conn.SetWriteDeadline(time.Now().Add(wsWriteDeadline)); err != nil { + wsc.logger.Debug().Err(err).Str("connId", wsc.id).Msg("failed to set websocket write deadline") + return + } + defer wsc.conn.SetWriteDeadline(time.Time{}) + + w, err := wsc.conn.NextWriter(websocket.TextMessage) + if err != nil { + return + } + + bw := NewBatchResponseWriter(responses) + _, _ = bw.WriteTo(w) + _ = w.Close() +} + +func (wsc *WsConnection) writeErrorResponse(nq *common.NormalizedRequest, origErr error, startedAt *time.Time, includeDetails *bool) { + errBody := processErrorBody(wsc.logger, startedAt, nq, origErr, includeDetails) + var err error + switch v := errBody.(type) { + case *HttpJsonRpcErrorResponse: + err = wsc.writeJSON(v) + default: + err = wsc.writeJSON(errBody) + } + if err != nil { + wsc.logger.Debug().Err(err).Str("connId", wsc.id).Msg("failed to write error response") + } +} + +// +// --- Keepalive --- +// + +func (wsc *WsConnection) pingLoop(interval time.Duration) { + ticker := time.NewTicker(interval) + defer ticker.Stop() + + for { + select { + case <-ticker.C: + // WriteControl uses gorilla's internal control-frame lock (separate + // from our writeMu), so pings don't have to wait for an in-flight + // notification / response to finish. Taking our writeMu here could + // delay the ping (and therefore our liveness signal) behind a slow + // data write — gorilla serializes control frames itself, so we can + // safely skip our mutex for this write. + err := wsc.conn.WriteControl(websocket.PingMessage, nil, time.Now().Add(wsWriteDeadline)) + wsc.logger.Trace().Str("connId", wsc.id).Msg("websocket ping sent") + if err != nil { + wsc.logger.Info().Err(err).Str("connId", wsc.id). + Str("errType", fmt.Sprintf("%T", err)). + Msg("websocket ping failed, closing connection") + wsc.cancel() + return + } + case <-wsc.appCtx.Done(): + return + } + } +} + +// +// --- Connection lifecycle --- +// + +// Close cleans up the WebSocket connection and all associated subscriptions. +func (wsc *WsConnection) Close() { + wsc.closeWithCode(websocket.CloseNormalClosure, "") +} + +// CloseWithGoingAway closes the connection with a GoingAway status code, +// indicating the server is shutting down. +func (wsc *WsConnection) CloseWithGoingAway() { + wsc.closeWithCode(websocket.CloseGoingAway, "server shutting down") +} + +func (wsc *WsConnection) closeWithCode(code int, reason string) { + if wsc.closed.Swap(true) { + return // already closed + } + + wsc.cancel() + + if wsc.subscriptionManager != nil && wsc.project != nil { + wsc.subscriptionManager.CleanupConnection(wsc, wsc.project) + } + + _ = wsc.conn.WriteControl( + websocket.CloseMessage, + websocket.FormatCloseMessage(code, reason), + time.Now().Add(unsubscribeTimeout), + ) + _ = wsc.conn.Close() + + wsc.logger.Info().Str("connId", wsc.id).Int("closeCode", code).Msg("websocket connection closed") +} + +// WriteSubscriptionNotification sends a subscription notification to the client. +// Used by the subscription manager to route upstream events to clients. +func (wsc *WsConnection) WriteSubscriptionNotification(clientSubId string, result json.RawMessage) error { + notification := map[string]interface{}{ + "jsonrpc": "2.0", + "method": "eth_subscription", + "params": map[string]interface{}{ + "subscription": clientSubId, + "result": result, + }, + } + return wsc.writeJSON(notification) +} diff --git a/erpc/ws_server_test.go b/erpc/ws_server_test.go new file mode 100644 index 000000000..8d732915e --- /dev/null +++ b/erpc/ws_server_test.go @@ -0,0 +1,1687 @@ +package erpc + +import ( + "context" + "encoding/json" + "fmt" + "net" + "net/http" + "net/http/httptest" + "strings" + "sync" + "sync/atomic" + "testing" + "time" + + "github.com/erpc/erpc/common" + "github.com/erpc/erpc/data" + "github.com/erpc/erpc/util" + "github.com/gorilla/websocket" + "github.com/h2non/gock" + "github.com/rs/zerolog/log" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// +// --- Test helpers --- +// + +func durationPtr(d time.Duration) *common.Duration { + v := common.Duration(d) + return &v +} + +func init() { + util.ConfigureTestLogger() +} + +// mockWsUpstream creates a test HTTP server that upgrades to WebSocket +// and delegates all message handling to the provided callback. +func mockWsUpstream(t *testing.T, handler func(conn *websocket.Conn)) *httptest.Server { + t.Helper() + upgrader := websocket.Upgrader{CheckOrigin: func(r *http.Request) bool { return true }} + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + c, err := upgrader.Upgrade(w, r, nil) + if err != nil { + t.Logf("mock ws upstream upgrade error: %v", err) + return + } + defer c.Close() + handler(c) + })) + return srv +} + +// standardWsConfig returns a config with both an HTTP upstream (gock) and a WS upstream. +func standardWsConfig(wsURL string) *common.Config { + return &common.Config{ + Server: &common.ServerConfig{ + ListenV4: util.BoolPtr(true), + }, + Projects: []*common.ProjectConfig{ + { + Id: "test_ws", + Networks: []*common.NetworkConfig{ + { + Architecture: common.ArchitectureEvm, + Evm: &common.EvmNetworkConfig{ChainId: 123}, + }, + }, + Upstreams: []*common.UpstreamConfig{ + { + Id: "http-upstream", + Type: common.UpstreamTypeEvm, + Endpoint: "http://rpc1.localhost", + Evm: &common.EvmUpstreamConfig{ChainId: 123}, + }, + { + Id: "ws-upstream", + Type: common.UpstreamTypeEvm, + Endpoint: wsURL, + Evm: &common.EvmUpstreamConfig{ChainId: 123}, + }, + }, + }, + }, + RateLimiters: &common.RateLimiterConfig{}, + } +} + +// multiWsConfig returns a config with one HTTP upstream (gock) and N WS upstreams. +func multiWsConfig(wsURLs ...string) *common.Config { + upstreams := []*common.UpstreamConfig{ + { + Id: "http-upstream", + Type: common.UpstreamTypeEvm, + Endpoint: "http://rpc1.localhost", + Evm: &common.EvmUpstreamConfig{ChainId: 123}, + }, + } + for i, wsURL := range wsURLs { + upstreams = append(upstreams, &common.UpstreamConfig{ + Id: fmt.Sprintf("ws-upstream-%d", i), + Type: common.UpstreamTypeEvm, + Endpoint: wsURL, + Evm: &common.EvmUpstreamConfig{ChainId: 123}, + }) + } + return &common.Config{ + Server: &common.ServerConfig{ListenV4: util.BoolPtr(true)}, + Projects: []*common.ProjectConfig{ + { + Id: "test_ws", + Networks: []*common.NetworkConfig{ + { + Architecture: common.ArchitectureEvm, + Evm: &common.EvmNetworkConfig{ChainId: 123}, + }, + }, + Upstreams: upstreams, + }, + }, + RateLimiters: &common.RateLimiterConfig{}, + } +} + +// httpOnlyConfig returns a config with only HTTP upstreams (no WS). +func httpOnlyConfig() *common.Config { + return &common.Config{ + Server: &common.ServerConfig{ + ListenV4: util.BoolPtr(true), + }, + Projects: []*common.ProjectConfig{ + { + Id: "test_ws", + Networks: []*common.NetworkConfig{ + { + Architecture: common.ArchitectureEvm, + Evm: &common.EvmNetworkConfig{ChainId: 123}, + }, + }, + Upstreams: []*common.UpstreamConfig{ + { + Type: common.UpstreamTypeEvm, + Endpoint: "http://rpc1.localhost", + Evm: &common.EvmUpstreamConfig{ChainId: 123}, + }, + }, + }, + }, + RateLimiters: &common.RateLimiterConfig{}, + } +} + +// setupGock sets up standard gock mocks (eth_getBalance) and EVM state poller stubs. +func setupGock() { + util.ResetGock() + gock.EnableNetworking() + gock.NetworkingFilter(func(req *http.Request) bool { + shouldMakeRealCall := strings.Split(req.URL.Host, ":")[0] == "127.0.0.1" + return shouldMakeRealCall + }) + util.SetupMocksForEvmStatePoller() + + gock.New("http://rpc1.localhost"). + Post("/"). + Persist(). + Filter(func(request *http.Request) bool { + body := util.SafeReadBody(request) + return strings.Contains(body, "eth_getBalance") + }). + Reply(200). + JSON(map[string]interface{}{ + "jsonrpc": "2.0", + "id": 1, + "result": "0xabc123", + }) +} + +// dialWs connects to the eRPC WebSocket endpoint for the test project. +func dialWs(t *testing.T, addr string) *websocket.Conn { + t.Helper() + wsURL := fmt.Sprintf("ws://%s/test_ws/evm/123", addr) + conn, _, err := websocket.DefaultDialer.Dial(wsURL, nil) + require.NoError(t, err, "WebSocket dial should succeed") + return conn +} + +// sendAndReceive sends a JSON-RPC request string and reads the JSON response. +func sendAndReceive(t *testing.T, conn *websocket.Conn, req string) map[string]interface{} { + t.Helper() + err := conn.WriteMessage(websocket.TextMessage, []byte(req)) + require.NoError(t, err) + conn.SetReadDeadline(time.Now().Add(5 * time.Second)) + _, msg, err := conn.ReadMessage() + require.NoError(t, err) + var resp map[string]interface{} + require.NoError(t, json.Unmarshal(msg, &resp)) + return resp +} + +// setupTestERPCServer boots an eRPC instance with the given config, returns +// the listen address and a cleanup function that shuts everything down. +func setupTestERPCServer(t *testing.T, cfg *common.Config) (string, context.CancelFunc) { + t.Helper() + + logger := log.Logger + ctx, cancel := context.WithCancel(context.Background()) + + err := cfg.SetDefaults(&common.DefaultOptions{}) + require.NoError(t, err) + + ssr, err := data.NewSharedStateRegistry(ctx, &logger, &common.SharedStateConfig{ + Connector: &common.ConnectorConfig{ + Driver: "memory", + Memory: &common.MemoryConnectorConfig{ + MaxItems: 100_000, MaxTotalSize: "1GB", + }, + }, + }) + require.NoError(t, err) + + erpcInstance, err := NewERPC(ctx, &logger, ssr, nil, cfg) + require.NoError(t, err) + + erpcInstance.Bootstrap(ctx) + + httpServer, err := NewHttpServer(ctx, &logger, cfg.Server, cfg.HealthCheck, cfg.Admin, cfg.Indexer, erpcInstance) + require.NoError(t, err) + + listener, err := net.Listen("tcp", "127.0.0.1:0") + require.NoError(t, err) + port := listener.Addr().(*net.TCPAddr).Port + + go func() { + err := httpServer.serverV4.Serve(listener) + if err != nil && err != http.ErrServerClosed { + t.Errorf("Server error: %v", err) + } + }() + + time.Sleep(500 * time.Millisecond) + + baseURL := fmt.Sprintf("127.0.0.1:%d", port) + + cleanup := func() { + _ = httpServer.Shutdown(&logger) + cancel() + } + + return baseURL, cleanup +} + +// standardMockWsHandler handles the common set of state poller methods +// (eth_chainId, eth_getBlockByNumber, eth_syncing) that the upstream +// must respond to before the WS client is considered ready. +func standardMockWsHandler(conn *websocket.Conn, customHandler func(method string, id interface{}, req map[string]interface{})) { + for { + _, msg, err := conn.ReadMessage() + if err != nil { + return + } + var req map[string]interface{} + if err := json.Unmarshal(msg, &req); err != nil { + continue + } + method, _ := req["method"].(string) + id := req["id"] + + switch method { + case "eth_chainId": + conn.WriteJSON(map[string]interface{}{"jsonrpc": "2.0", "id": id, "result": "0x7b"}) + case "eth_getBlockByNumber": + conn.WriteJSON(map[string]interface{}{"jsonrpc": "2.0", "id": id, "result": map[string]interface{}{"number": "0x100", "timestamp": "0x6702a8f0"}}) + case "eth_syncing": + conn.WriteJSON(map[string]interface{}{"jsonrpc": "2.0", "id": id, "result": false}) + default: + if customHandler != nil { + customHandler(method, id, req) + } else { + conn.WriteJSON(map[string]interface{}{"jsonrpc": "2.0", "id": id, "result": "0x1"}) + } + } + } +} + +// +// --- Tests: basic JSON-RPC over WebSocket --- +// + +func TestWebSocket_BasicRPC(t *testing.T) { + // Verifies a single JSON-RPC request/response over a WebSocket connection + t.Run("SingleRequestOverWebSocket", func(t *testing.T) { + setupGock() + defer util.ResetGock() + + addr, cleanup := setupTestERPCServer(t, httpOnlyConfig()) + defer cleanup() + + conn := dialWs(t, addr) + defer conn.Close() + + resp := sendAndReceive(t, conn, `{"jsonrpc":"2.0","id":1,"method":"eth_getBalance","params":["0x1234567890abcdef1234567890abcdef12345678","latest"]}`) + assert.Equal(t, "2.0", resp["jsonrpc"]) + assert.Equal(t, float64(1), resp["id"]) + assert.Equal(t, "0xabc123", resp["result"]) + }) + + // Verifies sequential requests on the same connection work correctly + t.Run("MultipleRequestsOnSameConnection", func(t *testing.T) { + setupGock() + defer util.ResetGock() + + gock.New("http://rpc1.localhost"). + Post("/"). + Persist(). + Filter(func(request *http.Request) bool { + body := util.SafeReadBody(request) + return strings.Contains(body, "eth_chainId") + }). + Reply(200). + JSON(map[string]interface{}{ + "jsonrpc": "2.0", + "id": 2, + "result": "0x7b", + }) + + addr, cleanup := setupTestERPCServer(t, httpOnlyConfig()) + defer cleanup() + + conn := dialWs(t, addr) + defer conn.Close() + + resp1 := sendAndReceive(t, conn, `{"jsonrpc":"2.0","id":1,"method":"eth_getBalance","params":["0xaaaa","latest"]}`) + assert.Equal(t, "0xabc123", resp1["result"]) + + resp2 := sendAndReceive(t, conn, `{"jsonrpc":"2.0","id":2,"method":"eth_chainId","params":[]}`) + assert.Equal(t, "0x7b", resp2["result"]) + }) + + // Verifies many concurrent writes/reads on a single connection + t.Run("ConcurrentRequestsOnSameConnection", func(t *testing.T) { + setupGock() + defer util.ResetGock() + + addr, cleanup := setupTestERPCServer(t, httpOnlyConfig()) + defer cleanup() + + conn := dialWs(t, addr) + defer conn.Close() + + const numRequests = 10 + var writeMu sync.Mutex + + for i := 0; i < numRequests; i++ { + writeMu.Lock() + msg := fmt.Sprintf(`{"jsonrpc":"2.0","id":%d,"method":"eth_getBalance","params":["0xaaaa","latest"]}`, i) + err := conn.WriteMessage(websocket.TextMessage, []byte(msg)) + writeMu.Unlock() + require.NoError(t, err) + } + + responses := make(map[float64]bool) + for i := 0; i < numRequests; i++ { + conn.SetReadDeadline(time.Now().Add(5 * time.Second)) + _, msg, err := conn.ReadMessage() + require.NoError(t, err) + + var resp map[string]interface{} + require.NoError(t, json.Unmarshal(msg, &resp)) + assert.Equal(t, "2.0", resp["jsonrpc"]) + if id, ok := resp["id"].(float64); ok { + responses[id] = true + } + } + + assert.Equal(t, numRequests, len(responses), "should receive all responses") + }) + + // Verifies JSON array batch requests work over WebSocket + t.Run("BatchRequestOverWebSocket", func(t *testing.T) { + setupGock() + defer util.ResetGock() + + gock.New("http://rpc1.localhost"). + Post("/"). + Persist(). + Filter(func(request *http.Request) bool { + body := util.SafeReadBody(request) + return strings.Contains(body, "eth_chainId") + }). + Reply(200). + JSON(map[string]interface{}{ + "jsonrpc": "2.0", + "id": 2, + "result": "0x7b", + }) + + addr, cleanup := setupTestERPCServer(t, httpOnlyConfig()) + defer cleanup() + + conn := dialWs(t, addr) + defer conn.Close() + + batch := `[{"jsonrpc":"2.0","id":1,"method":"eth_getBalance","params":["0xaaaa","latest"]},{"jsonrpc":"2.0","id":2,"method":"eth_chainId","params":[]}]` + err := conn.WriteMessage(websocket.TextMessage, []byte(batch)) + require.NoError(t, err) + + conn.SetReadDeadline(time.Now().Add(5 * time.Second)) + _, respMsg, err := conn.ReadMessage() + require.NoError(t, err) + + var responses []map[string]interface{} + err = json.Unmarshal(respMsg, &responses) + require.NoError(t, err, "Response should be a JSON array") + assert.Equal(t, 2, len(responses), "Batch response should contain 2 items") + }) + + // Verifies the server handles client disconnect and allows reconnection + t.Run("WebSocketClientDisconnect", func(t *testing.T) { + setupGock() + defer util.ResetGock() + + addr, cleanup := setupTestERPCServer(t, httpOnlyConfig()) + defer cleanup() + + conn := dialWs(t, addr) + + resp := sendAndReceive(t, conn, `{"jsonrpc":"2.0","id":1,"method":"eth_getBalance","params":["0xaaaa","latest"]}`) + assert.Contains(t, resp["result"], "0xabc123") + + err := conn.WriteMessage(websocket.CloseMessage, websocket.FormatCloseMessage(websocket.CloseNormalClosure, "")) + assert.NoError(t, err) + conn.Close() + + time.Sleep(200 * time.Millisecond) + conn2 := dialWs(t, addr) + defer conn2.Close() + + resp2 := sendAndReceive(t, conn2, `{"jsonrpc":"2.0","id":1,"method":"eth_getBalance","params":["0xaaaa","latest"]}`) + assert.Contains(t, resp2["result"], "0xabc123") + }) + + // Verifies HTTP and WebSocket work simultaneously on the same server + t.Run("HTTPStillWorksAlongsideWebSocket", func(t *testing.T) { + setupGock() + defer util.ResetGock() + + addr, cleanup := setupTestERPCServer(t, httpOnlyConfig()) + defer cleanup() + + conn := dialWs(t, addr) + defer conn.Close() + + wsResp := sendAndReceive(t, conn, `{"jsonrpc":"2.0","id":1,"method":"eth_getBalance","params":["0xaaaa","latest"]}`) + assert.Equal(t, "0xabc123", wsResp["result"]) + + httpURL := fmt.Sprintf("http://%s/test_ws/evm/123", addr) + cleanClient := &http.Client{Transport: &http.Transport{}} + httpResp, err := cleanClient.Post(httpURL, "application/json", strings.NewReader(`{"jsonrpc":"2.0","id":1,"method":"eth_getBalance","params":["0xaaaa","latest"]}`)) + require.NoError(t, err) + defer httpResp.Body.Close() + + assert.Equal(t, http.StatusOK, httpResp.StatusCode) + }) +} + +// +// --- Tests: error handling --- +// + +func TestWebSocket_ErrorHandling(t *testing.T) { + // Verifies invalid JSON returns a parse error response + t.Run("InvalidJSON", func(t *testing.T) { + setupGock() + defer util.ResetGock() + + addr, cleanup := setupTestERPCServer(t, httpOnlyConfig()) + defer cleanup() + + conn := dialWs(t, addr) + defer conn.Close() + + err := conn.WriteMessage(websocket.TextMessage, []byte(`{not valid json`)) + require.NoError(t, err) + + conn.SetReadDeadline(time.Now().Add(5 * time.Second)) + _, msg, err := conn.ReadMessage() + require.NoError(t, err) + + var resp map[string]interface{} + require.NoError(t, json.Unmarshal(msg, &resp)) + assert.NotNil(t, resp["error"], "should return error for invalid JSON") + }) + + // Verifies an empty message body returns an error response + t.Run("EmptyBody", func(t *testing.T) { + setupGock() + defer util.ResetGock() + + addr, cleanup := setupTestERPCServer(t, httpOnlyConfig()) + defer cleanup() + + conn := dialWs(t, addr) + defer conn.Close() + + err := conn.WriteMessage(websocket.TextMessage, []byte(``)) + require.NoError(t, err) + + conn.SetReadDeadline(time.Now().Add(5 * time.Second)) + _, msg, err := conn.ReadMessage() + require.NoError(t, err) + + var resp map[string]interface{} + require.NoError(t, json.Unmarshal(msg, &resp)) + assert.NotNil(t, resp["error"], "should return error for empty body") + }) + + // Verifies upstream JSON-RPC errors are forwarded to the client + t.Run("UpstreamError", func(t *testing.T) { + util.ResetGock() + gock.EnableNetworking() + gock.NetworkingFilter(func(req *http.Request) bool { + return strings.Split(req.URL.Host, ":")[0] == "127.0.0.1" + }) + util.SetupMocksForEvmStatePoller() + defer util.ResetGock() + + gock.New("http://rpc1.localhost"). + Post("/"). + Persist(). + Filter(func(request *http.Request) bool { + body := util.SafeReadBody(request) + return strings.Contains(body, "eth_getBalance") + }). + Reply(200). + JSON(map[string]interface{}{ + "jsonrpc": "2.0", + "id": 1, + "error": map[string]interface{}{ + "code": -32000, + "message": "execution reverted", + }, + }) + + addr, cleanup := setupTestERPCServer(t, httpOnlyConfig()) + defer cleanup() + + conn := dialWs(t, addr) + defer conn.Close() + + resp := sendAndReceive(t, conn, `{"jsonrpc":"2.0","id":1,"method":"eth_getBalance","params":["0xaaaa","latest"]}`) + assert.NotNil(t, resp["error"], "should forward upstream error") + }) +} + +// +// --- Tests: multiple independent connections --- +// + +func TestWebSocket_MultipleConnections(t *testing.T) { + // Verifies multiple simultaneous connections each get correct responses + t.Run("IndependentConnections", func(t *testing.T) { + setupGock() + defer util.ResetGock() + + addr, cleanup := setupTestERPCServer(t, httpOnlyConfig()) + defer cleanup() + + conn1 := dialWs(t, addr) + defer conn1.Close() + conn2 := dialWs(t, addr) + defer conn2.Close() + conn3 := dialWs(t, addr) + defer conn3.Close() + + resp1 := sendAndReceive(t, conn1, `{"jsonrpc":"2.0","id":1,"method":"eth_getBalance","params":["0xaaaa","latest"]}`) + resp2 := sendAndReceive(t, conn2, `{"jsonrpc":"2.0","id":2,"method":"eth_getBalance","params":["0xaaaa","latest"]}`) + resp3 := sendAndReceive(t, conn3, `{"jsonrpc":"2.0","id":3,"method":"eth_getBalance","params":["0xaaaa","latest"]}`) + + assert.Equal(t, float64(1), resp1["id"]) + assert.Equal(t, float64(2), resp2["id"]) + assert.Equal(t, float64(3), resp3["id"]) + assert.Equal(t, "0xabc123", resp1["result"]) + assert.Equal(t, "0xabc123", resp2["result"]) + assert.Equal(t, "0xabc123", resp3["result"]) + }) + + // Verifies closing one connection does not affect others + t.Run("OneDisconnectDoesNotAffectOthers", func(t *testing.T) { + setupGock() + defer util.ResetGock() + + addr, cleanup := setupTestERPCServer(t, httpOnlyConfig()) + defer cleanup() + + conn1 := dialWs(t, addr) + conn2 := dialWs(t, addr) + defer conn2.Close() + + resp1 := sendAndReceive(t, conn1, `{"jsonrpc":"2.0","id":1,"method":"eth_getBalance","params":["0xaaaa","latest"]}`) + assert.Equal(t, "0xabc123", resp1["result"]) + + conn1.WriteMessage(websocket.CloseMessage, websocket.FormatCloseMessage(websocket.CloseNormalClosure, "")) + conn1.Close() + time.Sleep(100 * time.Millisecond) + + resp2 := sendAndReceive(t, conn2, `{"jsonrpc":"2.0","id":2,"method":"eth_getBalance","params":["0xaaaa","latest"]}`) + assert.Equal(t, "0xabc123", resp2["result"]) + }) +} + +// +// --- Tests: subscriptions (eth_subscribe / eth_unsubscribe) --- +// + +func TestWebSocket_Subscriptions(t *testing.T) { + // Verifies the full subscribe -> receive notification -> unsubscribe lifecycle + t.Run("SubscribeReceiveUnsubscribe", func(t *testing.T) { + notifCh := make(chan struct{}, 1) + mockUpstream := mockWsUpstream(t, func(conn *websocket.Conn) { + standardMockWsHandler(conn, func(method string, id interface{}, req map[string]interface{}) { + switch method { + case "eth_subscribe": + subId := "0xdeadbeef12345678" + conn.WriteJSON(map[string]interface{}{"jsonrpc": "2.0", "id": id, "result": subId}) + + go func() { + time.Sleep(200 * time.Millisecond) + conn.WriteJSON(map[string]interface{}{ + "jsonrpc": "2.0", + "method": "eth_subscription", + "params": map[string]interface{}{ + "subscription": subId, + "result": map[string]interface{}{"number": "0x101", "hash": "0xaaa"}, + }, + }) + select { + case notifCh <- struct{}{}: + default: + } + }() + case "eth_unsubscribe": + conn.WriteJSON(map[string]interface{}{"jsonrpc": "2.0", "id": id, "result": true}) + default: + conn.WriteJSON(map[string]interface{}{"jsonrpc": "2.0", "id": id, "result": "0x1"}) + } + }) + }) + defer mockUpstream.Close() + + setupGock() + defer util.ResetGock() + + wsUpstreamURL := "ws" + strings.TrimPrefix(mockUpstream.URL, "http") + addr, cleanup := setupTestERPCServer(t, standardWsConfig(wsUpstreamURL)) + defer cleanup() + + time.Sleep(2 * time.Second) + + conn := dialWs(t, addr) + defer conn.Close() + + // Subscribe + resp := sendAndReceive(t, conn, `{"jsonrpc":"2.0","id":1,"method":"eth_subscribe","params":["newHeads"]}`) + assert.NotNil(t, resp["result"], "should return subscription ID") + assert.Nil(t, resp["error"], "should not have error") + clientSubId, ok := resp["result"].(string) + require.True(t, ok, "subscription ID should be a string") + assert.True(t, strings.HasPrefix(clientSubId, "0x"), "subscription ID should start with 0x") + + // Wait for notification from upstream + select { + case <-notifCh: + case <-time.After(5 * time.Second): + t.Fatal("timeout waiting for upstream to send notification") + } + + // Read the notification delivered to the client + conn.SetReadDeadline(time.Now().Add(5 * time.Second)) + _, notifMsg, err := conn.ReadMessage() + require.NoError(t, err, "should receive notification") + + var notif map[string]interface{} + require.NoError(t, json.Unmarshal(notifMsg, ¬if)) + assert.Equal(t, "eth_subscription", notif["method"]) + params, ok := notif["params"].(map[string]interface{}) + require.True(t, ok) + assert.Equal(t, clientSubId, params["subscription"], "notification should use client subscription ID") + + // Unsubscribe + unsubResp := sendAndReceive(t, conn, fmt.Sprintf(`{"jsonrpc":"2.0","id":2,"method":"eth_unsubscribe","params":["%s"]}`, clientSubId)) + assert.Equal(t, true, unsubResp["result"]) + }) + + // Verifies eth_subscribe returns an error when no WS upstream is configured + t.Run("SubscribeWithNoWsUpstreamFails", func(t *testing.T) { + setupGock() + defer util.ResetGock() + + addr, cleanup := setupTestERPCServer(t, httpOnlyConfig()) + defer cleanup() + + conn := dialWs(t, addr) + defer conn.Close() + + resp := sendAndReceive(t, conn, `{"jsonrpc":"2.0","id":1,"method":"eth_subscribe","params":["newHeads"]}`) + assert.NotNil(t, resp["error"], "should return error when no WS upstream available") + }) + + // Verifies eth_unsubscribe with a nonexistent ID returns an error + t.Run("UnsubscribeUnknownIDFails", func(t *testing.T) { + setupGock() + defer util.ResetGock() + + addr, cleanup := setupTestERPCServer(t, httpOnlyConfig()) + defer cleanup() + + conn := dialWs(t, addr) + defer conn.Close() + + resp := sendAndReceive(t, conn, `{"jsonrpc":"2.0","id":1,"method":"eth_unsubscribe","params":["0xnonexistent"]}`) + assert.NotNil(t, resp["error"], "should return error for unknown subscription ID") + }) + + // Verifies client disconnect cleanly removes the client from the fan-out + // group without disturbing the persistent upstream newHeads subscription. + t.Run("SubscriptionCleanupOnDisconnect", func(t *testing.T) { + mockUpstream := mockWsUpstream(t, func(conn *websocket.Conn) { + standardMockWsHandler(conn, func(method string, id interface{}, req map[string]interface{}) { + switch method { + case "eth_subscribe": + conn.WriteJSON(map[string]interface{}{"jsonrpc": "2.0", "id": id, "result": "0xsub123"}) + default: + conn.WriteJSON(map[string]interface{}{"jsonrpc": "2.0", "id": id, "result": "0x1"}) + } + }) + }) + defer mockUpstream.Close() + + setupGock() + defer util.ResetGock() + + wsUpstreamURL := "ws" + strings.TrimPrefix(mockUpstream.URL, "http") + addr, cleanup := setupTestERPCServer(t, standardWsConfig(wsUpstreamURL)) + defer cleanup() + + time.Sleep(2 * time.Second) + + conn := dialWs(t, addr) + + resp := sendAndReceive(t, conn, `{"jsonrpc":"2.0","id":1,"method":"eth_subscribe","params":["newHeads"]}`) + require.NotNil(t, resp["result"]) + + // Disconnect without unsubscribing — newHeads upstream sub stays active + conn.Close() + + // Allow cleanup to complete + time.Sleep(500 * time.Millisecond) + + // Verify a new client can still subscribe (upstream sub still alive) + conn2 := dialWs(t, addr) + defer conn2.Close() + + resp2 := sendAndReceive(t, conn2, `{"jsonrpc":"2.0","id":1,"method":"eth_subscribe","params":["newHeads"]}`) + assert.NotNil(t, resp2["result"], "new client should be able to subscribe after previous client disconnected") + }) +} + +// +// --- Tests: WebSocket upstream client --- +// + +func TestWebSocket_UpstreamClient(t *testing.T) { + // Verifies regular RPC requests can be routed through a WS upstream + t.Run("RPCThroughWsUpstream", func(t *testing.T) { + mockUpstream := mockWsUpstream(t, func(conn *websocket.Conn) { + standardMockWsHandler(conn, func(method string, id interface{}, req map[string]interface{}) { + switch method { + case "eth_getBalance": + conn.WriteJSON(map[string]interface{}{"jsonrpc": "2.0", "id": id, "result": "0xws_upstream_balance"}) + default: + conn.WriteJSON(map[string]interface{}{"jsonrpc": "2.0", "id": id, "result": "0x1"}) + } + }) + }) + defer mockUpstream.Close() + + setupGock() + defer util.ResetGock() + + wsUpstreamURL := "ws" + strings.TrimPrefix(mockUpstream.URL, "http") + + // Config with ONLY the WS upstream to ensure requests go through WS + cfg := &common.Config{ + Server: &common.ServerConfig{ + ListenV4: util.BoolPtr(true), + }, + Projects: []*common.ProjectConfig{ + { + Id: "test_ws", + Networks: []*common.NetworkConfig{ + { + Architecture: common.ArchitectureEvm, + Evm: &common.EvmNetworkConfig{ChainId: 123}, + }, + }, + Upstreams: []*common.UpstreamConfig{ + { + Id: "ws-only", + Type: common.UpstreamTypeEvm, + Endpoint: wsUpstreamURL, + Evm: &common.EvmUpstreamConfig{ChainId: 123}, + }, + }, + }, + }, + RateLimiters: &common.RateLimiterConfig{}, + } + + addr, cleanup := setupTestERPCServer(t, cfg) + defer cleanup() + + time.Sleep(2 * time.Second) + + httpURL := fmt.Sprintf("http://%s/test_ws/evm/123", addr) + cleanClient := &http.Client{Transport: &http.Transport{}} + httpResp, err := cleanClient.Post(httpURL, "application/json", + strings.NewReader(`{"jsonrpc":"2.0","id":1,"method":"eth_getBalance","params":["0xaaaa","latest"]}`)) + require.NoError(t, err) + defer httpResp.Body.Close() + + var resp map[string]interface{} + require.NoError(t, json.NewDecoder(httpResp.Body).Decode(&resp)) + assert.Equal(t, "0xws_upstream_balance", resp["result"], "response should come from WS upstream") + }) + + // Verifies the WS upstream client automatically reconnects after disconnect + t.Run("WsUpstreamReconnects", func(t *testing.T) { + connCount := 0 + var connMu sync.Mutex + + mockUpstream := mockWsUpstream(t, func(conn *websocket.Conn) { + connMu.Lock() + connCount++ + count := connCount + connMu.Unlock() + + if count == 1 { + // First connection: accept one request then close abruptly + _, msg, err := conn.ReadMessage() + if err != nil { + return + } + var req map[string]interface{} + json.Unmarshal(msg, &req) + id := req["id"] + method, _ := req["method"].(string) + if method == "eth_chainId" { + conn.WriteJSON(map[string]interface{}{"jsonrpc": "2.0", "id": id, "result": "0x7b"}) + } + time.Sleep(100 * time.Millisecond) + conn.Close() + return + } + + // Subsequent connections: handle normally + standardMockWsHandler(conn, func(method string, id interface{}, req map[string]interface{}) { + switch method { + case "eth_getBalance": + conn.WriteJSON(map[string]interface{}{"jsonrpc": "2.0", "id": id, "result": "0xreconnected"}) + default: + conn.WriteJSON(map[string]interface{}{"jsonrpc": "2.0", "id": id, "result": "0x1"}) + } + }) + }) + defer mockUpstream.Close() + + setupGock() + defer util.ResetGock() + + wsUpstreamURL := "ws" + strings.TrimPrefix(mockUpstream.URL, "http") + addr, cleanup := setupTestERPCServer(t, standardWsConfig(wsUpstreamURL)) + defer cleanup() + + time.Sleep(5 * time.Second) + + connMu.Lock() + assert.GreaterOrEqual(t, connCount, 2, "WS upstream should have reconnected") + connMu.Unlock() + + conn := dialWs(t, addr) + defer conn.Close() + + resp := sendAndReceive(t, conn, `{"jsonrpc":"2.0","id":1,"method":"eth_getBalance","params":["0xaaaa","latest"]}`) + assert.NotNil(t, resp["result"], "should get a response after upstream reconnect") + }) +} + +// +// --- Tests: subscription recovery after upstream reconnect --- +// + +func TestWebSocket_SubscriptionRecovery(t *testing.T) { + // Verifies that when the upstream WS connection drops, eRPC closes the + // client connection with CloseGoingAway (1001) so the client can reconnect + // and re-subscribe cleanly instead of holding a zombie subscription. + t.Run("ClientDisconnectedOnUpstreamDrop", func(t *testing.T) { + closeUpstream := make(chan struct{}) + + mockUpstream := mockWsUpstream(t, func(conn *websocket.Conn) { + for { + _, msg, err := conn.ReadMessage() + if err != nil { + return + } + var req map[string]interface{} + json.Unmarshal(msg, &req) + method, _ := req["method"].(string) + id := req["id"] + + switch method { + case "eth_chainId": + conn.WriteJSON(map[string]interface{}{"jsonrpc": "2.0", "id": id, "result": "0x7b"}) + case "eth_getBlockByNumber": + conn.WriteJSON(map[string]interface{}{"jsonrpc": "2.0", "id": id, "result": map[string]interface{}{"number": "0x1"}}) + case "eth_syncing": + conn.WriteJSON(map[string]interface{}{"jsonrpc": "2.0", "id": id, "result": false}) + case "eth_subscribe": + conn.WriteJSON(map[string]interface{}{"jsonrpc": "2.0", "id": id, "result": "0xsub123"}) + // Wait for signal then kill the connection + go func() { + <-closeUpstream + conn.Close() + }() + default: + conn.WriteJSON(map[string]interface{}{"jsonrpc": "2.0", "id": id, "result": "0x1"}) + } + } + }) + defer mockUpstream.Close() + + setupGock() + defer util.ResetGock() + + wsUpstreamURL := "ws" + strings.TrimPrefix(mockUpstream.URL, "http") + addr, cleanup := setupTestERPCServer(t, standardWsConfig(wsUpstreamURL)) + defer cleanup() + + time.Sleep(2 * time.Second) + + conn := dialWs(t, addr) + defer conn.Close() + + // Subscribe successfully + resp := sendAndReceive(t, conn, `{"jsonrpc":"2.0","id":1,"method":"eth_subscribe","params":["newHeads"]}`) + require.NotNil(t, resp["result"], "should get a subscription ID") + + // Kill the upstream WS connection + close(closeUpstream) + + // Client should receive a close frame with GoingAway (1001) + conn.SetReadDeadline(time.Now().Add(10 * time.Second)) + _, _, err := conn.ReadMessage() + require.Error(t, err, "client should be disconnected") + closeErr, ok := err.(*websocket.CloseError) + if ok { + assert.Equal(t, websocket.CloseGoingAway, closeErr.Code, "close code should be 1001 GoingAway") + t.Logf("client received close frame: code=%d reason=%q", closeErr.Code, closeErr.Text) + } else { + t.Logf("client disconnected with error: %v", err) + } + }) + + // Verifies that eth_subscribe returns an error when the upstream WS + // connection isn't established yet (instead of creating a zombie subscription). + t.Run("SubscribeFailsWhenUpstreamDisconnected", func(t *testing.T) { + // Create a mock that immediately closes the WS connection, + // so eRPC's upstream WS stays disconnected. + mockUpstream := mockWsUpstream(t, func(conn *websocket.Conn) { + conn.Close() + }) + defer mockUpstream.Close() + + setupGock() + defer util.ResetGock() + + wsUpstreamURL := "ws" + strings.TrimPrefix(mockUpstream.URL, "http") + addr, cleanup := setupTestERPCServer(t, standardWsConfig(wsUpstreamURL)) + defer cleanup() + + time.Sleep(2 * time.Second) + + conn := dialWs(t, addr) + defer conn.Close() + + resp := sendAndReceive(t, conn, `{"jsonrpc":"2.0","id":1,"method":"eth_subscribe","params":["newHeads"]}`) + assert.NotNil(t, resp["error"], "should return error when upstream WS is not connected") + t.Logf("got expected error: %v", resp["error"]) + }) +} + +// +// --- Tests: subscription deduplication --- +// + +func TestWebSocket_SubscriptionDedup(t *testing.T) { + // Verifies two clients subscribing to the same event share one upstream subscription + t.Run("TwoClientsShareOneUpstreamSubscription", func(t *testing.T) { + subscribeCount := 0 + var subMu sync.Mutex + upstreamSubId := "0xsharedsub123" + + mockUpstream := mockWsUpstream(t, func(conn *websocket.Conn) { + standardMockWsHandler(conn, func(method string, id interface{}, req map[string]interface{}) { + switch method { + case "eth_subscribe": + subMu.Lock() + subscribeCount++ + count := subscribeCount + subMu.Unlock() + conn.WriteJSON(map[string]interface{}{"jsonrpc": "2.0", "id": id, "result": upstreamSubId}) + + // Send a notification only on first subscribe to avoid duplicates + if count == 1 { + go func() { + time.Sleep(500 * time.Millisecond) + conn.WriteJSON(map[string]interface{}{ + "jsonrpc": "2.0", + "method": "eth_subscription", + "params": map[string]interface{}{ + "subscription": upstreamSubId, + "result": map[string]interface{}{"number": "0x999"}, + }, + }) + }() + } + case "eth_unsubscribe": + conn.WriteJSON(map[string]interface{}{"jsonrpc": "2.0", "id": id, "result": true}) + default: + conn.WriteJSON(map[string]interface{}{"jsonrpc": "2.0", "id": id, "result": "0x1"}) + } + }) + }) + defer mockUpstream.Close() + + setupGock() + defer util.ResetGock() + + wsUpstreamURL := "ws" + strings.TrimPrefix(mockUpstream.URL, "http") + addr, cleanup := setupTestERPCServer(t, standardWsConfig(wsUpstreamURL)) + defer cleanup() + time.Sleep(2 * time.Second) + + // Client 1 subscribes + conn1 := dialWs(t, addr) + defer conn1.Close() + resp1 := sendAndReceive(t, conn1, `{"jsonrpc":"2.0","id":1,"method":"eth_subscribe","params":["newHeads"]}`) + require.NotNil(t, resp1["result"]) + clientSubId1 := resp1["result"].(string) + + // Client 2 subscribes to the same event + conn2 := dialWs(t, addr) + defer conn2.Close() + resp2 := sendAndReceive(t, conn2, `{"jsonrpc":"2.0","id":1,"method":"eth_subscribe","params":["newHeads"]}`) + require.NotNil(t, resp2["result"]) + clientSubId2 := resp2["result"].(string) + + assert.NotEqual(t, clientSubId1, clientSubId2, "each client should get a unique subscription ID") + + // Only ONE eth_subscribe should have been sent to the upstream (dedup) + subMu.Lock() + assert.Equal(t, 1, subscribeCount, "upstream should only receive one eth_subscribe (dedup)") + subMu.Unlock() + + // Both clients should receive the notification + conn1.SetReadDeadline(time.Now().Add(5 * time.Second)) + _, notif1, err := conn1.ReadMessage() + require.NoError(t, err) + + conn2.SetReadDeadline(time.Now().Add(5 * time.Second)) + _, notif2, err := conn2.ReadMessage() + require.NoError(t, err) + + var n1, n2 map[string]interface{} + require.NoError(t, json.Unmarshal(notif1, &n1)) + require.NoError(t, json.Unmarshal(notif2, &n2)) + + p1 := n1["params"].(map[string]interface{}) + p2 := n2["params"].(map[string]interface{}) + assert.Equal(t, clientSubId1, p1["subscription"]) + assert.Equal(t, clientSubId2, p2["subscription"]) + }) +} + +// +// --- Tests: rate limiting --- +// + +func TestWebSocket_RateLimiting(t *testing.T) { + // Verifies project-level rate limits are enforced on WebSocket requests + t.Run("ProjectRateLimitAppliedOverWs", func(t *testing.T) { + setupGock() + defer util.ResetGock() + + cfg := &common.Config{ + Server: &common.ServerConfig{ + ListenV4: util.BoolPtr(true), + }, + Projects: []*common.ProjectConfig{ + { + Id: "test_ws", + RateLimitBudget: "ws-test-budget", + Networks: []*common.NetworkConfig{ + { + Architecture: common.ArchitectureEvm, + Evm: &common.EvmNetworkConfig{ChainId: 123}, + }, + }, + Upstreams: []*common.UpstreamConfig{ + { + Type: common.UpstreamTypeEvm, + Endpoint: "http://rpc1.localhost", + Evm: &common.EvmUpstreamConfig{ChainId: 123}, + }, + }, + }, + }, + RateLimiters: &common.RateLimiterConfig{ + Store: &common.RateLimitStoreConfig{ + Driver: "memory", + }, + Budgets: []*common.RateLimitBudgetConfig{ + { + Id: "ws-test-budget", + Rules: []*common.RateLimitRuleConfig{ + { + Method: "*", + MaxCount: 3, + Period: common.RateLimitPeriodMinute, + }, + }, + }, + }, + }, + } + + addr, cleanup := setupTestERPCServer(t, cfg) + defer cleanup() + + conn := dialWs(t, addr) + defer conn.Close() + + // Align to start of the next minute to avoid rate limit window rollover + now := time.Now() + time.Sleep(time.Until(now.Truncate(time.Minute).Add(time.Minute))) + + var lastResp map[string]interface{} + rateLimited := false + for i := 0; i < 10; i++ { + msg := fmt.Sprintf(`{"jsonrpc":"2.0","id":%d,"method":"eth_getBalance","params":["0xaaaa","latest"]}`, i) + err := conn.WriteMessage(websocket.TextMessage, []byte(msg)) + require.NoError(t, err) + + conn.SetReadDeadline(time.Now().Add(5 * time.Second)) + _, respBytes, err := conn.ReadMessage() + require.NoError(t, err) + + require.NoError(t, json.Unmarshal(respBytes, &lastResp)) + if lastResp["error"] != nil { + errStr, _ := json.Marshal(lastResp["error"]) + if strings.Contains(string(errStr), "RateLimitRuleExceeded") || + strings.Contains(string(errStr), "rate limit") || + strings.Contains(string(errStr), "ErrProjectRateLimitRuleExceeded") { + rateLimited = true + break + } + } + } + + assert.True(t, rateLimited, "should hit rate limit when sending requests over WS") + }) +} + +// +// --- Tests: graceful shutdown --- +// + +func TestWebSocket_GracefulShutdown(t *testing.T) { + // Verifies clients receive a GoingAway close frame on server shutdown + t.Run("ServerShutdownClosesWsWithGoingAway", func(t *testing.T) { + setupGock() + defer util.ResetGock() + + addr, cleanup := setupTestERPCServer(t, httpOnlyConfig()) + + conn := dialWs(t, addr) + defer conn.Close() + + resp := sendAndReceive(t, conn, `{"jsonrpc":"2.0","id":1,"method":"eth_getBalance","params":["0xaaaa","latest"]}`) + assert.Equal(t, "0xabc123", resp["result"]) + + cleanup() + + conn.SetReadDeadline(time.Now().Add(5 * time.Second)) + _, _, err := conn.ReadMessage() + require.Error(t, err) + + closeErr, ok := err.(*websocket.CloseError) + if ok { + assert.Equal(t, websocket.CloseGoingAway, closeErr.Code, + "should receive CloseGoingAway (1001) on server shutdown") + } + }) + + // Verifies the server unsubscribes from upstreams during shutdown + // Verifies server shutdown closes the upstream WS connection (which implicitly + // terminates all upstream subscriptions) and closes all client connections. + t.Run("ServerShutdownCleansUpSubscriptions", func(t *testing.T) { + mockUpstream := mockWsUpstream(t, func(conn *websocket.Conn) { + standardMockWsHandler(conn, func(method string, id interface{}, req map[string]interface{}) { + switch method { + case "eth_subscribe": + conn.WriteJSON(map[string]interface{}{"jsonrpc": "2.0", "id": id, "result": "0xshutdownsub"}) + default: + conn.WriteJSON(map[string]interface{}{"jsonrpc": "2.0", "id": id, "result": "0x1"}) + } + }) + }) + defer mockUpstream.Close() + + setupGock() + defer util.ResetGock() + + wsUpstreamURL := "ws" + strings.TrimPrefix(mockUpstream.URL, "http") + addr, cleanup := setupTestERPCServer(t, standardWsConfig(wsUpstreamURL)) + time.Sleep(2 * time.Second) + + conn := dialWs(t, addr) + defer conn.Close() + + resp := sendAndReceive(t, conn, `{"jsonrpc":"2.0","id":1,"method":"eth_subscribe","params":["newHeads"]}`) + require.NotNil(t, resp["result"]) + + cleanup() + + // After shutdown, the client connection should be closed + conn.SetReadDeadline(time.Now().Add(5 * time.Second)) + _, _, err := conn.ReadMessage() + assert.Error(t, err, "client should be disconnected after server shutdown") + }) + + // Verifies all connections are closed when the server shuts down + t.Run("MultipleConnectionsClosedOnShutdown", func(t *testing.T) { + setupGock() + defer util.ResetGock() + + addr, cleanup := setupTestERPCServer(t, httpOnlyConfig()) + + conn1 := dialWs(t, addr) + defer conn1.Close() + conn2 := dialWs(t, addr) + defer conn2.Close() + conn3 := dialWs(t, addr) + defer conn3.Close() + + resp1 := sendAndReceive(t, conn1, `{"jsonrpc":"2.0","id":1,"method":"eth_getBalance","params":["0xaaaa","latest"]}`) + assert.Equal(t, "0xabc123", resp1["result"]) + resp2 := sendAndReceive(t, conn2, `{"jsonrpc":"2.0","id":2,"method":"eth_getBalance","params":["0xaaaa","latest"]}`) + assert.Equal(t, "0xabc123", resp2["result"]) + resp3 := sendAndReceive(t, conn3, `{"jsonrpc":"2.0","id":3,"method":"eth_getBalance","params":["0xaaaa","latest"]}`) + assert.Equal(t, "0xabc123", resp3["result"]) + + cleanup() + + closedCount := 0 + for _, conn := range []*websocket.Conn{conn1, conn2, conn3} { + conn.SetReadDeadline(time.Now().Add(5 * time.Second)) + _, _, err := conn.ReadMessage() + if err != nil { + closedCount++ + } + } + assert.Equal(t, 3, closedCount, "all connections should be closed on shutdown") + }) +} + +// +// --- Tests: method filtering --- +// + +func TestWebSocket_MethodFiltering(t *testing.T) { + // Verifies ignored methods return an unsupported error over WebSocket + t.Run("IgnoredMethodReturnsError", func(t *testing.T) { + setupGock() + defer util.ResetGock() + + cfg := &common.Config{ + Server: &common.ServerConfig{ + ListenV4: util.BoolPtr(true), + }, + Projects: []*common.ProjectConfig{ + { + Id: "test_ws", + IgnoreMethods: []string{"debug_*"}, + Networks: []*common.NetworkConfig{ + { + Architecture: common.ArchitectureEvm, + Evm: &common.EvmNetworkConfig{ChainId: 123}, + }, + }, + Upstreams: []*common.UpstreamConfig{ + { + Type: common.UpstreamTypeEvm, + Endpoint: "http://rpc1.localhost", + Evm: &common.EvmUpstreamConfig{ChainId: 123}, + }, + }, + }, + }, + RateLimiters: &common.RateLimiterConfig{}, + } + + addr, cleanup := setupTestERPCServer(t, cfg) + defer cleanup() + + conn := dialWs(t, addr) + defer conn.Close() + + resp := sendAndReceive(t, conn, `{"jsonrpc":"2.0","id":1,"method":"debug_traceTransaction","params":["0xabc"]}`) + assert.NotNil(t, resp["error"], "ignored method should return error") + errObj := resp["error"].(map[string]interface{}) + assert.Contains(t, errObj["message"], "not supported") + }) +} + +// +// --- Regression tests --- +// + +// TestWebSocket_RegressionFailedSubscribeKeepsConnectionOpen verifies that a +// failed subscribe does NOT close the entire client connection. A regression +// where eRPC closed the WS on subscribe failure caused cascading failures — +// one bad subscribe killed all working ones for downstream consumers. +func TestWebSocket_RegressionFailedSubscribeKeepsConnectionOpen(t *testing.T) { + setupGock() + defer util.ResetGock() + + addr, cleanup := setupTestERPCServer(t, httpOnlyConfig()) + defer cleanup() + + conn := dialWs(t, addr) + defer conn.Close() + + // Subscribe MUST fail (no WS upstream configured) but the connection + // must remain open for subsequent requests. + resp := sendAndReceive(t, conn, `{"jsonrpc":"2.0","id":1,"method":"eth_subscribe","params":["newHeads"]}`) + require.NotNil(t, resp["error"], "subscribe should fail") + + // Connection should still work — send a normal RPC after failed subscribe. + conn.SetReadDeadline(time.Now().Add(3 * time.Second)) + resp2 := sendAndReceive(t, conn, `{"jsonrpc":"2.0","id":2,"method":"eth_getBalance","params":["0xaaaa","latest"]}`) + assert.Equal(t, "0xabc123", resp2["result"], "connection should remain open after subscribe failure") +} + +// TestWebSocket_RegressionBootstrapRetriedOnEverySubscribe verifies that a +// failed initial bootstrap doesn't permanently break newHeads delivery — +// subsequent client subscribes must re-attempt the upstream subscription. +// +// Previously a sync.Map "bootstrapped" flag was set even on failure, so the +// first failure caused permanent silence on the network. +func TestWebSocket_RegressionBootstrapRetriedOnEverySubscribe(t *testing.T) { + subscribeCount := int64(0) + mockUpstream := mockWsUpstream(t, func(conn *websocket.Conn) { + standardMockWsHandler(conn, func(method string, id interface{}, req map[string]interface{}) { + switch method { + case "eth_subscribe": + atomic.AddInt64(&subscribeCount, 1) + conn.WriteJSON(map[string]interface{}{"jsonrpc": "2.0", "id": id, "result": "0xabc"}) + default: + conn.WriteJSON(map[string]interface{}{"jsonrpc": "2.0", "id": id, "result": "0x1"}) + } + }) + }) + defer mockUpstream.Close() + + setupGock() + defer util.ResetGock() + + wsUpstreamURL := "ws" + strings.TrimPrefix(mockUpstream.URL, "http") + addr, cleanup := setupTestERPCServer(t, standardWsConfig(wsUpstreamURL)) + defer cleanup() + time.Sleep(2 * time.Second) + + // First subscribe triggers bootstrap on the WS upstream. + conn1 := dialWs(t, addr) + defer conn1.Close() + resp1 := sendAndReceive(t, conn1, `{"jsonrpc":"2.0","id":1,"method":"eth_subscribe","params":["newHeads"]}`) + require.NotNil(t, resp1["result"]) + first := atomic.LoadInt64(&subscribeCount) + require.GreaterOrEqual(t, first, int64(1), "first client should trigger upstream subscribe") + + // Subsequent subscribes (different params) must also call BootstrapNetwork's + // idempotent path — no new upstream subscribe expected since the upstream + // newHeads sub already exists, but the call must not be skipped due to a + // "bootstrapped once" gate. + conn2 := dialWs(t, addr) + defer conn2.Close() + resp2 := sendAndReceive(t, conn2, `{"jsonrpc":"2.0","id":1,"method":"eth_subscribe","params":["newHeads"]}`) + require.NotNil(t, resp2["result"]) + assert.Equal(t, first, atomic.LoadInt64(&subscribeCount), + "second client should reuse existing newHeads sub (idempotent bootstrap)") +} + +// TestWebSocket_RegressionSuggestLatestBlockOnEveryUpstream verifies that +// SuggestLatestBlock fires for every WS upstream that delivers a block, +// regardless of the network-level dedup. Previously only the first upstream +// to deliver got its poller updated, causing other upstreams to appear +// behind and be rejected by the block availability check. +func TestWebSocket_RegressionSuggestLatestBlockOnEveryUpstream(t *testing.T) { + // Two WS upstreams that both deliver the same block. + upSubId := "0xupsub" + + deliverNotification := func(conn *websocket.Conn, blockHash string) { + conn.WriteJSON(map[string]interface{}{ + "jsonrpc": "2.0", + "method": "eth_subscription", + "params": map[string]interface{}{ + "subscription": upSubId, + "result": map[string]interface{}{ + "number": "0x123", + "hash": blockHash, + }, + }, + }) + } + + makeMock := func() *httptest.Server { + return mockWsUpstream(t, func(conn *websocket.Conn) { + standardMockWsHandler(conn, func(method string, id interface{}, req map[string]interface{}) { + switch method { + case "eth_subscribe": + conn.WriteJSON(map[string]interface{}{"jsonrpc": "2.0", "id": id, "result": upSubId}) + go func() { + time.Sleep(300 * time.Millisecond) + deliverNotification(conn, "0xsamehash") + }() + default: + conn.WriteJSON(map[string]interface{}{"jsonrpc": "2.0", "id": id, "result": "0x1"}) + } + }) + }) + } + mock1 := makeMock() + defer mock1.Close() + mock2 := makeMock() + defer mock2.Close() + + setupGock() + defer util.ResetGock() + + url1 := "ws" + strings.TrimPrefix(mock1.URL, "http") + url2 := "ws" + strings.TrimPrefix(mock2.URL, "http") + addr, cleanup := setupTestERPCServer(t, multiWsConfig(url1, url2)) + defer cleanup() + time.Sleep(2 * time.Second) + + conn := dialWs(t, addr) + defer conn.Close() + resp := sendAndReceive(t, conn, `{"jsonrpc":"2.0","id":1,"method":"eth_subscribe","params":["newHeads"]}`) + require.NotNil(t, resp["result"], "subscribe should succeed with multiple WS upstreams") + + // Client should receive exactly one fan-out (deduped across upstreams). + conn.SetReadDeadline(time.Now().Add(3 * time.Second)) + _, msg, err := conn.ReadMessage() + require.NoError(t, err, "should receive deduped notification") + var notif map[string]interface{} + require.NoError(t, json.Unmarshal(msg, ¬if)) + assert.Equal(t, "eth_subscription", notif["method"]) + + // A second notification should NOT arrive — dedup must drop the duplicate + // from the second upstream. + conn.SetReadDeadline(time.Now().Add(800 * time.Millisecond)) + _, _, err = conn.ReadMessage() + assert.Error(t, err, "duplicate block from second upstream must be deduped") +} + +// TestWebSocket_RegressionFilterFanOutAcrossUpstreams verifies that logs +// subscriptions subscribe on ALL WS upstreams and dedup notifications by +// content (blockHash + txHash + logIndex). Previously logs subscribed on a +// single "best" upstream; one disconnect caused gaps until re-route. +func TestWebSocket_RegressionFilterFanOutAcrossUpstreams(t *testing.T) { + subscribeCount := int64(0) + logSubId := "0xlogsub" + + deliverLog := func(conn *websocket.Conn, blockHash, txHash, logIndex string) { + conn.WriteJSON(map[string]interface{}{ + "jsonrpc": "2.0", + "method": "eth_subscription", + "params": map[string]interface{}{ + "subscription": logSubId, + "result": map[string]interface{}{ + "address": "0xabc", + "blockHash": blockHash, + "transactionHash": txHash, + "logIndex": logIndex, + "removed": false, + "topics": []string{}, + "data": "0x", + "blockNumber": "0x123", + "transactionIndex": "0x0", + }, + }, + }) + } + + makeMock := func() *httptest.Server { + return mockWsUpstream(t, func(conn *websocket.Conn) { + standardMockWsHandler(conn, func(method string, id interface{}, req map[string]interface{}) { + switch method { + case "eth_subscribe": + params, _ := req["params"].([]interface{}) + if len(params) > 0 && params[0] == "logs" { + atomic.AddInt64(&subscribeCount, 1) + conn.WriteJSON(map[string]interface{}{"jsonrpc": "2.0", "id": id, "result": logSubId}) + go func() { + time.Sleep(300 * time.Millisecond) + // Both upstreams deliver the same log. + deliverLog(conn, "0xblock1", "0xtx1", "0x0") + }() + } else { + conn.WriteJSON(map[string]interface{}{"jsonrpc": "2.0", "id": id, "result": "0xothersub"}) + } + default: + conn.WriteJSON(map[string]interface{}{"jsonrpc": "2.0", "id": id, "result": "0x1"}) + } + }) + }) + } + mock1 := makeMock() + defer mock1.Close() + mock2 := makeMock() + defer mock2.Close() + + setupGock() + defer util.ResetGock() + + url1 := "ws" + strings.TrimPrefix(mock1.URL, "http") + url2 := "ws" + strings.TrimPrefix(mock2.URL, "http") + addr, cleanup := setupTestERPCServer(t, multiWsConfig(url1, url2)) + defer cleanup() + time.Sleep(2 * time.Second) + + conn := dialWs(t, addr) + defer conn.Close() + resp := sendAndReceive(t, conn, `{"jsonrpc":"2.0","id":1,"method":"eth_subscribe","params":["logs",{}]}`) + require.NotNil(t, resp["result"]) + + // Wait for both upstreams to be subscribed. + require.Eventually(t, func() bool { + return atomic.LoadInt64(&subscribeCount) >= 2 + }, 5*time.Second, 100*time.Millisecond, "both WS upstreams should be subscribed for logs") + + // Client should receive exactly one log notification (deduped). + conn.SetReadDeadline(time.Now().Add(3 * time.Second)) + _, msg, err := conn.ReadMessage() + require.NoError(t, err, "should receive deduped log notification") + var notif map[string]interface{} + require.NoError(t, json.Unmarshal(msg, ¬if)) + assert.Equal(t, "eth_subscription", notif["method"]) + + // Duplicate from second upstream must be deduped. + conn.SetReadDeadline(time.Now().Add(800 * time.Millisecond)) + _, _, err = conn.ReadMessage() + assert.Error(t, err, "duplicate log from second upstream must be deduped by blockHash+txHash+logIndex") +} + +// TestWebSocket_RegressionUnsubscribeDoesNotPanicOnReconnect verifies that +// when a filter group is torn down via UnsubscribeFilter, a subsequent +// reconnect callback firing does not panic with "assignment to entry in nil +// map". The reconnect callback closes over the group pointer and may fire +// after the group is removed from the parent map. +func TestWebSocket_RegressionUnsubscribeDoesNotPanicOnReconnect(t *testing.T) { + logSubId := "0xunsublog" + mockUpstream := mockWsUpstream(t, func(conn *websocket.Conn) { + standardMockWsHandler(conn, func(method string, id interface{}, req map[string]interface{}) { + if method == "eth_subscribe" { + params, _ := req["params"].([]interface{}) + if len(params) > 0 && params[0] == "logs" { + conn.WriteJSON(map[string]interface{}{"jsonrpc": "2.0", "id": id, "result": logSubId}) + return + } + } + conn.WriteJSON(map[string]interface{}{"jsonrpc": "2.0", "id": id, "result": "0xothersub"}) + }) + }) + defer mockUpstream.Close() + + setupGock() + defer util.ResetGock() + + wsUpstreamURL := "ws" + strings.TrimPrefix(mockUpstream.URL, "http") + addr, cleanup := setupTestERPCServer(t, standardWsConfig(wsUpstreamURL)) + defer cleanup() + time.Sleep(2 * time.Second) + + conn := dialWs(t, addr) + defer conn.Close() + + // Subscribe to logs (creates filter sub group). + resp := sendAndReceive(t, conn, `{"jsonrpc":"2.0","id":1,"method":"eth_subscribe","params":["logs",{}]}`) + require.NotNil(t, resp["result"]) + clientSubId := resp["result"].(string) + + // Unsubscribe (tears down the filter group). + unsubResp := sendAndReceive(t, conn, fmt.Sprintf(`{"jsonrpc":"2.0","id":2,"method":"eth_unsubscribe","params":["%s"]}`, clientSubId)) + assert.Equal(t, true, unsubResp["result"]) + + // Force the upstream WS to drop, triggering the reconnect callback that + // closes over the now-torn-down group. Without the tornDown flag check, + // this would panic with "assignment to entry in nil map". + mockUpstream.CloseClientConnections() + + // Sleep long enough for reconnect + callback to fire. + time.Sleep(3 * time.Second) + + // eRPC should still be alive — make a normal RPC call. + resp2 := sendAndReceive(t, conn, `{"jsonrpc":"2.0","id":3,"method":"eth_getBalance","params":["0xaaaa","latest"]}`) + assert.NotNil(t, resp2, "eRPC should not have crashed from reconnect-after-unsubscribe") +} + +// TestWebSocket_RegressionInternalRequestIdsDontCollide verifies that internal +// eth_subscribe requests use unique IDs that don't collide with state poller +// or client requests on the same WS connection. Previously a hardcoded id=1 +// caused responses to be misrouted in the WsJsonRpcClient pending map. +func TestWebSocket_RegressionInternalRequestIdsDontCollide(t *testing.T) { + subscribeIds := make([]int64, 0) + var idMu sync.Mutex + + mockUpstream := mockWsUpstream(t, func(conn *websocket.Conn) { + standardMockWsHandler(conn, func(method string, id interface{}, req map[string]interface{}) { + if method == "eth_subscribe" { + idMu.Lock() + if f, ok := id.(float64); ok { + subscribeIds = append(subscribeIds, int64(f)) + } + idMu.Unlock() + conn.WriteJSON(map[string]interface{}{"jsonrpc": "2.0", "id": id, "result": "0xabc"}) + return + } + conn.WriteJSON(map[string]interface{}{"jsonrpc": "2.0", "id": id, "result": "0x1"}) + }) + }) + defer mockUpstream.Close() + + setupGock() + defer util.ResetGock() + + wsUpstreamURL := "ws" + strings.TrimPrefix(mockUpstream.URL, "http") + addr, cleanup := setupTestERPCServer(t, standardWsConfig(wsUpstreamURL)) + defer cleanup() + time.Sleep(2 * time.Second) + + // Trigger an internal upstream subscribe via a client subscribe. + conn := dialWs(t, addr) + defer conn.Close() + resp := sendAndReceive(t, conn, `{"jsonrpc":"2.0","id":1,"method":"eth_subscribe","params":["newHeads"]}`) + require.NotNil(t, resp["result"]) + + idMu.Lock() + defer idMu.Unlock() + require.NotEmpty(t, subscribeIds, "should have observed at least one internal eth_subscribe") + + // Internal IDs use a large offset (>= 900M) to avoid collisions with the + // state poller's small integer IDs. + for _, id := range subscribeIds { + assert.GreaterOrEqual(t, id, int64(900_000_000), + "internal eth_subscribe id %d should be offset to avoid collision with poller IDs", id) + } + + // All IDs unique. + seen := make(map[int64]bool) + for _, id := range subscribeIds { + assert.False(t, seen[id], "internal id %d duplicated", id) + seen[id] = true + } +} diff --git a/go.mod b/go.mod index a9481ff44..23efa7edb 100644 --- a/go.mod +++ b/go.mod @@ -98,6 +98,7 @@ require ( github.com/google/pprof v0.0.0-20240727154555-813a5fbdbec8 // indirect github.com/google/uuid v1.6.0 // indirect github.com/gorilla/mux v1.8.1 // indirect + github.com/gorilla/websocket v1.5.3 // indirect github.com/grpc-ecosystem/go-grpc-middleware v1.4.0 // indirect github.com/grpc-ecosystem/grpc-gateway/v2 v2.27.7 // indirect github.com/h2non/parth v0.0.0-20190131123155-b4df798d6542 // indirect diff --git a/go.sum b/go.sum index 82f001328..b89a04ef0 100644 --- a/go.sum +++ b/go.sum @@ -195,6 +195,8 @@ github.com/google/uuid v1.6.0 h1:NIvaJDMOsjHA8n1jAhLSgzrAzy1Hgr+hNrb57e+94F0= github.com/google/uuid v1.6.0/go.mod h1:TIyPZe4MgqvfeYDBFedMoGGpEw/LqOeaOT+nhxU+yHo= github.com/gorilla/mux v1.8.1 h1:TuBL49tXwgrFYWhqrNgrUNEY92u81SPhu7sTdzQEiWY= github.com/gorilla/mux v1.8.1/go.mod h1:AKf9I4AEqPTmMytcMc0KkNouC66V3BtZ4qD5fmWSiMQ= +github.com/gorilla/websocket v1.5.3 h1:saDtZ6Pbx/0u+bgYQ3q96pZgCzfhKXGPqt7kZ72aNNg= +github.com/gorilla/websocket v1.5.3/go.mod h1:YR8l580nyteQvAITg2hZ9XVh4b55+EU/adAjf1fMHhE= github.com/grafana/sobek v0.0.0-20241024150027-d91f02b05e9b h1:hzfIt1lf19Zx1jIYdeHvuWS266W+jL+7dxbpvH2PZMQ= github.com/grafana/sobek v0.0.0-20241024150027-d91f02b05e9b/go.mod h1:FmcutBFPLiGgroH42I4/HBahv7GxVjODcVWFTw1ISes= github.com/grpc-ecosystem/go-grpc-middleware v1.4.0 h1:UH//fgunKIs4JdUbpDl1VZCDaL56wXCB/5+wF6uHfaI= diff --git a/indexer/adapters/nullingress/adapter.go b/indexer/adapters/nullingress/adapter.go new file mode 100644 index 000000000..ed2b39811 --- /dev/null +++ b/indexer/adapters/nullingress/adapter.go @@ -0,0 +1,119 @@ +// Package nullingress is a minimal, transport-free EventIngress +// implementation. It exists to (a) prove indexer.EventIngress isn't +// secretly shaped around WebSocket semantics and (b) give tests a +// deterministic way to inject StreamEvents into the indexer pipeline. +// +// It is NOT a production ingress — the Push method lets tests feed raw +// events with no backing transport. A future Kafka consumer would slot +// in at the same seam. +package nullingress + +import ( + "context" + "sync" + + "github.com/erpc/erpc/indexer" +) + +// Adapter implements indexer.EventIngress as an in-memory feed. +type Adapter struct { + name string + + mu sync.Mutex + started bool + cancel context.CancelFunc + sink indexer.Sink + events chan indexer.StreamEvent + + // Active filter bookkeeping — tests can assert on this to verify + // EnsureFilter/RemoveFilter were invoked by the indexer. + filters map[string]int // subType:paramsHash -> refcount (the indexer itself refcounts, this is just for observation) +} + +// New constructs a nullingress adapter. name is surfaced via Name() +// for indexer registries/logs ("null:mock1"). +func New(name string) *Adapter { + return &Adapter{ + name: name, + events: make(chan indexer.StreamEvent, 64), + filters: make(map[string]int), + } +} + +// Name implements indexer.EventIngress. +func (a *Adapter) Name() string { return a.name } + +// Start implements indexer.EventIngress. Spins up a goroutine that +// forwards any Push()ed event to the indexer's Sink. +func (a *Adapter) Start(ctx context.Context, _ indexer.NetworkHandle, sink indexer.Sink) error { + a.mu.Lock() + defer a.mu.Unlock() + if a.started { + return nil + } + ctx, cancel := context.WithCancel(ctx) + a.cancel = cancel + a.sink = sink + a.started = true + go a.pump(ctx) + return nil +} + +func (a *Adapter) pump(ctx context.Context) { + for { + select { + case <-ctx.Done(): + return + case ev := <-a.events: + a.sink.Ingest(ev) + } + } +} + +// EnsureFilter implements indexer.EventIngress. +func (a *Adapter) EnsureFilter(_ context.Context, subType, paramsHash string, _ []interface{}) error { + key := subType + ":" + paramsHash + a.mu.Lock() + a.filters[key]++ + a.mu.Unlock() + return nil +} + +// RemoveFilter implements indexer.EventIngress. +func (a *Adapter) RemoveFilter(_ context.Context, subType, paramsHash string) error { + key := subType + ":" + paramsHash + a.mu.Lock() + delete(a.filters, key) + a.mu.Unlock() + return nil +} + +// Stop implements indexer.EventIngress. +func (a *Adapter) Stop(_ context.Context) error { + a.mu.Lock() + defer a.mu.Unlock() + if a.cancel != nil { + a.cancel() + } + a.started = false + return nil +} + +// Push injects a StreamEvent into the ingress. Test-only. Will block if +// the internal channel fills up — intentional, so slow-consumer bugs in +// the indexer show up as a test timeout rather than silent drops. +func (a *Adapter) Push(ev indexer.StreamEvent) { + a.events <- ev +} + +// ActiveFilters returns a snapshot of currently-active filter keys, for +// test assertions on EnsureFilter / RemoveFilter plumbing. +func (a *Adapter) ActiveFilters() []string { + a.mu.Lock() + defer a.mu.Unlock() + out := make([]string, 0, len(a.filters)) + for k := range a.filters { + out = append(out, k) + } + return out +} diff --git a/indexer/adapters/wsclient/adapter.go b/indexer/adapters/wsclient/adapter.go new file mode 100644 index 000000000..3841404e9 --- /dev/null +++ b/indexer/adapters/wsclient/adapter.go @@ -0,0 +1,271 @@ +// Package wsclient adapts a single client WebSocket connection into an +// indexer.EventEgress. One adapter per WsConnection. The adapter owns +// per-subscription write buffers and writer goroutines — the drop policy +// (oldest-drop for slow clients) lives here rather than in the indexer +// core because future egresses (Kafka, webhook) will want different +// policies. +package wsclient + +import ( + "encoding/json" + "sync" + "sync/atomic" + + "github.com/erpc/erpc/indexer" + "github.com/rs/zerolog" +) + +// clientNotifyBufferSize is the per-subscription buffer depth. Slow +// clients drop their oldest queued notification when this is full. +// +// Sized small because retention scales with depth, not throughput: +// each buffered notification pins the upstream-allocated +// json.RawMessage until the writer drains it, so total GC-reachable +// heap from this path is roughly N_subs × depth × payload_size. Block +// header payloads can run tens of KB on chains that pack extended +// header fields, and long-lived client subscriptions hold their +// buffers across many newHead notifications. A small depth keeps the +// drop-oldest threshold tight: a consumer that hasn't drained this +// many events has already fallen far enough behind to need a re-sync. +const clientNotifyBufferSize = 8 + +// NotificationWriter is what the adapter uses to push a delivered event +// onto the wire. The indirection lets the adapter stay unaware of +// gorilla/websocket specifically — any transport that can deliver a +// (clientSubId, result) pair to a single consumer works. +type NotificationWriter interface { + WriteSubscriptionNotification(clientSubId string, result json.RawMessage) error +} + +// Adapter is the per-connection EventEgress. Subscribe/Unsubscribe calls +// on the underlying connection translate into AddSubscription / +// RemoveSubscription here; the indexer calls InterestedIn on every event +// and Deliver for matches. +type Adapter struct { + connID string + writer NotificationWriter + logger *zerolog.Logger + + mu sync.RWMutex + // subs keyed by clientSubId. A single map supports both newHeads + // and filter subs. + subs map[string]*clientSub + // routes is an interest-index: (kind, networkId, filterHash) -> set + // of clientSubIds. Kept in sync with subs for O(1) InterestedIn. + routes map[routeKey]map[string]struct{} +} + +type routeKey struct { + kind indexer.EventKind + networkID string + filterHash string +} + +type clientSub struct { + id string + kind indexer.EventKind + networkID string + // filterHash is "" for newHeads. + filterHash string + + notify chan json.RawMessage + done chan struct{} + closed atomic.Bool +} + +// New creates an adapter for a single client connection. The writer is +// invoked from the per-subscription goroutine; it must be safe for +// concurrent use (typically an internal sync.Mutex on the WS connection). +func New(connID string, writer NotificationWriter, logger *zerolog.Logger) *Adapter { + lg := logger.With().Str("connId", connID).Logger() + return &Adapter{ + connID: connID, + writer: writer, + logger: &lg, + subs: make(map[string]*clientSub), + routes: make(map[routeKey]map[string]struct{}), + } +} + +// Name identifies the adapter in indexer registries and logs. +func (a *Adapter) Name() string { return "ws:client:" + a.connID } + +// InterestedIn: O(1) map lookup. Called by the indexer on every event. +func (a *Adapter) InterestedIn(kind indexer.EventKind, networkID, filterHash string) bool { + a.mu.RLock() + defer a.mu.RUnlock() + _, ok := a.routes[routeKey{kind: kind, networkID: networkID, filterHash: filterHash}] + return ok +} + +// Deliver enqueues the event on every matching per-sub channel. If a +// sub's buffer is full the oldest queued notification is dropped to make +// room (fresher data is preferred, and the indexer must not block on +// this egress). +func (a *Adapter) Deliver(ev indexer.IndexedEvent) { + a.mu.RLock() + routes, ok := a.routes[routeKey{kind: ev.Kind, networkID: ev.NetworkId, filterHash: ev.FilterHash}] + if !ok || len(routes) == 0 { + a.mu.RUnlock() + return + } + ids := make([]string, 0, len(routes)) + for id := range routes { + ids = append(ids, id) + } + a.mu.RUnlock() + + for _, id := range ids { + a.mu.RLock() + sub, ok := a.subs[id] + a.mu.RUnlock() + if !ok { + continue + } + enqueue(sub, ev.Payload) + } +} + +// AddSubscription registers a client subscription on this connection and +// starts its writer goroutine. clientSubId is the erpc-generated opaque +// ID the caller already returned to the client. filterHash is "" for +// newHeads. +func (a *Adapter) AddSubscription(clientSubID, networkID string, kind indexer.EventKind, filterHash string) { + sub := &clientSub{ + id: clientSubID, + kind: kind, + networkID: networkID, + filterHash: filterHash, + notify: make(chan json.RawMessage, clientNotifyBufferSize), + done: make(chan struct{}), + } + + a.mu.Lock() + a.subs[clientSubID] = sub + key := routeKey{kind: kind, networkID: networkID, filterHash: filterHash} + set, ok := a.routes[key] + if !ok { + set = make(map[string]struct{}) + a.routes[key] = set + } + set[clientSubID] = struct{}{} + a.mu.Unlock() + + go a.runWriter(sub) +} + +// RemoveSubscription deregisters a subscription and stops its writer. It +// returns the (kind, filterHash) of the removed sub so the caller can +// decide whether to call indexer.ReleaseFilter (for filter subs once the +// refcount drops to zero on its side). +func (a *Adapter) RemoveSubscription(clientSubID string) (kind indexer.EventKind, networkID, filterHash string, existed bool) { + a.mu.Lock() + sub, ok := a.subs[clientSubID] + if !ok { + a.mu.Unlock() + return 0, "", "", false + } + delete(a.subs, clientSubID) + key := routeKey{kind: sub.kind, networkID: sub.networkID, filterHash: sub.filterHash} + if set, ok := a.routes[key]; ok { + delete(set, clientSubID) + if len(set) == 0 { + delete(a.routes, key) + } + } + a.mu.Unlock() + + if sub.closed.CompareAndSwap(false, true) { + close(sub.done) + } + return sub.kind, sub.networkID, sub.filterHash, true +} + +// Drain stops every sub's writer — call on connection close before +// detaching from the indexer. +func (a *Adapter) Drain() { + a.mu.Lock() + subs := a.subs + a.subs = make(map[string]*clientSub) + a.routes = make(map[routeKey]map[string]struct{}) + a.mu.Unlock() + + for _, sub := range subs { + if sub.closed.CompareAndSwap(false, true) { + close(sub.done) + } + } +} + +// Subscriptions returns a snapshot of the active subscriptions. Used by +// connection cleanup paths to iterate subs without racing with Deliver. +func (a *Adapter) Subscriptions() []Subscription { + a.mu.RLock() + defer a.mu.RUnlock() + out := make([]Subscription, 0, len(a.subs)) + for _, sub := range a.subs { + out = append(out, Subscription{ + ClientSubID: sub.id, + Kind: sub.kind, + NetworkID: sub.networkID, + FilterHash: sub.filterHash, + }) + } + return out +} + +// Subscription is a caller-visible snapshot of one active subscription. +type Subscription struct { + ClientSubID string + Kind indexer.EventKind + NetworkID string + FilterHash string +} + +// Count returns the number of active subscriptions — used by the +// client-facing manager to enforce per-connection limits. +func (a *Adapter) Count() int { + a.mu.RLock() + defer a.mu.RUnlock() + return len(a.subs) +} + +// --- writer goroutine ------------------------------------------------- + +func (a *Adapter) runWriter(sub *clientSub) { + for { + select { + case <-sub.done: + return + case payload, ok := <-sub.notify: + if !ok { + return + } + if err := a.writer.WriteSubscriptionNotification(sub.id, payload); err != nil { + a.logger.Debug().Err(err).Str("clientSubId", sub.id). + Msg("failed to write subscription notification") + // Errors are per-sub; the connection-close path will + // Drain us when the peer is truly gone. + } + } + } +} + +// enqueue pushes a payload onto the sub's buffer, evicting the oldest +// element when the buffer is full. Never blocks. +func enqueue(sub *clientSub, payload json.RawMessage) { + for { + select { + case sub.notify <- payload: + return + default: + // Buffer full; drop oldest to make room. + select { + case <-sub.notify: + default: + // Concurrent drain won the race — drop this message. + return + } + } + } +} diff --git a/indexer/adapters/wsupstream/adapter.go b/indexer/adapters/wsupstream/adapter.go new file mode 100644 index 000000000..4bc3b19d3 --- /dev/null +++ b/indexer/adapters/wsupstream/adapter.go @@ -0,0 +1,467 @@ +// Package wsupstream adapts a single WebSocket JSON-RPC upstream into an +// indexer.EventIngress. One adapter per (upstream, network). The adapter +// owns the eth_subscribe / eth_unsubscribe RPCs, the per-subscription +// handler registration, and the re-subscribe-on-reconnect hook. Dedup, +// lifecycle tagging, and client fan-out live in the indexer core — this +// package is strictly a WS ↔ StreamEvent bridge. +package wsupstream + +import ( + "context" + "encoding/json" + "fmt" + "strconv" + "strings" + "sync" + "sync/atomic" + "time" + + "github.com/erpc/erpc/clients" + "github.com/erpc/erpc/common" + "github.com/erpc/erpc/indexer" + "github.com/erpc/erpc/upstream" + "github.com/rs/zerolog" +) + +const ( + methodEthSubscribe = "eth_subscribe" + methodEthUnsubscribe = "eth_unsubscribe" +) + +// internalReqIDOffset keeps our JSON-RPC IDs out of the range clients +// typically use (small incrementing ints). 900M gives us ~9.2×10^18 +// distinct internal IDs before the int64 overflows. +const internalReqIDOffset = 900_000_000 + +var internalReqIDCounter atomic.Int64 + +// Adapter bridges one WS upstream into the indexer pipeline. It is +// created per (upstream, network) at startup and registered via +// indexer.AddIngress. +type Adapter struct { + upstreamID string + networkID string + upstream *upstream.Upstream + wsClient *clients.WsJsonRpcClient + logger *zerolog.Logger + + // stripSubscribeFromBlockZero controls whether fromBlock: "0x0" is + // removed from eth_subscribe logs filters before forwarding upstream. + // See common.EvmNetworkConfig.StripSubscribeFromBlockZero for details. + stripSubscribeFromBlockZero bool + + nw indexer.NetworkHandle + sink indexer.Sink + + // subsMu guards all mutable state below. Kept coarse-grained because + // (a) the hot path (handleNotification) doesn't touch it and (b) the + // reconnect path needs consistency across all maps. + subsMu sync.Mutex + // newHeadsSubID is the upstream-assigned ID for our newHeads sub, or "" + // when not (yet) subscribed. + newHeadsSubID string + // filters keyed by `subType + ":" + paramsHash`. Survives disconnects + // so we can re-subscribe on reconnect. + filters map[string]*filterSub +} + +// Options carries optional settings for New. Fields zero-valued by default +// preserve the adapter's standard behaviour — only set what you want to +// override. +type Options struct { + // StripSubscribeFromBlockZero, when true, removes fromBlock: "0x0" from + // eth_subscribe logs filters before sending to the upstream. See + // common.EvmNetworkConfig.StripSubscribeFromBlockZero. + StripSubscribeFromBlockZero bool +} + +type filterSub struct { + subType string + paramsHash string + params []interface{} + upstreamSub string // assigned on each (re)subscribe; empty before first attempt +} + +// New constructs an adapter for one upstream. Returns nil if the upstream +// is not backed by a WsJsonRpcClient (i.e. it's HTTP) — callers filter +// upstream lists up-front but this is a cheap safety net. +// +// Pass opts for network-level behaviour overrides; a nil opts preserves +// default behaviour. +func New(up *upstream.Upstream, networkID string, logger *zerolog.Logger, opts *Options) *Adapter { + wsClient, ok := up.Client.(*clients.WsJsonRpcClient) + if !ok { + return nil + } + lg := logger.With().Str("upstreamId", up.Id()).Str("networkId", networkID).Logger() + a := &Adapter{ + upstreamID: up.Id(), + networkID: networkID, + upstream: up, + wsClient: wsClient, + logger: &lg, + filters: make(map[string]*filterSub), + } + if opts != nil { + a.stripSubscribeFromBlockZero = opts.StripSubscribeFromBlockZero + } + return a +} + +// Name identifies the adapter in indexer registries and logs. Unique per +// upstream — two networks never share a WS client today. +func (a *Adapter) Name() string { return "ws:" + a.upstreamID } + +// Start wires the adapter to the indexer's sink, registers the +// reconnect/disconnect hooks, and kicks off the initial newHeads +// subscribe in a background goroutine. Per the EventIngress contract +// Start returns once background workers are set up, not once the first +// event arrives — the initial subscribe uses its own detached context +// so a cancelled caller ctx doesn't abort the upstream RPC. +func (a *Adapter) Start(_ context.Context, nw indexer.NetworkHandle, sink indexer.Sink) error { + a.nw = nw + a.sink = sink + + cbID := a.Name() + a.wsClient.SetOnReconnect(cbID, func() { + a.logger.Info().Msg("WS reconnected — re-subscribing to all active subs") + a.resubscribeAll(context.Background()) + }) + a.wsClient.SetOnDisconnect(cbID, func() { + a.logger.Info().Msg("WS disconnected — active subs will re-subscribe on reconnect") + }) + + go a.initialSubscribe() + return nil +} + +// initialSubscribe attempts the first newHeads subscribe. Retries on its +// own are unnecessary — if the subscribe fails, the reconnect callback +// picks it up the next time the WS client reconnects. If the WS is down +// at Start time, we wait briefly; no-op after that since the reconnect +// hook will fire Start's equivalent. +func (a *Adapter) initialSubscribe() { + ctx, cancel := context.WithTimeout(context.Background(), 15*time.Second) + defer cancel() + if !a.wsClient.IsConnected() { + a.logger.Debug().Msg("WS not yet connected at Start; newHeads will subscribe on first connect") + return + } + a.subscribeNewHeads(ctx) +} + +// EnsureFilter (re)subscribes a filter on this upstream. Safe to call more +// than once for the same paramsHash — the second call replaces the first. +func (a *Adapter) EnsureFilter(ctx context.Context, subType, paramsHash string, params []interface{}) error { + key := filterKey(subType, paramsHash) + + a.subsMu.Lock() + sub, exists := a.filters[key] + if !exists { + sub = &filterSub{subType: subType, paramsHash: paramsHash, params: params} + a.filters[key] = sub + } + a.subsMu.Unlock() + + if !a.wsClient.IsConnected() { + a.logger.Debug().Str("subType", subType).Str("paramsHash", paramsHash). + Msg("WS not connected; filter will subscribe on reconnect") + return nil + } + return a.subscribeFilter(ctx, sub) +} + +// RemoveFilter unsubscribes a filter from the upstream and drops it from +// the adapter's state so it won't be resubscribed on reconnect. Any +// notifications that arrive in the race window are dropped by the +// indexer (no refcount → no fan-out). +func (a *Adapter) RemoveFilter(ctx context.Context, subType, paramsHash string) error { + key := filterKey(subType, paramsHash) + + a.subsMu.Lock() + sub, ok := a.filters[key] + delete(a.filters, key) + a.subsMu.Unlock() + if !ok { + return nil + } + if sub.upstreamSub != "" { + a.wsClient.UnregisterSubscriptionHandler(sub.upstreamSub) + a.sendUnsubscribe(ctx, sub.upstreamSub) + } + return nil +} + +// Stop tears down all upstream subscriptions and removes the reconnect +// hook. Best-effort — upstream RPCs may fail if the connection is +// already gone; we only care that the adapter's state is released. +func (a *Adapter) Stop(ctx context.Context) error { + cbID := a.Name() + a.wsClient.RemoveOnReconnect(cbID) + a.wsClient.RemoveOnDisconnect(cbID) + + a.subsMu.Lock() + subs := a.filters + a.filters = make(map[string]*filterSub) + newHeads := a.newHeadsSubID + a.newHeadsSubID = "" + a.subsMu.Unlock() + + if newHeads != "" { + a.wsClient.UnregisterSubscriptionHandler(newHeads) + a.sendUnsubscribe(ctx, newHeads) + } + for _, sub := range subs { + if sub.upstreamSub == "" { + continue + } + a.wsClient.UnregisterSubscriptionHandler(sub.upstreamSub) + a.sendUnsubscribe(ctx, sub.upstreamSub) + } + return nil +} + +// --- internals --------------------------------------------------------- + +func filterKey(subType, paramsHash string) string { + return subType + ":" + paramsHash +} + +// resubscribeAll re-subscribes newHeads + every tracked filter after a +// WS reconnect. Runs in a dedicated goroutine (reconnect callback) so it +// must be self-sufficient w.r.t. context. +func (a *Adapter) resubscribeAll(ctx context.Context) { + a.subscribeNewHeads(ctx) + a.subsMu.Lock() + subs := make([]*filterSub, 0, len(a.filters)) + for _, s := range a.filters { + subs = append(subs, s) + } + a.subsMu.Unlock() + + for _, sub := range subs { + if err := a.subscribeFilter(ctx, sub); err != nil { + a.logger.Warn().Err(err).Str("subType", sub.subType).Str("paramsHash", sub.paramsHash). + Msg("failed to re-subscribe filter after reconnect") + } + } +} + +func (a *Adapter) subscribeNewHeads(ctx context.Context) { + subID, err := a.sendSubscribe(ctx, []interface{}{indexer.SubTypeNewHeads}) + if err != nil { + a.logger.Warn().Err(err).Msg("failed to subscribe newHeads") + return + } + a.subsMu.Lock() + if a.newHeadsSubID != "" { + a.wsClient.UnregisterSubscriptionHandler(a.newHeadsSubID) + } + a.newHeadsSubID = subID + a.subsMu.Unlock() + + a.wsClient.RegisterSubscriptionHandler(subID, func(params []byte) { + a.handleNewHeads(params) + }) + a.logger.Info().Str("upstreamSubId", subID).Msg("subscribed to newHeads") +} + +func (a *Adapter) subscribeFilter(ctx context.Context, sub *filterSub) error { + outParams := append([]interface{}{sub.subType}, sub.params[1:]...) + if a.stripSubscribeFromBlockZero { + if cleaned, changed := stripFromBlockZero(outParams); changed { + a.logger.Info(). + Str("subType", sub.subType). + Str("paramsHash", sub.paramsHash). + Msg("stripping fromBlock:0x0 from eth_subscribe filter (stripSubscribeFromBlockZero)") + outParams = cleaned + } + } + subID, err := a.sendSubscribe(ctx, outParams) + if err != nil { + return fmt.Errorf("filter subscribe: %w", err) + } + a.subsMu.Lock() + // Replace any previous upstreamSub for this (subType, paramsHash). + if sub.upstreamSub != "" { + a.wsClient.UnregisterSubscriptionHandler(sub.upstreamSub) + } + sub.upstreamSub = subID + a.subsMu.Unlock() + + a.wsClient.RegisterSubscriptionHandler(subID, func(params []byte) { + a.handleFilter(sub.subType, sub.paramsHash, params) + }) + a.logger.Info().Str("upstreamSubId", subID).Str("subType", sub.subType). + Str("paramsHash", sub.paramsHash).Msg("subscribed filter") + return nil +} + +// handleNewHeads converts a newHeads notification into a StreamEvent and +// pushes it at the indexer's Sink. +func (a *Adapter) handleNewHeads(raw []byte) { + var outer struct { + Subscription string `json:"subscription"` + Result json.RawMessage `json:"result"` + } + if err := common.SonicCfg.Unmarshal(raw, &outer); err != nil { + a.logger.Warn().Err(err).Msg("failed to parse newHeads notification envelope") + return + } + var header struct { + Number string `json:"number"` + Hash string `json:"hash"` + ParentHash string `json:"parentHash"` + } + if err := common.SonicCfg.Unmarshal(outer.Result, &header); err != nil { + a.logger.Warn().Err(err).Msg("failed to parse newHeads result") + return + } + num, err := common.HexToInt64(header.Number) + if err != nil { + a.logger.Warn().Err(err).Str("number", header.Number).Msg("failed to parse block number") + return + } + a.sink.Ingest(indexer.StreamEvent{ + Kind: indexer.KindNewHead, + NetworkId: a.networkID, + SourceId: a.Name(), + Block: indexer.BlockRef{Number: num, Hash: header.Hash, ParentHash: header.ParentHash}, + Payload: outer.Result, + ObservedAt: time.Now(), + }) +} + +// handleFilter converts a filter notification into a StreamEvent. +func (a *Adapter) handleFilter(subType, paramsHash string, raw []byte) { + var outer struct { + Subscription string `json:"subscription"` + Result json.RawMessage `json:"result"` + } + if err := common.SonicCfg.Unmarshal(raw, &outer); err != nil { + a.logger.Warn().Err(err).Str("subType", subType).Msg("failed to parse filter notification envelope") + return + } + kind := indexer.KindLog + if subType == indexer.SubTypeNewPendingTransactions { + kind = indexer.KindPendingTx + } + ev := indexer.StreamEvent{ + Kind: kind, + NetworkId: a.networkID, + SourceId: a.Name(), + FilterHash: paramsHash, + Payload: outer.Result, + ObservedAt: time.Now(), + } + // For logs, opportunistically extract the BlockRef so the indexer + // can lifecycle-tag it and feed the canonical-chain tracker. + if kind == indexer.KindLog { + var logProbe struct { + BlockNumber string `json:"blockNumber"` + BlockHash string `json:"blockHash"` + } + if err := common.SonicCfg.Unmarshal(outer.Result, &logProbe); err == nil { + if n, err := common.HexToInt64(logProbe.BlockNumber); err == nil { + ev.Block = indexer.BlockRef{Number: n, Hash: logProbe.BlockHash} + } + } + } + a.sink.Ingest(ev) +} + +// sendSubscribe performs the eth_subscribe RPC via the upstream's +// failsafe executor so retry/timeout policies still apply. +func (a *Adapter) sendSubscribe(ctx context.Context, params []interface{}) (string, error) { + body, err := buildJSONRPCBody(methodEthSubscribe, params) + if err != nil { + return "", err + } + nq := common.NewNormalizedRequest(body) + resp, err := a.upstream.Forward(ctx, nq, false, false) + if err != nil { + return "", err + } + jrResp, err := resp.JsonRpcResponse() + if err != nil { + return "", err + } + if jrResp.Error != nil { + return "", jrResp.Error + } + subID := strings.Trim(string(jrResp.GetResultBytes()), "\"") + if subID == "" { + return "", fmt.Errorf("upstream returned empty subscription ID") + } + return subID, nil +} + +// sendUnsubscribe is best-effort; errors are logged by the WS client and +// not surfaced (cleanup paths shouldn't fail on upstream errors). +func (a *Adapter) sendUnsubscribe(ctx context.Context, subID string) { + body, err := buildJSONRPCBody(methodEthUnsubscribe, []interface{}{subID}) + if err != nil { + return + } + nq := common.NewNormalizedRequest(body) + _, _ = a.upstream.Forward(ctx, nq, false, false) +} + +// buildJSONRPCBody marshals a JSON-RPC request with a unique internal +// ID. Kept self-contained here rather than imported from erpc/ to +// avoid a circular dependency (erpc depends on the adapter for wiring). +func buildJSONRPCBody(method string, params interface{}) ([]byte, error) { + id := internalReqIDCounter.Add(1) + internalReqIDOffset + return common.SonicCfg.Marshal(map[string]interface{}{ + "jsonrpc": "2.0", + "id": id, + "method": method, + "params": params, + }) +} + +// stripFromBlockZero returns a copy of params with fromBlock removed from +// any filter object whose fromBlock equals "0x0" or "0". toBlock is left +// alone. Other fromBlock values (including "latest", "finalized", or a +// specific hex block number) are not touched. The input slice is not +// mutated — filter maps are shallow-copied so the caller's params remain +// stable for paramsHash computation and resubscribe. +func stripFromBlockZero(params []interface{}) ([]interface{}, bool) { + out := make([]interface{}, len(params)) + changed := false + for i, p := range params { + f, ok := p.(map[string]interface{}) + if !ok { + out[i] = p + continue + } + fb, hasFrom := f["fromBlock"] + if !hasFrom { + out[i] = f + continue + } + s, isStr := fb.(string) + if !isStr || !isZeroBlockRef(s) { + out[i] = f + continue + } + clean := make(map[string]interface{}, len(f)) + for k, v := range f { + if k == "fromBlock" { + continue + } + clean[k] = v + } + out[i] = clean + changed = true + } + return out, changed +} + +// isZeroBlockRef reports whether s parses to zero in any form eth clients +// typically emit ("0", "0x0", "0x00", …). strconv.ParseInt with base 0 +// auto-detects the 0x prefix for hex. +func isZeroBlockRef(s string) bool { + n, err := strconv.ParseInt(strings.TrimSpace(s), 0, 64) + return err == nil && n == 0 +} diff --git a/indexer/adapters/wsupstream/adapter_test.go b/indexer/adapters/wsupstream/adapter_test.go new file mode 100644 index 000000000..7c92b006f --- /dev/null +++ b/indexer/adapters/wsupstream/adapter_test.go @@ -0,0 +1,90 @@ +package wsupstream + +import ( + "testing" + + "github.com/stretchr/testify/assert" +) + +func TestIsZeroBlockRef(t *testing.T) { + cases := []struct { + in string + want bool + }{ + {"0", true}, + {"0x0", true}, + {"0X0", true}, + {"0x00", true}, + {"0x0000", true}, + {" 0x0 ", true}, + {"0x1", false}, + {"0x100", false}, + {"latest", false}, + {"finalized", false}, + {"", false}, + {"0x", false}, + } + for _, c := range cases { + t.Run(c.in, func(t *testing.T) { + assert.Equal(t, c.want, isZeroBlockRef(c.in)) + }) + } +} + +func TestStripFromBlockZero(t *testing.T) { + t.Run("strips fromBlock:0x0 and keeps toBlock", func(t *testing.T) { + params := []interface{}{ + map[string]interface{}{ + "address": "0xabc", + "topics": []string{"0xdef"}, + "fromBlock": "0x0", + "toBlock": "latest", + }, + } + out, changed := stripFromBlockZero(params) + assert.True(t, changed) + m := out[0].(map[string]interface{}) + _, hasFrom := m["fromBlock"] + assert.False(t, hasFrom) + assert.Equal(t, "latest", m["toBlock"]) + assert.Equal(t, "0xabc", m["address"]) + }) + + t.Run("non-zero fromBlock passes through untouched", func(t *testing.T) { + params := []interface{}{ + map[string]interface{}{"fromBlock": "0x100", "toBlock": "latest"}, + } + out, changed := stripFromBlockZero(params) + assert.False(t, changed) + m := out[0].(map[string]interface{}) + assert.Equal(t, "0x100", m["fromBlock"]) + }) + + t.Run("no fromBlock returns changed=false", func(t *testing.T) { + params := []interface{}{ + map[string]interface{}{"address": "0xabc", "topics": []string{"0xdef"}}, + } + out, changed := stripFromBlockZero(params) + assert.False(t, changed) + assert.Equal(t, params, out) + }) + + t.Run("non-map params are untouched", func(t *testing.T) { + params := []interface{}{"logs", "notAMap", 42} + out, changed := stripFromBlockZero(params) + assert.False(t, changed) + assert.Equal(t, params, out) + }) + + t.Run("input is not mutated", func(t *testing.T) { + original := []interface{}{ + map[string]interface{}{"fromBlock": "0x0", "address": "0xabc"}, + } + _, changed := stripFromBlockZero(original) + assert.True(t, changed) + // Original still has fromBlock. + orig := original[0].(map[string]interface{}) + _, hasFrom := orig["fromBlock"] + assert.True(t, hasFrom, "stripFromBlockZero must not mutate its input") + }) +} diff --git a/indexer/constants.go b/indexer/constants.go new file mode 100644 index 000000000..f6cd0489e --- /dev/null +++ b/indexer/constants.go @@ -0,0 +1,22 @@ +// Package indexer hosts the transport-neutral event-stream core: dedup, +// per-source subscription bookkeeping, canonical event shape, and (in a +// follow-up) reorg-aware canonical-chain tracking. Packages in this tree +// MUST NOT import erpc, clients, or any transport-specific library — that +// invariant is what keeps ingress/egress implementations swappable. +package indexer + +// Supported eth_subscribe subscription types. These are surfaced by the +// indexer's public API rather than being transport-specific (JSON-RPC +// method names live in the WS adapter), so a future Kafka / gRPC egress +// can describe events in the same vocabulary. +const ( + SubTypeNewHeads = "newHeads" + SubTypeLogs = "logs" + SubTypeNewPendingTransactions = "newPendingTransactions" +) + +// DefaultDedupWindowSize bounds the per-filter seen-set used to drop +// duplicate notifications delivered by multiple sources. The value is +// sized to cover realistic burst windows on fast chains without keeping +// an unbounded memory footprint per filter. +const DefaultDedupWindowSize = 8192 diff --git a/indexer/dedup.go b/indexer/dedup.go new file mode 100644 index 000000000..8278842f1 --- /dev/null +++ b/indexer/dedup.go @@ -0,0 +1,91 @@ +package indexer + +import ( + "encoding/json" + "fmt" + "sync" + + "github.com/erpc/erpc/common" +) + +// DedupKeyForFilter returns a stable identifier for a filter-subscription +// notification payload, or "" when the payload cannot be parsed (caller +// should then fan out without deduping, since we can't prove it's a dupe). +// +// - logs: blockHash + txHash + logIndex + removed flag (the removed +// flag is part of the key so a reorg-out log isn't collapsed with its +// matching reorg-in log — both are delivered to clients). +// - newPendingTransactions: the tx hash, whether the upstream returned +// a raw string or an object with a .hash field. +func DedupKeyForFilter(subType string, result json.RawMessage) string { + switch subType { + case SubTypeLogs: + var log struct { + BlockHash string `json:"blockHash"` + TxHash string `json:"transactionHash"` + LogIndex string `json:"logIndex"` + Removed bool `json:"removed"` + } + if err := common.SonicCfg.Unmarshal(result, &log); err != nil { + return "" + } + return fmt.Sprintf("%s:%s:%s:%t", log.BlockHash, log.TxHash, log.LogIndex, log.Removed) + case SubTypeNewPendingTransactions: + var asString string + if err := common.SonicCfg.Unmarshal(result, &asString); err == nil && asString != "" { + return asString + } + var asObj struct { + Hash string `json:"hash"` + } + if err := common.SonicCfg.Unmarshal(result, &asObj); err == nil { + return asObj.Hash + } + } + return "" +} + +// DedupWindow is a bounded FIFO seen-set: keys added past the window's +// capacity evict the oldest entries. It is safe for concurrent use; callers +// typically hold one per (network, subType, paramsHash) fan-out group. +type DedupWindow struct { + size int + + mu sync.Mutex + seen map[string]struct{} + order []string +} + +// NewDedupWindow returns a DedupWindow sized to hold up to `size` keys +// before the oldest entries are evicted. Pass 0 to use DefaultDedupWindowSize. +func NewDedupWindow(size int) *DedupWindow { + if size <= 0 { + size = DefaultDedupWindowSize + } + return &DedupWindow{ + size: size, + seen: make(map[string]struct{}, size), + order: make([]string, 0, size), + } +} + +// Mark records a key and returns true if it's newly seen. Returns false if +// the key is already in the window (caller should drop the duplicate). +func (w *DedupWindow) Mark(key string) bool { + w.mu.Lock() + defer w.mu.Unlock() + if _, ok := w.seen[key]; ok { + return false + } + w.seen[key] = struct{}{} + w.order = append(w.order, key) + + if len(w.order) > w.size { + evict := len(w.order) - w.size + for _, old := range w.order[:evict] { + delete(w.seen, old) + } + w.order = w.order[evict:] + } + return true +} diff --git a/indexer/dedup_test.go b/indexer/dedup_test.go new file mode 100644 index 000000000..b90ffc8ee --- /dev/null +++ b/indexer/dedup_test.go @@ -0,0 +1,84 @@ +package indexer + +import ( + "encoding/json" + "testing" +) + +func TestDedupWindow_MarkNewKey(t *testing.T) { + w := NewDedupWindow(4) + if !w.Mark("a") { + t.Fatalf("first Mark of a new key must return true") + } + if w.Mark("a") { + t.Fatalf("second Mark of the same key must return false") + } +} + +func TestDedupWindow_EvictsOldestPastCapacity(t *testing.T) { + w := NewDedupWindow(3) + for _, k := range []string{"a", "b", "c"} { + if !w.Mark(k) { + t.Fatalf("Mark(%q) must be true while window has room", k) + } + } + // Window: a, b, c. Adding "d" evicts "a"; window becomes b, c, d. + if !w.Mark("d") { + t.Fatalf("Mark(d) must be true") + } + // "a" was evicted and is therefore newly markable. + if !w.Mark("a") { + t.Fatalf("a should be evicted and therefore newly markable") + } + // "c" has not been evicted yet — still a dupe. + if w.Mark("c") { + t.Fatalf("c should still be in-window") + } +} + +func TestDedupWindow_ZeroSizeUsesDefault(t *testing.T) { + w := NewDedupWindow(0) + if w.size != DefaultDedupWindowSize { + t.Fatalf("size with 0 arg should default to %d, got %d", DefaultDedupWindowSize, w.size) + } +} + +func TestDedupKeyForFilter_Logs(t *testing.T) { + payload := json.RawMessage(`{"blockHash":"0xabc","transactionHash":"0xdef","logIndex":"0x1","removed":false}`) + got := DedupKeyForFilter(SubTypeLogs, payload) + want := "0xabc:0xdef:0x1:false" + if got != want { + t.Fatalf("log dedup key: got %q want %q", got, want) + } +} + +func TestDedupKeyForFilter_LogsRemovedFlagIsPartOfKey(t *testing.T) { + // A reorged-out log must not collapse with its matching reorged-in log; + // both are delivered to clients so the removed flag must be in the key. + in := DedupKeyForFilter(SubTypeLogs, json.RawMessage(`{"blockHash":"0xabc","transactionHash":"0xdef","logIndex":"0x1","removed":false}`)) + out := DedupKeyForFilter(SubTypeLogs, json.RawMessage(`{"blockHash":"0xabc","transactionHash":"0xdef","logIndex":"0x1","removed":true}`)) + if in == out { + t.Fatalf("removed=true / removed=false must produce distinct dedup keys") + } +} + +func TestDedupKeyForFilter_PendingTxAsString(t *testing.T) { + got := DedupKeyForFilter(SubTypeNewPendingTransactions, json.RawMessage(`"0xbeef"`)) + if got != "0xbeef" { + t.Fatalf("pendingTx string key: got %q", got) + } +} + +func TestDedupKeyForFilter_PendingTxAsObject(t *testing.T) { + got := DedupKeyForFilter(SubTypeNewPendingTransactions, json.RawMessage(`{"hash":"0xbeef"}`)) + if got != "0xbeef" { + t.Fatalf("pendingTx object key: got %q", got) + } +} + +func TestDedupKeyForFilter_UnknownSubTypeReturnsEmpty(t *testing.T) { + got := DedupKeyForFilter("unknown", json.RawMessage(`{}`)) + if got != "" { + t.Fatalf("unknown subType should return empty key, got %q", got) + } +} diff --git a/indexer/egress.go b/indexer/egress.go new file mode 100644 index 000000000..4767a98a5 --- /dev/null +++ b/indexer/egress.go @@ -0,0 +1,27 @@ +package indexer + +// EventEgress receives IndexedEvents that survive dedup + lifecycle tagging. +// Adapters own their own buffering, ordering guarantees, and drop policy — +// the indexer passes events through synchronously and moves on. +// +// Today's WS-client adapter uses oldest-drop (slow client discards the +// oldest queued notification to keep up). A future Kafka producer would +// block on backpressure; a webhook adapter would queue-and-retry. Keeping +// the policy per-adapter is deliberate. +type EventEgress interface { + // Name is a stable identifier for logs/metrics ("ws:client:", + // "kafka:topic:", …). + Name() string + // InterestedIn lets the indexer skip the Deliver call entirely for + // events this egress doesn't route. Return true if the event matches + // some active subscription on this egress; false to drop it. + // + // Called on the indexer's hot path — must not allocate, lock, or + // block. Typical implementation is a sync.Map lookup on a pre-built + // key. + InterestedIn(kind EventKind, networkId, filterHash string) bool + // Deliver hands the event to the egress. The indexer treats Deliver + // as non-blocking; a slow Deliver hurts *this* egress only (via the + // adapter's own buffer), never other egresses or the ingress. + Deliver(ev IndexedEvent) +} diff --git a/indexer/errors.go b/indexer/errors.go new file mode 100644 index 000000000..004dffcb8 --- /dev/null +++ b/indexer/errors.go @@ -0,0 +1,11 @@ +package indexer + +import "fmt" + +// errNetworkNotRegistered is returned by indexer calls that reference a +// networkId never passed to RegisterNetwork. It's a programmer error at +// wiring time — wrapped rather than a sentinel because callers have no +// useful action beyond failing the startup path. +func errNetworkNotRegistered(networkId string) error { + return fmt.Errorf("indexer: network %q not registered", networkId) +} diff --git a/indexer/event.go b/indexer/event.go new file mode 100644 index 000000000..26048b010 --- /dev/null +++ b/indexer/event.go @@ -0,0 +1,126 @@ +package indexer + +import ( + "encoding/json" + "time" +) + +// EventKind is a tagged-union discriminant for events flowing through the +// indexer pipeline. Separate types for each kind would force sum-type +// dispatch at every stage (dedup, lifecycle tagging, fan-out) — the +// discriminant lets every stage operate on a uniform value, which is +// worth the modest loss of compile-time guarantees on per-kind fields. +type EventKind uint8 + +const ( + KindUnknown EventKind = iota + // KindNewHead: a canonical block header. One-shot per new block. + KindNewHead + // KindLog: an event log delivered by a filter subscription. + KindLog + // KindPendingTx: a newly-seen pending transaction hash or object. + KindPendingTx + // KindReorg: a synthetic event emitted by the indexer when the + // canonical chain diverges. The payload describes the evicted segment. + KindReorg +) + +// String returns the lowercase name matching Ethereum's eth_subscribe +// surface (or "reorg"/"unknown" for synthetic/unknown kinds). +func (k EventKind) String() string { + switch k { + case KindNewHead: + return "newHeads" + case KindLog: + return "logs" + case KindPendingTx: + return "newPendingTransactions" + case KindReorg: + return "reorg" + default: + return "unknown" + } +} + +// Lifecycle records whether an event sits above or below the network's +// finality boundary at the moment it's emitted. A consumer that only +// cares about finalized data can filter on `Lifecycle == LifeFinalized`. +type Lifecycle uint8 + +const ( + // LifeSoft: event has not yet passed the network's finality depth. + // May be reorged out. This is the default for head-tracking. + LifeSoft Lifecycle = iota + // LifeFinalized: event is at or below the finalized block. Treated + // as immutable by consumers. + LifeFinalized +) + +// String renders Lifecycle for log/metric labels. +func (l Lifecycle) String() string { + switch l { + case LifeFinalized: + return "finalized" + default: + return "soft" + } +} + +// BlockRef identifies a block on a network. Zero-valued for KindPendingTx +// (pending txs don't carry a block reference until mined). +type BlockRef struct { + Number int64 + Hash string + ParentHash string +} + +// Zero reports whether the BlockRef is the zero value — i.e. no block +// reference is attached (pending tx). +func (b BlockRef) Zero() bool { + return b.Number == 0 && b.Hash == "" && b.ParentHash == "" +} + +// StreamEvent is what an ingress emits. It is pre-dedup, pre-lifecycle, +// and may duplicate events delivered by sibling sources covering the +// same network. The Indexer converts StreamEvents → IndexedEvents. +type StreamEvent struct { + Kind EventKind + NetworkId string + // SourceId names the ingress that produced this event ("ws:", + // "kafka:", …). Used for per-source bookkeeping inside the + // indexer (e.g. state-poller updates are per upstream) but never + // surfaced to egresses. + SourceId string + Block BlockRef + // FilterHash identifies the filter subscription this event belongs to + // for Kind{Log,PendingTx}. Empty for KindNewHead. Matches the hash + // returned by BuildParamsKey when clients subscribe. + FilterHash string + // Payload is the upstream-provided notification result, verbatim JSON. + // Adapters must not re-marshal — both to preserve upstream formatting + // quirks and to let non-JSON egresses (protobuf, flatbuf) decode once + // and cache beside the event. + Payload json.RawMessage + ObservedAt time.Time +} + +// IndexedEvent is what the Indexer emits to every registered egress after +// dedup, sequencing, and lifecycle tagging. One IndexedEvent per observed +// StreamEvent (minus duplicates), with at most one synthetic KindReorg +// injected between the old and new canonical heads. +type IndexedEvent struct { + StreamEvent + + // Seq is a monotonically-increasing per (NetworkId, Kind, FilterHash) + // counter, stable for the life of an indexer instance. Egresses that + // need resumption (Kafka producer, webhooks with retry) key off Seq. + // It is NOT globally unique across indexer restarts. + Seq uint64 + // Lifecycle is computed on the way out from the NetworkHandle's + // finality depth. See LifeSoft / LifeFinalized. + Lifecycle Lifecycle + // Removed is true when the upstream signalled the event was reorged + // out (for logs) or when the indexer's canonical chain tracker + // evicted a previously-emitted block. + Removed bool +} diff --git a/indexer/filters.go b/indexer/filters.go new file mode 100644 index 000000000..4a9844dde --- /dev/null +++ b/indexer/filters.go @@ -0,0 +1,64 @@ +package indexer + +import ( + "crypto/rand" + "encoding/hex" + "fmt" + "hash/fnv" + + "github.com/erpc/erpc/common" +) + +// BuildParamsKey returns a stable short hash of eth_subscribe params. The +// same (subType, params) tuple always hashes to the same key across +// processes, so the key doubles as a fan-out lookup across sources and +// across pods. +func BuildParamsKey(params []interface{}) string { + data, err := common.SonicCfg.Marshal(params) + if err != nil { + // Fall back to Go's default formatting. Worse than JSON for + // cross-process stability, but only hit on marshal errors that + // would already have broken the upstream subscribe call. + return fmt.Sprintf("%v", params) + } + h := fnv.New64a() + _, _ = h.Write(data) + return fmt.Sprintf("%x", h.Sum64()) +} + +// ExtractSubscriptionType pulls the subscription type (e.g. "newHeads", +// "logs") from the first element of an eth_subscribe params array. +// Returns "" when params is empty or the first element isn't a string. +func ExtractSubscriptionType(params []interface{}) string { + if len(params) == 0 { + return "" + } + st, _ := params[0].(string) + return st +} + +// ExtractClientSubID pulls the client-facing subscription ID from an +// eth_unsubscribe params array. Returns a typed "subscription not found" +// error when the ID is missing or not a string. +func ExtractClientSubID(params []interface{}) (string, error) { + if len(params) == 0 { + return "", common.NewErrSubscriptionNotFound("") + } + id, ok := params[0].(string) + if !ok { + return "", common.NewErrSubscriptionNotFound(fmt.Sprintf("%v", params[0])) + } + return id, nil +} + +// GenerateClientSubID creates a cryptographically random subscription ID +// formatted as "0x" + 32 hex characters (16 random bytes). The format +// matches what Ethereum clients emit so consumers can treat our IDs as +// opaque strings. +func GenerateClientSubID() (string, error) { + b := make([]byte, 16) + if _, err := rand.Read(b); err != nil { + return "", err + } + return "0x" + hex.EncodeToString(b), nil +} diff --git a/indexer/filters_test.go b/indexer/filters_test.go new file mode 100644 index 000000000..bfcfd7fc5 --- /dev/null +++ b/indexer/filters_test.go @@ -0,0 +1,86 @@ +package indexer + +import ( + "strings" + "testing" +) + +func TestBuildParamsKey_StableAcrossCalls(t *testing.T) { + a := BuildParamsKey([]interface{}{"logs", map[string]interface{}{"topics": []string{"0x1"}}}) + b := BuildParamsKey([]interface{}{"logs", map[string]interface{}{"topics": []string{"0x1"}}}) + if a != b { + t.Fatalf("same params must hash identically, got %q vs %q", a, b) + } +} + +func TestBuildParamsKey_DifferentParamsDiffer(t *testing.T) { + a := BuildParamsKey([]interface{}{"logs", map[string]interface{}{"topics": []string{"0x1"}}}) + b := BuildParamsKey([]interface{}{"logs", map[string]interface{}{"topics": []string{"0x2"}}}) + if a == b { + t.Fatalf("different params must produce different keys") + } +} + +func TestExtractSubscriptionType(t *testing.T) { + cases := []struct { + name string + params []interface{} + want string + }{ + {"newHeads", []interface{}{"newHeads"}, "newHeads"}, + {"logs with filter", []interface{}{"logs", map[string]interface{}{}}, "logs"}, + {"empty params", []interface{}{}, ""}, + {"non-string first", []interface{}{123}, ""}, + } + for _, c := range cases { + if got := ExtractSubscriptionType(c.params); got != c.want { + t.Errorf("%s: got %q want %q", c.name, got, c.want) + } + } +} + +func TestExtractClientSubID_ValidString(t *testing.T) { + got, err := ExtractClientSubID([]interface{}{"0xabc"}) + if err != nil { + t.Fatalf("unexpected err: %v", err) + } + if got != "0xabc" { + t.Fatalf("got %q want 0xabc", got) + } +} + +func TestExtractClientSubID_EmptyParams(t *testing.T) { + _, err := ExtractClientSubID(nil) + if err == nil { + t.Fatalf("expected error on empty params") + } +} + +func TestExtractClientSubID_NonString(t *testing.T) { + _, err := ExtractClientSubID([]interface{}{123}) + if err == nil { + t.Fatalf("expected error on non-string param") + } +} + +func TestGenerateClientSubID_Shape(t *testing.T) { + id, err := GenerateClientSubID() + if err != nil { + t.Fatalf("unexpected err: %v", err) + } + if !strings.HasPrefix(id, "0x") { + t.Fatalf("id must start with 0x: %q", id) + } + // 2 for "0x" + 32 hex chars (16 bytes * 2) + if len(id) != 34 { + t.Fatalf("id must be 34 chars, got %d (%q)", len(id), id) + } +} + +func TestGenerateClientSubID_Unique(t *testing.T) { + a, _ := GenerateClientSubID() + b, _ := GenerateClientSubID() + if a == b { + t.Fatalf("two consecutive IDs must differ") + } +} diff --git a/indexer/indexer.go b/indexer/indexer.go new file mode 100644 index 000000000..a686fecad --- /dev/null +++ b/indexer/indexer.go @@ -0,0 +1,541 @@ +package indexer + +import ( + "context" + "errors" + "sync" + "sync/atomic" + "time" + + "github.com/rs/zerolog" +) + +// Options configures Indexer construction. Defaults (zero values) are +// sensible for all fields; override only for tuning. +type Options struct { + // DedupWindowSize is the per-filter seen-set capacity. 0 = default. + DedupWindowSize int + // CanonicalChainDepth is the ring-buffer size of the per-network + // canonical-chain tracker. 0 = default. Controls how deep a reorg + // we can fully resolve. + CanonicalChainDepth int + // Now is the clock used for event timestamps. Tests inject a fake + // clock; nil falls back to time.Now. + Now func() time.Time +} + +// Indexer is the transport-neutral core: ingresses push StreamEvents via +// Sink.Ingest; the indexer dedupes, sequences, and lifecycle-tags them, +// then fans out to every interested egress. +// +// The Indexer itself implements Sink so adapters can call indexer.Ingest +// directly — the choice is deliberate: synchronous Ingest keeps +// "update-before-dedup" ordering guarantees from the ingress goroutine +// without goroutine-per-event channel gymnastics. +type Indexer struct { + logger *zerolog.Logger + opts Options + + networks sync.Map // networkId -> *networkState + egresses sync.Map // egress Name() -> EventEgress + seq atomic.Uint64 +} + +// networkState holds the indexer's per-network bookkeeping: the +// headMarker packages the most-recent head seen for a network so +// networkState can store (num, hash) atomically. Treated as immutable +// once published via lastHead.Store / CompareAndSwap; readers load +// and may see nil before the first head arrives. +type headMarker struct { + num int64 + hash string +} + +// NetworkHandle for finality lookups, the ingresses feeding it, and the +// dedup windows for each active filter (plus the newHeads window). +type networkState struct { + handle NetworkHandle + + ingressMu sync.RWMutex + ingresses map[string]EventIngress // Name() -> ingress + // selector, when non-nil, narrows per-filter fan-out to a chosen + // subset of ingresses (defaults first, fallbacks on total failure). + // Nil means "treat every registered ingress as a default." + selector IngressSelector + + // newHeads dedup: most-recently-delivered (blockNumber, blockHash) + // packed into a single pointer so the check+advance is a single + // atomic CAS — separate atomics on num and hash would leave readers + // able to see num updated before hash, and producers able to both + // pass the "load, check, store" sequence with the same stale num. + // We saw the latter in prod against evm:1101 where four upstream WS + // sources delivered the same head within ~1ms and both raced past + // the dedup. Fallback DedupWindow still catches the rare + // "older-but-not-newest" case (out-of-order delivery on a reorg + // boundary). + lastHead atomic.Pointer[headMarker] + headFallback *DedupWindow + + // Per-filter dedup windows: filterHash -> *DedupWindow. + filterMu sync.RWMutex + filterDedup map[string]*DedupWindow + filterRefcnt map[string]int // clients × filterHash; triggers RemoveFilter at 0 + + // chain records the last N canonical heads for this network plus + // an index of delivered logs keyed by blockHash. Drives reorg + // detection and Removed=true re-emission. + chain *canonicalChain +} + +// New returns an empty Indexer. Networks must be registered via +// RegisterNetwork before any ingress can push events. +func New(logger *zerolog.Logger, opts Options) *Indexer { + if opts.Now == nil { + opts.Now = time.Now + } + if opts.DedupWindowSize <= 0 { + opts.DedupWindowSize = DefaultDedupWindowSize + } + return &Indexer{ + logger: logger, + opts: opts, + } +} + +// RegisterNetwork installs a NetworkHandle for the indexer to consult when +// tagging lifecycle and routing per-source state-poller updates. Safe to +// call more than once for the same network (idempotent on handle identity). +func (i *Indexer) RegisterNetwork(nw NetworkHandle) *networkState { + if ns, ok := i.networks.Load(nw.Id()); ok { + return ns.(*networkState) + } + ns := &networkState{ + handle: nw, + ingresses: make(map[string]EventIngress), + headFallback: NewDedupWindow(i.opts.DedupWindowSize), + filterDedup: make(map[string]*DedupWindow), + filterRefcnt: make(map[string]int), + chain: newCanonicalChain(i.opts.CanonicalChainDepth), + } + actual, _ := i.networks.LoadOrStore(nw.Id(), ns) + return actual.(*networkState) +} + +// AddIngress starts an ingress and hands it the NetworkHandle registered +// for networkId. The ingress pushes events at the indexer (which +// implements Sink). Returns an error if the network has not been +// registered yet. +func (i *Indexer) AddIngress(ctx context.Context, networkId string, ing EventIngress) error { + nsRaw, ok := i.networks.Load(networkId) + if !ok { + return errNetworkNotRegistered(networkId) + } + ns := nsRaw.(*networkState) + ns.ingressMu.Lock() + ns.ingresses[ing.Name()] = ing + ns.ingressMu.Unlock() + return ing.Start(ctx, ns.handle, i) +} + +// Attach registers an egress. The returned detach function removes the +// egress — callers that care about cleanup (client-connection closes) +// must invoke it. +func (i *Indexer) Attach(eg EventEgress) (detach func()) { + i.egresses.Store(eg.Name(), eg) + return func() { i.egresses.Delete(eg.Name()) } +} + +// RegisterNetworkSelector installs a per-network IngressSelector that +// EnsureFilter consults when deciding which ingresses to subscribe. Passing +// nil restores the legacy "fan out to every registered ingress" behaviour. +// The network must have been registered first. +func (i *Indexer) RegisterNetworkSelector(networkId string, sel IngressSelector) { + nsRaw, ok := i.networks.Load(networkId) + if !ok { + return + } + ns := nsRaw.(*networkState) + ns.ingressMu.Lock() + ns.selector = sel + ns.ingressMu.Unlock() +} + +// EnsureFilter subscribes a filter on this network's ingresses and tracks +// a per-filter refcount so ReleaseFilter can decide when to tear it down +// upstream. Selection rules: +// +// - Ingresses named in the selector's default tier are grouped as +// pooled defaults; those named in the fallback tier are pooled +// fallbacks. +// - Ingresses the selector does not name in either tier are treated as +// "always include" — they receive EnsureFilter unconditionally. This +// preserves the original fan-out for standalone transports (Kafka, +// gRPC streams, …) that a WS-only selector doesn't know about. +// - With no selector registered, every ingress is an "always include", +// matching the pre-selector behaviour. +// +// Success semantics: the subscription is considered established — and +// EnsureFilter returns (paramsHash, nil) — as soon as any ingress +// (always-include or pooled default) successfully subscribed. Fallbacks +// are only attempted when every always-include and every default failed. +// An error is returned only if every attempted ingress failed; per-ingress +// errors are otherwise logged and not propagated. +func (i *Indexer) EnsureFilter(ctx context.Context, networkId, subType string, params []interface{}) (paramsHash string, err error) { + nsRaw, ok := i.networks.Load(networkId) + if !ok { + return "", errNetworkNotRegistered(networkId) + } + ns := nsRaw.(*networkState) + paramsHash = BuildParamsKey(params) + + ns.filterMu.Lock() + if _, ok := ns.filterDedup[paramsHash]; !ok { + ns.filterDedup[paramsHash] = NewDedupWindow(i.opts.DedupWindowSize) + } + ns.filterRefcnt[paramsHash]++ + first := ns.filterRefcnt[paramsHash] == 1 + ns.filterMu.Unlock() + + if !first { + return paramsHash, nil + } + + // Snapshot the ingress set and selector under rlock, then do the + // potentially slow per-ingress RPC calls without holding any lock. + ns.ingressMu.RLock() + ings := make(map[string]EventIngress, len(ns.ingresses)) + for name, ing := range ns.ingresses { + ings[name] = ing + } + sel := ns.selector + ns.ingressMu.RUnlock() + + always, defaults, fallbacks := partitionIngresses(sel, ings, networkId, subType, params) + + // Nothing to subscribe on — bootstrap ordering or a selector returning + // nothing and no standalone ingresses. Preserve the dedup window so + // any ingress attached later still delivers into it. + if len(always) == 0 && len(defaults) == 0 && len(fallbacks) == 0 { + return paramsHash, nil + } + + var ( + successCount int + errs []error + ) + tryOne := func(ing EventIngress) { + if err := ing.EnsureFilter(ctx, subType, paramsHash, params); err != nil { + errs = append(errs, err) + i.logger.Warn().Err(err).Str("ingress", ing.Name()).Str("networkId", networkId). + Str("subType", subType).Str("paramsHash", paramsHash). + Msg("ingress EnsureFilter failed") + return + } + successCount++ + } + + for _, ing := range always { + tryOne(ing) + } + for _, ing := range defaults { + tryOne(ing) + } + if successCount == 0 { + for _, ing := range fallbacks { + tryOne(ing) + } + } + + if successCount > 0 { + return paramsHash, nil + } + + // Every chosen ingress failed. Roll the refcount this caller added + // back so the next EnsureFilter attempt starts a fresh subscribe; + // only clean the dedup window when no concurrent subscriber is + // waiting on this filterHash. + ns.filterMu.Lock() + ns.filterRefcnt[paramsHash]-- + if ns.filterRefcnt[paramsHash] <= 0 { + delete(ns.filterRefcnt, paramsHash) + delete(ns.filterDedup, paramsHash) + } + ns.filterMu.Unlock() + + return paramsHash, errors.Join(errs...) +} + +// partitionIngresses splits the registered ingress set into three groups +// based on the network's selector: always-include (ingresses the selector +// does not mention at all), pooled defaults, and pooled fallbacks. Names +// returned by the selector that do not correspond to a registered ingress +// are silently dropped. Without a selector, every ingress is treated as +// always-include. +func partitionIngresses(sel IngressSelector, ings map[string]EventIngress, networkId, subType string, params []interface{}) (always, defaults, fallbacks []EventIngress) { + if sel == nil { + always = make([]EventIngress, 0, len(ings)) + for _, ing := range ings { + always = append(always, ing) + } + return always, nil, nil + } + dNames, fNames := sel.Select(networkId, subType, params) + named := make(map[string]struct{}, len(dNames)+len(fNames)) + pick := func(names []string) []EventIngress { + out := make([]EventIngress, 0, len(names)) + for _, n := range names { + if _, dup := named[n]; dup { + continue + } + named[n] = struct{}{} + if ing, ok := ings[n]; ok { + out = append(out, ing) + } + } + return out + } + defaults = pick(dNames) + fallbacks = pick(fNames) + for name, ing := range ings { + if _, claimed := named[name]; claimed { + continue + } + always = append(always, ing) + } + return always, defaults, fallbacks +} + +// ReleaseFilter decrements the refcount on the filter and, when it hits +// zero, tears the subscription down on every registered ingress. Callers +// supply paramsHash (returned by EnsureFilter) rather than params. +// Ingresses that never received EnsureFilter for this paramsHash are +// expected to no-op on RemoveFilter. +func (i *Indexer) ReleaseFilter(ctx context.Context, networkId, subType, paramsHash string) { + nsRaw, ok := i.networks.Load(networkId) + if !ok { + return + } + ns := nsRaw.(*networkState) + + ns.filterMu.Lock() + ns.filterRefcnt[paramsHash]-- + remove := ns.filterRefcnt[paramsHash] <= 0 + if remove { + delete(ns.filterRefcnt, paramsHash) + delete(ns.filterDedup, paramsHash) + } + ns.filterMu.Unlock() + + if !remove { + return + } + ns.ingressMu.RLock() + ings := make([]EventIngress, 0, len(ns.ingresses)) + for _, ing := range ns.ingresses { + ings = append(ings, ing) + } + ns.ingressMu.RUnlock() + + for _, ing := range ings { + if err := ing.RemoveFilter(ctx, subType, paramsHash); err != nil { + i.logger.Warn().Err(err).Str("ingress", ing.Name()).Str("networkId", networkId). + Str("subType", subType).Str("paramsHash", paramsHash). + Msg("ingress RemoveFilter failed") + } + } +} + +// Ingest is the hot-path entry point for ingress adapters. It updates +// per-source state (via NetworkHandle.SuggestLatestBlock for headed +// events), dedupes, lifecycle-tags, emits reorg invalidations, and +// fans out to every interested egress. +func (i *Indexer) Ingest(ev StreamEvent) { + nsRaw, ok := i.networks.Load(ev.NetworkId) + if !ok { + return + } + ns := nsRaw.(*networkState) + + // State-poller update before dedup. Every observation feeds the + // per-upstream latest-block tracker, even if the head is a dup at + // the indexer level — otherwise a lagging source's state poller + // stalls on the first dup. + if ev.Kind == KindNewHead && !ev.Block.Zero() && ev.SourceId != "" { + ns.handle.SuggestLatestBlock(ev.SourceId, ev.Block.Number) + } + + // Dedup. + if !i.dedupe(ns, &ev) { + return + } + + // Detect and emit reorg invalidations BEFORE delivering the new + // head. Consumers see: (removed logs) → reorg summary → new head. + if ev.Kind == KindNewHead && !ev.Block.Zero() { + if evicted := ns.chain.observeHead(ev.Block); len(evicted) > 0 { + i.emitReorgInvalidations(ns, evicted) + } + } + + // Index logs so a later reorg can re-emit them with Removed=true. + if ev.Kind == KindLog { + // Best-effort: if the payload has a parseable blockHash, record + // the log against it. Logs without a blockHash can't be + // invalidated — safe to drop from the index. + if blockRef := parseLogBlockRef(ev.Payload); blockRef.Hash != "" { + ns.chain.indexLog(loggedLog{ + filterHash: ev.FilterHash, + networkID: ev.NetworkId, + sourceID: ev.SourceId, + block: blockRef, + payload: ev.Payload, + }) + } + } + + // Build the indexed event. + out := IndexedEvent{ + StreamEvent: ev, + Seq: i.seq.Add(1), + Lifecycle: i.classify(ns, ev), + } + // Upstream-asserted removed flag. Chain-tracker-driven removal goes + // through emitReorgInvalidations; this handles the case where the + // upstream itself has already classified the log as reorged-out. + if ev.Kind == KindLog { + out.Removed = logRemoved(ev.Payload) + } + + i.fanOut(out) +} + +// emitReorgInvalidations fans out a KindReorg summary followed by +// Removed=true copies of every log indexed against the evicted blocks. +// Called under ingest's caller goroutine; Deliver is non-blocking by +// contract so this stays cheap even under deep reorgs. +func (i *Indexer) emitReorgInvalidations(ns *networkState, evicted []BlockRef) { + if len(evicted) == 0 { + return + } + networkID := ns.handle.Id() + + // One KindReorg summary carrying the evicted BlockRefs. + reorgPayload, err := reorgSummaryPayload(evicted) + if err == nil { + i.fanOut(IndexedEvent{ + StreamEvent: StreamEvent{ + Kind: KindReorg, + NetworkId: networkID, + Payload: reorgPayload, + }, + Seq: i.seq.Add(1), + Lifecycle: LifeSoft, + }) + } + + // One Removed=true emission per indexed log in each evicted block. + for _, blk := range evicted { + logs := ns.chain.drainLogsFor(blk.Hash) + for _, lg := range logs { + i.fanOut(IndexedEvent{ + StreamEvent: StreamEvent{ + Kind: KindLog, + NetworkId: lg.networkID, + SourceId: lg.sourceID, + FilterHash: lg.filterHash, + Block: lg.block, + Payload: lg.payload, + }, + Seq: i.seq.Add(1), + Lifecycle: LifeSoft, + Removed: true, + }) + } + } +} + +// dedupe returns true if the event should be delivered, false if it's a +// dupe. For newHeads we use an optimistic fast-path on (last number, +// last hash); misses fall through to a bounded DedupWindow. +func (i *Indexer) dedupe(ns *networkState, ev *StreamEvent) bool { + switch ev.Kind { + case KindNewHead: + // CAS-retry on the packed (num, hash) pointer. On the happy path + // exactly one goroutine per distinct head wins the swap and falls + // through to mark+deliver; any concurrent ingest of the same head + // sees its CAS fail, reloads, and drops as a dupe on the next + // iteration. Reorgs at the same height (same num, different hash) + // win a second CAS and are delivered. + next := &headMarker{num: ev.Block.Number, hash: ev.Block.Hash} + for { + prev := ns.lastHead.Load() + if prev != nil { + if ev.Block.Number < prev.num { + return false + } + if ev.Block.Number == prev.num && prev.hash == ev.Block.Hash { + return false + } + } + if ns.lastHead.CompareAndSwap(prev, next) { + break + } + } + // Store in the fallback window too for the rare "older-but-not- + // newest" case (out-of-order delivery on a reorg boundary). + ns.headFallback.Mark(ev.Block.Hash) + return true + case KindLog, KindPendingTx: + ns.filterMu.RLock() + win := ns.filterDedup[ev.FilterHash] + ns.filterMu.RUnlock() + if win == nil { + // Filter not registered with this indexer instance (e.g. an + // ingress delivered an event for a filter we never EnsureFilter'd). + // Allow through — upstream subs we didn't request are rare. + return true + } + key := DedupKeyForFilter(ev.Kind.String(), ev.Payload) + if key == "" { + // Couldn't extract a key; don't pretend we deduped. + return true + } + return win.Mark(key) + default: + return true + } +} + +// classify computes Lifecycle for an event using the network's finality +// depth. Events at or below (latest - depth) are considered finalized. +// Applied only to events with a BlockRef; KindPendingTx and KindReorg +// default to LifeSoft. +func (i *Indexer) classify(ns *networkState, ev StreamEvent) Lifecycle { + if ev.Block.Zero() { + return LifeSoft + } + depth := ns.handle.FinalityDepth() + if depth <= 0 { + return LifeSoft + } + var latest int64 + if head := ns.lastHead.Load(); head != nil { + latest = head.num + } + if latest-ev.Block.Number >= depth { + return LifeFinalized + } + return LifeSoft +} + +// fanOut dispatches to every registered egress whose InterestedIn matches. +func (i *Indexer) fanOut(ev IndexedEvent) { + i.egresses.Range(func(_, v any) bool { + eg := v.(EventEgress) + if !eg.InterestedIn(ev.Kind, ev.NetworkId, ev.FilterHash) { + return true + } + eg.Deliver(ev) + return true + }) +} diff --git a/indexer/indexer_test.go b/indexer/indexer_test.go new file mode 100644 index 000000000..b586fb935 --- /dev/null +++ b/indexer/indexer_test.go @@ -0,0 +1,584 @@ +package indexer + +import ( + "context" + "encoding/json" + "fmt" + "sync" + "sync/atomic" + "testing" + + "github.com/rs/zerolog" +) + +// --- fakes ----------------------------------------------------------- + +type fakeNetwork struct { + id string + finalityDepth int64 + + mu sync.Mutex + suggestedBy map[string][]int64 // sourceId -> block nums seen +} + +func newFakeNetwork(id string, depth int64) *fakeNetwork { + return &fakeNetwork{ + id: id, + finalityDepth: depth, + suggestedBy: make(map[string][]int64), + } +} + +func (n *fakeNetwork) Id() string { return n.id } +func (n *fakeNetwork) FinalityDepth() int64 { return n.finalityDepth } +func (n *fakeNetwork) SuggestLatestBlock(sourceId string, block int64) { + n.mu.Lock() + n.suggestedBy[sourceId] = append(n.suggestedBy[sourceId], block) + n.mu.Unlock() +} + +type fakeEgress struct { + name string + filters map[string]struct{} // filterHash -> interested + acceptAllHeads bool + acceptReorgs bool + + mu sync.Mutex + received []IndexedEvent +} + +func (e *fakeEgress) Name() string { return e.name } +func (e *fakeEgress) InterestedIn(kind EventKind, networkId, filterHash string) bool { + switch kind { + case KindNewHead: + return e.acceptAllHeads + case KindReorg: + return e.acceptReorgs + } + _, ok := e.filters[filterHash] + return ok +} +func (e *fakeEgress) Deliver(ev IndexedEvent) { + e.mu.Lock() + e.received = append(e.received, ev) + e.mu.Unlock() +} +func (e *fakeEgress) count() int { + e.mu.Lock() + defer e.mu.Unlock() + return len(e.received) +} + +type fakeIngress struct { + name string + ensureCalls atomic.Int32 + removeCalls atomic.Int32 + lastParamsHash atomic.Value // string + // ensureErr, when set, is returned from every EnsureFilter call. + errMu sync.Mutex + ensureErr error + + startedFor NetworkHandle + sink Sink +} + +func (i *fakeIngress) Name() string { return i.name } +func (i *fakeIngress) Start(_ context.Context, nw NetworkHandle, sink Sink) error { + i.startedFor = nw + i.sink = sink + return nil +} +func (i *fakeIngress) setErr(err error) { + i.errMu.Lock() + i.ensureErr = err + i.errMu.Unlock() +} +func (i *fakeIngress) getErr() error { + i.errMu.Lock() + defer i.errMu.Unlock() + return i.ensureErr +} +func (i *fakeIngress) EnsureFilter(_ context.Context, _ string, paramsHash string, _ []interface{}) error { + i.ensureCalls.Add(1) + i.lastParamsHash.Store(paramsHash) + return i.getErr() +} +func (i *fakeIngress) RemoveFilter(_ context.Context, _, _ string) error { + i.removeCalls.Add(1) + return nil +} +func (i *fakeIngress) Stop(_ context.Context) error { return nil } + +// fakeSelector is a static IngressSelector for tests. +type fakeSelector struct { + defaults []string + fallbacks []string +} + +func (s *fakeSelector) Select(_, _ string, _ []interface{}) ([]string, []string) { + return s.defaults, s.fallbacks +} + +// --- tests ----------------------------------------------------------- + +func newIndexer(t *testing.T) *Indexer { + t.Helper() + logger := zerolog.New(zerolog.NewTestWriter(t)) + return New(&logger, Options{}) +} + +func TestIndexer_NewHead_FanOutAndDedup(t *testing.T) { + idx := newIndexer(t) + nw := newFakeNetwork("evm:1", 0) + idx.RegisterNetwork(nw) + eg := &fakeEgress{name: "eg1", filters: map[string]struct{}{}, acceptAllHeads: true} + idx.Attach(eg) + + ev := StreamEvent{ + Kind: KindNewHead, + NetworkId: "evm:1", + SourceId: "ws:up1", + Block: BlockRef{Number: 100, Hash: "0xAAA"}, + } + idx.Ingest(ev) + idx.Ingest(ev) // dup + + if got := eg.count(); got != 1 { + t.Fatalf("want 1 delivery after dup, got %d", got) + } + // Second unique head advances. + idx.Ingest(StreamEvent{Kind: KindNewHead, NetworkId: "evm:1", SourceId: "ws:up1", Block: BlockRef{Number: 101, Hash: "0xBBB"}}) + if got := eg.count(); got != 2 { + t.Fatalf("want 2 after advance, got %d", got) + } +} + +// TestIndexer_NewHead_ConcurrentIngestDedupe reproduces the TOCTOU race +// that prod evm:1101 hit: four WS upstream sources delivered the same +// newHead within ~1ms, and concurrent Ingest calls both read the stale +// lastHeadNum + Store the same new value + fell through to fanOut, so +// clients saw every head twice. The regression asserts that no matter +// how many goroutines race with the same (number, hash), the egress +// receives exactly one delivery. +func TestIndexer_NewHead_ConcurrentIngestDedupe(t *testing.T) { + idx := newIndexer(t) + nw := newFakeNetwork("evm:1", 0) + idx.RegisterNetwork(nw) + eg := &fakeEgress{name: "eg1", filters: map[string]struct{}{}, acceptAllHeads: true} + idx.Attach(eg) + + const sources = 8 + ev := StreamEvent{ + Kind: KindNewHead, + NetworkId: "evm:1", + Block: BlockRef{Number: 100, Hash: "0xAAA"}, + } + + var wg sync.WaitGroup + start := make(chan struct{}) + for i := 0; i < sources; i++ { + wg.Add(1) + ev := ev + ev.SourceId = fmt.Sprintf("ws:up%d", i) + go func() { + defer wg.Done() + <-start + idx.Ingest(ev) + }() + } + close(start) + wg.Wait() + + if got := eg.count(); got != 1 { + t.Fatalf("concurrent ingest of identical head must dedupe to 1 delivery, got %d", got) + } +} + +func TestIndexer_NewHead_StalerDroppedKeepsStatePollerFed(t *testing.T) { + idx := newIndexer(t) + nw := newFakeNetwork("evm:1", 0) + idx.RegisterNetwork(nw) + eg := &fakeEgress{name: "eg1", acceptAllHeads: true} + idx.Attach(eg) + + idx.Ingest(StreamEvent{Kind: KindNewHead, NetworkId: "evm:1", SourceId: "ws:up1", Block: BlockRef{Number: 100, Hash: "0xAAA"}}) + // A sibling source sees a newer head; different source, same block — + // state poller must still receive the update even though fan-out dedupes. + idx.Ingest(StreamEvent{Kind: KindNewHead, NetworkId: "evm:1", SourceId: "ws:up2", Block: BlockRef{Number: 100, Hash: "0xAAA"}}) + + if got := eg.count(); got != 1 { + t.Fatalf("duplicate head fan-out: want 1, got %d", got) + } + nw.mu.Lock() + defer nw.mu.Unlock() + if len(nw.suggestedBy["ws:up1"]) != 1 || len(nw.suggestedBy["ws:up2"]) != 1 { + t.Fatalf("each source must see its own SuggestLatestBlock, got %v", nw.suggestedBy) + } +} + +func TestIndexer_Log_RefcountFanOutAndTeardown(t *testing.T) { + idx := newIndexer(t) + nw := newFakeNetwork("evm:1", 0) + idx.RegisterNetwork(nw) + ing := &fakeIngress{name: "ws:up1"} + if err := idx.AddIngress(context.Background(), "evm:1", ing); err != nil { + t.Fatal(err) + } + + params := []interface{}{"logs", map[string]interface{}{"topics": []string{"0x1"}}} + h1, err := idx.EnsureFilter(context.Background(), "evm:1", "logs", params) + if err != nil { + t.Fatal(err) + } + // Second subscriber on same filter: refcount bumps, no new ingress call. + h2, _ := idx.EnsureFilter(context.Background(), "evm:1", "logs", params) + if h1 != h2 { + t.Fatalf("same params must hash identically, got %q vs %q", h1, h2) + } + if got := ing.ensureCalls.Load(); got != 1 { + t.Fatalf("EnsureFilter on ingress should be called exactly once, got %d", got) + } + + // First release decrements; still refcnt 1 → no tear-down. + idx.ReleaseFilter(context.Background(), "evm:1", "logs", h1) + if got := ing.removeCalls.Load(); got != 0 { + t.Fatalf("RemoveFilter must not fire while refcnt > 0, got %d calls", got) + } + // Second release drops to 0 → tear-down. + idx.ReleaseFilter(context.Background(), "evm:1", "logs", h1) + if got := ing.removeCalls.Load(); got != 1 { + t.Fatalf("RemoveFilter must fire exactly once at refcnt 0, got %d", got) + } +} + +func TestIndexer_Log_DedupFanOut(t *testing.T) { + idx := newIndexer(t) + nw := newFakeNetwork("evm:1", 0) + idx.RegisterNetwork(nw) + + params := []interface{}{"logs", map[string]interface{}{}} + h, _ := idx.EnsureFilter(context.Background(), "evm:1", "logs", params) + + eg := &fakeEgress{name: "eg1", filters: map[string]struct{}{h: {}}} + idx.Attach(eg) + + payload := json.RawMessage(`{"blockHash":"0xB","transactionHash":"0xT","logIndex":"0x0","removed":false}`) + ev := StreamEvent{ + Kind: KindLog, + NetworkId: "evm:1", + SourceId: "ws:up1", + FilterHash: h, + Payload: payload, + } + idx.Ingest(ev) + idx.Ingest(ev) // dup by (blockHash, txHash, logIndex, removed) + + if got := eg.count(); got != 1 { + t.Fatalf("log dedup: want 1, got %d", got) + } + + // Same log with removed=true must deliver (distinct dedup key). + idx.Ingest(StreamEvent{ + Kind: KindLog, NetworkId: "evm:1", SourceId: "ws:up1", FilterHash: h, + Payload: json.RawMessage(`{"blockHash":"0xB","transactionHash":"0xT","logIndex":"0x0","removed":true}`), + }) + if got := eg.count(); got != 2 { + t.Fatalf("log with removed=true must be delivered, got %d", got) + } + // Last delivered must carry Removed=true. + eg.mu.Lock() + last := eg.received[len(eg.received)-1] + eg.mu.Unlock() + if !last.Removed { + t.Fatalf("removed flag must propagate to IndexedEvent.Removed") + } +} + +func TestIndexer_Lifecycle_FinalityBoundary(t *testing.T) { + idx := newIndexer(t) + nw := newFakeNetwork("evm:1", 10) + idx.RegisterNetwork(nw) + eg := &fakeEgress{name: "eg1", acceptAllHeads: true} + idx.Attach(eg) + + // Latest head: 100. + idx.Ingest(StreamEvent{Kind: KindNewHead, NetworkId: "evm:1", SourceId: "ws:up1", Block: BlockRef{Number: 100, Hash: "0xH100"}}) + // A log at block 90 → latest-90 = 10 = depth → finalized. + idx.Ingest(StreamEvent{ + Kind: KindLog, NetworkId: "evm:1", SourceId: "ws:up1", + Block: BlockRef{Number: 90, Hash: "0xH90"}, + Payload: json.RawMessage(`{"blockHash":"0xH90","transactionHash":"0xT","logIndex":"0x0"}`), + }) + // A log at 95 → latest-95 = 5 < depth → soft. + idx.Ingest(StreamEvent{ + Kind: KindLog, NetworkId: "evm:1", SourceId: "ws:up1", + Block: BlockRef{Number: 95, Hash: "0xH95"}, + Payload: json.RawMessage(`{"blockHash":"0xH95","transactionHash":"0xT","logIndex":"0x1"}`), + }) + + eg.mu.Lock() + defer eg.mu.Unlock() + // Only KindNewHead at 100 got through the egress (logs don't pass + // through the egress without a matching filter). So assert lifecycle + // on the head event. + if len(eg.received) != 1 { + t.Fatalf("egress should have 1 newHead event, got %d", len(eg.received)) + } + head := eg.received[0] + // latest-latest = 0 < depth → soft. + if head.Lifecycle != LifeSoft { + t.Fatalf("latest head must be soft, got %v", head.Lifecycle) + } +} + +func TestIndexer_ReorgEmitsRemovedLogs(t *testing.T) { + idx := newIndexer(t) + nw := newFakeNetwork("evm:1", 0) + idx.RegisterNetwork(nw) + + params := []interface{}{"logs", map[string]interface{}{}} + h, _ := idx.EnsureFilter(context.Background(), "evm:1", "logs", params) + + eg := &fakeEgress{ + name: "eg1", + filters: map[string]struct{}{h: {}}, + acceptAllHeads: true, + acceptReorgs: true, + } + idx.Attach(eg) + + // newHead 100/0xA. + idx.Ingest(StreamEvent{ + Kind: KindNewHead, NetworkId: "evm:1", SourceId: "src1", + Block: BlockRef{Number: 100, Hash: "0xA", ParentHash: "0xZ"}, + }) + // newHead 101/0xB, child of 0xA. + idx.Ingest(StreamEvent{ + Kind: KindNewHead, NetworkId: "evm:1", SourceId: "src1", + Block: BlockRef{Number: 101, Hash: "0xB", ParentHash: "0xA"}, + }) + // Log in block 0xB. + logPayload := json.RawMessage(`{"blockHash":"0xB","blockNumber":"0x65","transactionHash":"0xT1","logIndex":"0x0"}`) + idx.Ingest(StreamEvent{ + Kind: KindLog, NetworkId: "evm:1", SourceId: "src1", FilterHash: h, + Block: BlockRef{Number: 101, Hash: "0xB"}, + Payload: logPayload, + }) + + // Reorg: new head 101/0xC replaces 0xB. + idx.Ingest(StreamEvent{ + Kind: KindNewHead, NetworkId: "evm:1", SourceId: "src1", + Block: BlockRef{Number: 101, Hash: "0xC", ParentHash: "0xA"}, + }) + + // Expected sequence in eg.received: + // newHead 100, newHead 101 (0xB), log 0xB, reorg summary, log 0xB (removed), newHead 101 (0xC) + // eg.interestedIn filters: newHeads + logs with filterHash h. + eg.mu.Lock() + defer eg.mu.Unlock() + if len(eg.received) < 6 { + t.Fatalf("want >= 6 events after reorg, got %d: %+v", len(eg.received), kinds(eg.received)) + } + // Find the reorg event. + var reorgIdx = -1 + for i, ev := range eg.received { + if ev.Kind == KindReorg { + reorgIdx = i + break + } + } + if reorgIdx < 0 { + t.Fatalf("KindReorg not emitted; got %+v", kinds(eg.received)) + } + // Next event after reorg (amongst this egress's filtered set) should + // be a log with Removed=true. + foundRemoved := false + for _, ev := range eg.received[reorgIdx+1:] { + if ev.Kind == KindLog && ev.Removed { + foundRemoved = true + break + } + } + if !foundRemoved { + t.Fatalf("expected a log with Removed=true after reorg, got %+v", kinds(eg.received)) + } +} + +func kinds(evs []IndexedEvent) []string { + out := make([]string, 0, len(evs)) + for _, ev := range evs { + out = append(out, ev.Kind.String()) + } + return out +} + +func TestIndexer_UnregisteredNetwork(t *testing.T) { + idx := newIndexer(t) + if err := idx.AddIngress(context.Background(), "evm:missing", &fakeIngress{name: "x"}); err == nil { + t.Fatal("expected error on unregistered network") + } + if _, err := idx.EnsureFilter(context.Background(), "evm:missing", "logs", []interface{}{"logs"}); err == nil { + t.Fatal("expected error on unregistered network") + } +} + +// registerThreeIngresses attaches three fakeIngresses named a, b, c on +// "evm:1" and returns them. Helper for selector tests. +func registerThreeIngresses(t *testing.T, idx *Indexer) (a, b, c *fakeIngress) { + t.Helper() + idx.RegisterNetwork(newFakeNetwork("evm:1", 0)) + a = &fakeIngress{name: "a"} + b = &fakeIngress{name: "b"} + c = &fakeIngress{name: "c"} + for _, ing := range []*fakeIngress{a, b, c} { + if err := idx.AddIngress(context.Background(), "evm:1", ing); err != nil { + t.Fatal(err) + } + } + return +} + +func TestIndexer_Selector_DefaultsOnlyHappyPath(t *testing.T) { + idx := newIndexer(t) + a, b, c := registerThreeIngresses(t, idx) + idx.RegisterNetworkSelector("evm:1", &fakeSelector{defaults: []string{"a", "b"}, fallbacks: []string{"c"}}) + + _, err := idx.EnsureFilter(context.Background(), "evm:1", "logs", []interface{}{"logs"}) + if err != nil { + t.Fatalf("unexpected error: %v", err) + } + if a.ensureCalls.Load() != 1 || b.ensureCalls.Load() != 1 { + t.Fatalf("defaults must be called once each, got a=%d b=%d", a.ensureCalls.Load(), b.ensureCalls.Load()) + } + if c.ensureCalls.Load() != 0 { + t.Fatalf("fallback must not be touched when defaults succeed, got %d calls", c.ensureCalls.Load()) + } +} + +func TestIndexer_Selector_PartialDefaultFailureReturnsNil(t *testing.T) { + idx := newIndexer(t) + a, _, c := registerThreeIngresses(t, idx) + a.setErr(errTest("a is sad")) + idx.RegisterNetworkSelector("evm:1", &fakeSelector{defaults: []string{"a", "b"}, fallbacks: []string{"c"}}) + + _, err := idx.EnsureFilter(context.Background(), "evm:1", "logs", []interface{}{"logs"}) + if err != nil { + t.Fatalf("partial failure must not surface: %v", err) + } + if a.ensureCalls.Load() != 1 { + t.Fatalf("failing default must still be attempted once, got %d", a.ensureCalls.Load()) + } + if c.ensureCalls.Load() != 0 { + t.Fatalf("fallback must not be touched when at least one default succeeded, got %d", c.ensureCalls.Load()) + } +} + +func TestIndexer_Selector_AllDefaultsFailEscalatesToFallback(t *testing.T) { + idx := newIndexer(t) + a, b, c := registerThreeIngresses(t, idx) + a.setErr(errTest("a")) + b.setErr(errTest("b")) + idx.RegisterNetworkSelector("evm:1", &fakeSelector{defaults: []string{"a", "b"}, fallbacks: []string{"c"}}) + + _, err := idx.EnsureFilter(context.Background(), "evm:1", "logs", []interface{}{"logs"}) + if err != nil { + t.Fatalf("escalation should return nil when fallback succeeds: %v", err) + } + if c.ensureCalls.Load() != 1 { + t.Fatalf("fallback must be tried after all defaults fail, got %d calls", c.ensureCalls.Load()) + } +} + +func TestIndexer_Selector_UnknownIngressIsAlwaysIncluded(t *testing.T) { + idx := newIndexer(t) + a, b, c := registerThreeIngresses(t, idx) + // Selector only knows about "a" and "b"; "c" stands in for a standalone + // ingress (e.g. Kafka) and must receive EnsureFilter unconditionally. + idx.RegisterNetworkSelector("evm:1", &fakeSelector{defaults: []string{"a"}, fallbacks: []string{"b"}}) + + if _, err := idx.EnsureFilter(context.Background(), "evm:1", "logs", []interface{}{"logs"}); err != nil { + t.Fatalf("unexpected error: %v", err) + } + if c.ensureCalls.Load() != 1 { + t.Fatalf("standalone ingress must be EnsureFiltered unconditionally, got %d calls", c.ensureCalls.Load()) + } + if a.ensureCalls.Load() != 1 { + t.Fatalf("default must be called, got %d", a.ensureCalls.Load()) + } + if b.ensureCalls.Load() != 0 { + t.Fatalf("fallback must not be called while a default or standalone succeeded, got %d", b.ensureCalls.Load()) + } +} + +func TestIndexer_Selector_AllFailReturnsJoinedError(t *testing.T) { + idx := newIndexer(t) + a, b, c := registerThreeIngresses(t, idx) + a.setErr(errTest("a")) + b.setErr(errTest("b")) + c.setErr(errTest("c")) + idx.RegisterNetworkSelector("evm:1", &fakeSelector{defaults: []string{"a", "b"}, fallbacks: []string{"c"}}) + + _, err := idx.EnsureFilter(context.Background(), "evm:1", "logs", []interface{}{"logs"}) + if err == nil { + t.Fatal("expected joined error when every tier fails") + } + msg := err.Error() + for _, want := range []string{"a", "b", "c"} { + if !containsStr(msg, want) { + t.Fatalf("joined error must include %q, got %q", want, msg) + } + } + // Re-attempt after fixing the ingresses must work — dedup window and + // refcount must have been rolled back. + a.setErr(nil) + b.setErr(nil) + c.setErr(nil) + if _, err := idx.EnsureFilter(context.Background(), "evm:1", "logs", []interface{}{"logs"}); err != nil { + t.Fatalf("second attempt must succeed after errors clear, got %v", err) + } +} + +func TestIndexer_NilSelector_FansOutToAllIngresses(t *testing.T) { + idx := newIndexer(t) + a, b, c := registerThreeIngresses(t, idx) + // No selector registered: every ingress must be tried. + + h, err := idx.EnsureFilter(context.Background(), "evm:1", "logs", []interface{}{"logs"}) + if err != nil { + t.Fatalf("unexpected error: %v", err) + } + for _, ing := range []*fakeIngress{a, b, c} { + if ing.ensureCalls.Load() != 1 { + t.Fatalf("ingress %q: want 1 EnsureFilter call, got %d", ing.name, ing.ensureCalls.Load()) + } + } + idx.ReleaseFilter(context.Background(), "evm:1", "logs", h) + for _, ing := range []*fakeIngress{a, b, c} { + if ing.removeCalls.Load() != 1 { + t.Fatalf("ingress %q: want 1 RemoveFilter call, got %d", ing.name, ing.removeCalls.Load()) + } + } +} + +// Test-only helpers. + +type errTest string + +func (e errTest) Error() string { return string(e) } + +func containsStr(haystack, needle string) bool { + return len(needle) == 0 || indexOf(haystack, needle) >= 0 +} + +func indexOf(s, substr string) int { + n := len(substr) + if n == 0 { + return 0 + } + for i := 0; i+n <= len(s); i++ { + if s[i:i+n] == substr { + return i + } + } + return -1 +} diff --git a/indexer/ingress.go b/indexer/ingress.go new file mode 100644 index 000000000..8026b7675 --- /dev/null +++ b/indexer/ingress.go @@ -0,0 +1,89 @@ +package indexer + +import "context" + +// Sink is the interface an ingress uses to push StreamEvents into the +// indexer pipeline. The indexer itself implements Sink — a pointer to the +// indexer is what adapters call on every upstream notification. +// +// Ingest is non-blocking and best-effort. Dedup, reorg handling, lifecycle +// tagging, and per-egress drop policy all live downstream; an ingress that +// re-delivers an already-seen notification is expected and handled. +type Sink interface { + Ingest(ev StreamEvent) +} + +// NetworkHandle is the narrow slice of network state an ingress is +// allowed to touch. It intentionally excludes almost everything on +// *erpc.Network — the invariant is that an ingress only needs +// identification, finality info, and per-source bookkeeping hooks. +type NetworkHandle interface { + // Id returns the network identifier ("evm:"). + Id() string + // FinalityDepth returns the number of blocks below the latest head + // that are considered finalized on this network. Used by the + // indexer when tagging IndexedEvent.Lifecycle. + FinalityDepth() int64 + // SuggestLatestBlock advances the per-source latest-block tracker + // before the indexer dedupes. Preserving "update-before-dedup" + // ordering is critical — the state poller needs to see every + // observation, even ones we'll drop in the fan-out stage. + SuggestLatestBlock(sourceId string, blockNumber int64) +} + +// EventIngress is an adapter that converts some transport-specific +// subscription (WS eth_subscribe, Kafka consumer, HTTP long-poll, …) +// into StreamEvents pushed at a Sink. +// +// Lifecycle expectations: +// +// - Start is called once when the indexer takes ownership. The ingress +// is expected to spin up its own goroutine(s) and push events at the +// sink until Stop is called or the context is cancelled. +// - EnsureFilter / RemoveFilter are invoked when the first/last client +// subscribes to a filter. Idempotent: repeated EnsureFilter calls for +// the same (subType, paramsHash) are no-ops. +// - Stop is best-effort; implementations should return promptly even +// if upstream unsubscribe RPCs time out. +type EventIngress interface { + // Name is a human-readable identifier used in logs/metrics + // ("ws:", "kafka:"). Must be stable for the life + // of the ingress. + Name() string + // Start begins pumping events. Returns only once the ingress has + // established its background workers — not once the first event + // arrives (that would race with upstream connectivity). + Start(ctx context.Context, nw NetworkHandle, sink Sink) error + // EnsureFilter subscribes (on the ingress's transport) to the given + // filter. The paramsHash is precomputed by the caller to match + // BuildParamsKey(params). Subsequent EnsureFilter calls with the + // same hash are no-ops. + EnsureFilter(ctx context.Context, subType string, paramsHash string, params []interface{}) error + // RemoveFilter unsubscribes the given filter from the transport. + // A no-op if the filter was never subscribed. Called when the last + // client for a filter unsubscribes. + RemoveFilter(ctx context.Context, subType string, paramsHash string) error + // Stop shuts down the ingress and releases resources. After Stop + // returns, no further events should be delivered to the sink. + Stop(ctx context.Context) error +} + +// IngressSelector chooses which of a network's pooled ingresses should +// carry a filter subscription. Select returns ingress names (matching +// EventIngress.Name()) split into two tiers: +// +// - defaults: tried first, as a group. At least one must succeed for +// the subscription to be considered established. +// - fallbacks: tried only if every default failed. +// +// Ingresses whose names appear in neither slice are treated as +// "always include" — they receive EnsureFilter unconditionally alongside +// the selected tier. This lets a selector that only understands WS +// upstreams leave standalone ingresses (Kafka topics, gRPC streams, etc.) +// untouched: they just get every filter automatically. +// +// A nil selector preserves the legacy behaviour: every registered ingress +// is treated as a default. +type IngressSelector interface { + Select(networkId, subType string, params []interface{}) (defaults, fallbacks []string) +} diff --git a/indexer/integration_test.go b/indexer/integration_test.go new file mode 100644 index 000000000..3eec40b0a --- /dev/null +++ b/indexer/integration_test.go @@ -0,0 +1,150 @@ +package indexer_test + +import ( + "context" + "encoding/json" + "sync" + "testing" + "time" + + "github.com/erpc/erpc/indexer" + "github.com/erpc/erpc/indexer/adapters/nullingress" + "github.com/rs/zerolog" +) + +// TestIntegration_NullIngress_EndToEnd wires a real indexer.Indexer to a +// transport-free nullingress.Adapter and asserts the full pipeline +// (ingest → dedup → lifecycle → fan-out) works with zero WebSocket-shaped +// dependencies. This is the forcing-function test proving the interface +// is transport-neutral. +func TestIntegration_NullIngress_EndToEnd(t *testing.T) { + logger := zerolog.New(zerolog.NewTestWriter(t)) + idx := indexer.New(&logger, indexer.Options{}) + + nw := &stubNetwork{id: "evm:1", finality: 10} + idx.RegisterNetwork(nw) + + ing := nullingress.New("null:test") + if err := idx.AddIngress(context.Background(), "evm:1", ing); err != nil { + t.Fatalf("AddIngress: %v", err) + } + + // Egress that records newHeads. + eg := &recordingEgress{interested: func(k indexer.EventKind, _, _ string) bool { + return k == indexer.KindNewHead + }} + idx.Attach(eg) + + // Push two distinct heads and a dup — expect 2 deliveries. + ing.Push(headEvent("evm:1", "null:test", 100, "0xA", "0x0")) + ing.Push(headEvent("evm:1", "null:test", 100, "0xA", "0x0")) // dup + ing.Push(headEvent("evm:1", "null:test", 101, "0xB", "0xA")) + + // Drain: give the pump goroutine + indexer a chance to run. + deadline := time.After(2 * time.Second) + for { + if eg.count() >= 2 { + break + } + select { + case <-deadline: + t.Fatalf("timed out waiting for 2 deliveries, got %d", eg.count()) + case <-time.After(10 * time.Millisecond): + } + } + if got := eg.count(); got != 2 { + t.Fatalf("want 2 deliveries (dup suppressed), got %d", got) + } + // Verify state poller was called per source. + nw.mu.Lock() + defer nw.mu.Unlock() + if n := len(nw.suggestions["null:test"]); n < 2 { + t.Fatalf("SuggestLatestBlock should fire on every observation (incl. dup), got %d", n) + } +} + +func TestIntegration_NullIngress_FilterRefcountAndTeardown(t *testing.T) { + logger := zerolog.New(zerolog.NewTestWriter(t)) + idx := indexer.New(&logger, indexer.Options{}) + + nw := &stubNetwork{id: "evm:1"} + idx.RegisterNetwork(nw) + ing := nullingress.New("null:test") + if err := idx.AddIngress(context.Background(), "evm:1", ing); err != nil { + t.Fatal(err) + } + + params := []interface{}{"logs", map[string]interface{}{"topics": []string{"0x1"}}} + h1, _ := idx.EnsureFilter(context.Background(), "evm:1", "logs", params) + h2, _ := idx.EnsureFilter(context.Background(), "evm:1", "logs", params) + if h1 != h2 { + t.Fatalf("same params must hash identically, got %q vs %q", h1, h2) + } + if n := len(ing.ActiveFilters()); n != 1 { + t.Fatalf("EnsureFilter should register exactly one filter on the ingress, got %d", n) + } + + // Only the second release tears down (refcount hits 0). + idx.ReleaseFilter(context.Background(), "evm:1", "logs", h1) + if n := len(ing.ActiveFilters()); n != 1 { + t.Fatalf("filter still held by 1 client — expected 1 active, got %d", n) + } + idx.ReleaseFilter(context.Background(), "evm:1", "logs", h1) + if n := len(ing.ActiveFilters()); n != 0 { + t.Fatalf("filter should be torn down, got %d active", n) + } +} + +// --- stubs ------------------------------------------------------------ + +type stubNetwork struct { + id string + finality int64 + + mu sync.Mutex + suggestions map[string][]int64 +} + +func (s *stubNetwork) Id() string { return s.id } +func (s *stubNetwork) FinalityDepth() int64 { return s.finality } +func (s *stubNetwork) SuggestLatestBlock(sourceID string, block int64) { + s.mu.Lock() + if s.suggestions == nil { + s.suggestions = make(map[string][]int64) + } + s.suggestions[sourceID] = append(s.suggestions[sourceID], block) + s.mu.Unlock() +} + +type recordingEgress struct { + interested func(indexer.EventKind, string, string) bool + + mu sync.Mutex + recv []indexer.IndexedEvent +} + +func (r *recordingEgress) Name() string { return "test:recording" } +func (r *recordingEgress) InterestedIn(k indexer.EventKind, networkID, filterHash string) bool { + return r.interested(k, networkID, filterHash) +} +func (r *recordingEgress) Deliver(ev indexer.IndexedEvent) { + r.mu.Lock() + r.recv = append(r.recv, ev) + r.mu.Unlock() +} +func (r *recordingEgress) count() int { + r.mu.Lock() + defer r.mu.Unlock() + return len(r.recv) +} + +func headEvent(networkID, sourceID string, num int64, hash, parent string) indexer.StreamEvent { + payload := json.RawMessage(`{"number":"0x0","hash":"` + hash + `","parentHash":"` + parent + `"}`) + return indexer.StreamEvent{ + Kind: indexer.KindNewHead, + NetworkId: networkID, + SourceId: sourceID, + Block: indexer.BlockRef{Number: num, Hash: hash, ParentHash: parent}, + Payload: payload, + } +} diff --git a/indexer/log_removed.go b/indexer/log_removed.go new file mode 100644 index 000000000..a94a77ec9 --- /dev/null +++ b/indexer/log_removed.go @@ -0,0 +1,23 @@ +package indexer + +import ( + "encoding/json" + + "github.com/erpc/erpc/common" +) + +// logRemoved extracts the "removed" flag from a log notification payload, +// defaulting to false on parse failure. Isolated into its own helper so +// the hot-path Ingest doesn't carry an inline json unmarshal. +func logRemoved(payload json.RawMessage) bool { + if len(payload) == 0 { + return false + } + var probe struct { + Removed bool `json:"removed"` + } + if err := common.SonicCfg.Unmarshal(payload, &probe); err != nil { + return false + } + return probe.Removed +} diff --git a/indexer/reorg.go b/indexer/reorg.go new file mode 100644 index 000000000..8f335bfee --- /dev/null +++ b/indexer/reorg.go @@ -0,0 +1,219 @@ +package indexer + +import ( + "encoding/json" + "sync" + + "github.com/erpc/erpc/common" +) + +// DefaultCanonicalChainDepth is the default ring-buffer size per +// network. Sized to cover deep-but-realistic reorgs (e.g. up to +// ~1 minute of blocks on a 12-second-block chain) without pinning too +// much memory per network. +const DefaultCanonicalChainDepth = 256 + +// canonicalChain is the per-network ring-buffer tracker. It records the +// last N canonical heads observed, detects reorgs on each new head via +// parentHash continuity, and maintains a per-blockHash index of indexed +// logs so reorged-out logs can be re-emitted with Removed=true. +// +// Scope limits: +// - Reorgs deeper than the ring depth are partially handled — the +// common ancestor is "not found", so we evict only the immediate +// head. Consumers that need stronger guarantees should use +// upstream-level consensus policies. +// - Gaps in observed heads (missed block between two observations) +// do not synthesize reorg events — the tracker treats them as +// forward progress. +type canonicalChain struct { + maxDepth int + + mu sync.Mutex + // ring is kept ordered by block number, oldest first. len <= maxDepth. + ring []BlockRef + // logs indexes delivered logs by block hash so we can re-emit them + // with Removed=true when the block is reorged out. Values point to + // the original StreamEvent payload (never mutated). + logs map[string][]loggedLog +} + +// loggedLog records enough detail to rebuild the IndexedEvent when a +// reorg invalidates the block. +type loggedLog struct { + filterHash string + networkID string + sourceID string + block BlockRef + payload json.RawMessage +} + +// newCanonicalChain constructs a tracker with the given ring depth. +// depth <= 0 falls back to DefaultCanonicalChainDepth. +func newCanonicalChain(depth int) *canonicalChain { + if depth <= 0 { + depth = DefaultCanonicalChainDepth + } + return &canonicalChain{ + maxDepth: depth, + ring: make([]BlockRef, 0, depth), + logs: make(map[string][]loggedLog), + } +} + +// observeHead pushes a new head onto the ring. Returns the set of blocks +// the ring evicted because the new head invalidated them — i.e. the +// reorged-out segment. Empty slice for the happy path. +// +// Semantics: +// +// - Ring empty or new head extends the tip cleanly: push, no +// evictions. +// - new.num < tip.num: caller has already ingested a newer head, this +// is stale — no-op, no evictions. +// - new.parentHash matches some ring entry: evict everything after +// that entry, push new — this is the reorg path. +// - new.parentHash unknown in the ring AND new.num <= tip.num: +// same-height or shallow reorg we can't fully walk; evict anything +// with num >= new.num and push. Best-effort. +// - Gap (new.num > tip.num+1): push without evictions, since we have +// no knowledge of the skipped blocks. +func (c *canonicalChain) observeHead(b BlockRef) (evicted []BlockRef) { + if b.Hash == "" || b.Number == 0 { + return nil + } + c.mu.Lock() + defer c.mu.Unlock() + + if len(c.ring) == 0 { + c.pushLocked(b) + return nil + } + + // parentHash continuity drives the primary decision: if we can find + // the incoming block's parent in the ring, we know exactly where the + // chain diverges and can evict everything after that point. This + // handles both the clean-extension case (parent == tip) and + // reorg-to-common-ancestor cases (parent deeper in ring). + if b.ParentHash != "" { + for i := len(c.ring) - 1; i >= 0; i-- { + if c.ring[i].Hash == b.ParentHash { + if i < len(c.ring)-1 { + evicted = append(evicted, c.ring[i+1:]...) + } + c.ring = c.ring[:i+1] + c.pushLocked(b) + return evicted + } + } + } + + // No parent match. Disambiguate: is this a stale/dup head or a + // genuine reorg whose ancestor fell out of our window? + tip := c.ring[len(c.ring)-1] + if b.Number < tip.Number { + // Stale: older height, unknown parent — caller has seen newer. + // Skip. + return nil + } + if b.Number == tip.Number && b.Hash == tip.Hash { + return nil + } + + // Fallback: unknown parent, new-ish height. Evict anything at or + // beyond the new height (same-level siblings are invalidated by + // the incoming head) and push. This is best-effort — deep reorgs + // beyond the ring window get partial handling here. + cut := len(c.ring) + for i := 0; i < len(c.ring); i++ { + if c.ring[i].Number >= b.Number { + cut = i + break + } + } + evicted = append(evicted, c.ring[cut:]...) + c.ring = c.ring[:cut] + c.pushLocked(b) + return evicted +} + +// pushLocked appends to the ring, evicting the oldest entry and its +// log index if we'd exceed maxDepth. Must be called with c.mu held. +func (c *canonicalChain) pushLocked(b BlockRef) { + c.ring = append(c.ring, b) + for len(c.ring) > c.maxDepth { + evictBlock := c.ring[0] + c.ring = c.ring[1:] + // Drop the log index for the evicted block — it's beyond our + // re-emission horizon anyway. + delete(c.logs, evictBlock.Hash) + } +} + +// indexLog records a log against its block hash so it can be re-emitted +// on reorg. No-op if the block hash is empty (ingress couldn't parse +// the log's blockHash, which shouldn't happen for well-formed payloads). +func (c *canonicalChain) indexLog(entry loggedLog) { + if entry.block.Hash == "" { + return + } + c.mu.Lock() + c.logs[entry.block.Hash] = append(c.logs[entry.block.Hash], entry) + c.mu.Unlock() +} + +// drainLogsFor returns and removes the recorded logs for a block hash. +// Used during reorg emission to rebuild IndexedEvents with Removed=true. +func (c *canonicalChain) drainLogsFor(blockHash string) []loggedLog { + c.mu.Lock() + defer c.mu.Unlock() + out := c.logs[blockHash] + delete(c.logs, blockHash) + return out +} + +// head returns the current tip or the zero BlockRef when empty. Test-only. +func (c *canonicalChain) head() BlockRef { + c.mu.Lock() + defer c.mu.Unlock() + if len(c.ring) == 0 { + return BlockRef{} + } + return c.ring[len(c.ring)-1] +} + +// reorgSummaryPayload serialises the evicted segment of a canonical +// chain as the payload for a KindReorg IndexedEvent. Consumers can +// unmarshal into an array of {number, hash, parentHash} objects. Small +// schema, forwards-compatible: the field set is a subset of BlockRef +// (additional fields later remain consumer-safe). +func reorgSummaryPayload(evicted []BlockRef) (json.RawMessage, error) { + items := make([]map[string]interface{}, 0, len(evicted)) + for _, b := range evicted { + items = append(items, map[string]interface{}{ + "number": b.Number, + "hash": b.Hash, + "parentHash": b.ParentHash, + }) + } + return common.SonicCfg.Marshal(map[string]interface{}{ + "evicted": items, + }) +} + +// parseLogBlockRef extracts the (number, hash) of a log payload's +// containing block. Returns BlockRef{} on parse failure. +func parseLogBlockRef(payload json.RawMessage) BlockRef { + if len(payload) == 0 { + return BlockRef{} + } + var probe struct { + BlockNumber string `json:"blockNumber"` + BlockHash string `json:"blockHash"` + } + if err := common.SonicCfg.Unmarshal(payload, &probe); err != nil { + return BlockRef{} + } + num, _ := common.HexToInt64(probe.BlockNumber) + return BlockRef{Number: num, Hash: probe.BlockHash} +} diff --git a/indexer/reorg_test.go b/indexer/reorg_test.go new file mode 100644 index 000000000..5ab5dba5a --- /dev/null +++ b/indexer/reorg_test.go @@ -0,0 +1,136 @@ +package indexer + +import ( + "encoding/json" + "reflect" + "testing" +) + +func TestCanonicalChain_NormalExtension(t *testing.T) { + c := newCanonicalChain(8) + ev := c.observeHead(BlockRef{Number: 100, Hash: "0xA", ParentHash: "0xZ"}) + if len(ev) != 0 { + t.Fatalf("first head, expected no evictions, got %d", len(ev)) + } + ev = c.observeHead(BlockRef{Number: 101, Hash: "0xB", ParentHash: "0xA"}) + if len(ev) != 0 { + t.Fatalf("clean extension, expected no evictions, got %v", ev) + } + if tip := c.head(); tip.Number != 101 || tip.Hash != "0xB" { + t.Fatalf("tip should be 101/0xB, got %+v", tip) + } +} + +func TestCanonicalChain_OneBlockReorg(t *testing.T) { + c := newCanonicalChain(8) + c.observeHead(BlockRef{Number: 100, Hash: "0xA", ParentHash: "0xZ"}) + c.observeHead(BlockRef{Number: 101, Hash: "0xB", ParentHash: "0xA"}) + + // A new head at 101 with a different hash than 0xB is a 1-block + // reorg. parentHash 0xA matches ring[0]; ring[1]=0xB evicted. + evicted := c.observeHead(BlockRef{Number: 101, Hash: "0xC", ParentHash: "0xA"}) + if len(evicted) != 1 || evicted[0].Hash != "0xB" { + t.Fatalf("expected 1 eviction of 0xB, got %+v", evicted) + } + if tip := c.head(); tip.Hash != "0xC" { + t.Fatalf("tip should be 0xC, got %+v", tip) + } +} + +func TestCanonicalChain_DeepReorgToCommonAncestor(t *testing.T) { + c := newCanonicalChain(8) + c.observeHead(BlockRef{Number: 100, Hash: "0xA", ParentHash: "0xZ"}) + c.observeHead(BlockRef{Number: 101, Hash: "0xB", ParentHash: "0xA"}) + c.observeHead(BlockRef{Number: 102, Hash: "0xC", ParentHash: "0xB"}) + c.observeHead(BlockRef{Number: 103, Hash: "0xD", ParentHash: "0xC"}) + + // Incoming head at 102 claims parentHash=0xA — reverts back two + // blocks. Expect ring[0xB, 0xC, 0xD] to be evicted. + evicted := c.observeHead(BlockRef{Number: 101, Hash: "0xX", ParentHash: "0xA"}) + if len(evicted) != 3 { + t.Fatalf("expected 3 evictions, got %d (%+v)", len(evicted), evicted) + } + hashes := []string{evicted[0].Hash, evicted[1].Hash, evicted[2].Hash} + if !reflect.DeepEqual(hashes, []string{"0xB", "0xC", "0xD"}) { + t.Fatalf("expected [0xB 0xC 0xD], got %v", hashes) + } +} + +func TestCanonicalChain_UnknownParentFallback(t *testing.T) { + c := newCanonicalChain(8) + c.observeHead(BlockRef{Number: 100, Hash: "0xA", ParentHash: "0xZ"}) + c.observeHead(BlockRef{Number: 101, Hash: "0xB", ParentHash: "0xA"}) + + // Incoming head at 102 with unknown parent. Fallback: evict anything + // with num >= 102 (none), push. No evictions. + ev := c.observeHead(BlockRef{Number: 102, Hash: "0xC", ParentHash: "0xUnknown"}) + if len(ev) != 0 { + t.Fatalf("expected 0 evictions (gap case), got %+v", ev) + } + // Incoming head at 101 with unknown parent: same-height reorg we + // can't fully reason about. Evict 0xB, push 0xD. The ring state is + // best-effort after this kind of event. + c = newCanonicalChain(8) + c.observeHead(BlockRef{Number: 100, Hash: "0xA", ParentHash: "0xZ"}) + c.observeHead(BlockRef{Number: 101, Hash: "0xB", ParentHash: "0xA"}) + ev = c.observeHead(BlockRef{Number: 101, Hash: "0xD", ParentHash: "0xUnknown"}) + if len(ev) != 1 || ev[0].Hash != "0xB" { + t.Fatalf("expected 1 eviction (0xB) on unknown-parent same-height, got %+v", ev) + } +} + +func TestCanonicalChain_StaleHeadIgnored(t *testing.T) { + c := newCanonicalChain(8) + c.observeHead(BlockRef{Number: 100, Hash: "0xA", ParentHash: "0xZ"}) + c.observeHead(BlockRef{Number: 101, Hash: "0xB", ParentHash: "0xA"}) + + // Stale head at 99 — ignore. + if ev := c.observeHead(BlockRef{Number: 99, Hash: "0xOld", ParentHash: "0xOlder"}); len(ev) != 0 { + t.Fatalf("stale head must produce 0 evictions, got %+v", ev) + } + if tip := c.head(); tip.Number != 101 { + t.Fatalf("tip must still be 101 after stale head, got %+v", tip) + } +} + +func TestCanonicalChain_RingCapacityEvictsOldest(t *testing.T) { + c := newCanonicalChain(3) + for i := int64(100); i <= 104; i++ { + c.observeHead(BlockRef{Number: i, Hash: stamp(i), ParentHash: stamp(i - 1)}) + } + if tip := c.head(); tip.Number != 104 { + t.Fatalf("tip should be 104, got %+v", tip) + } + // Ring should hold the last 3: 102, 103, 104. + c.mu.Lock() + defer c.mu.Unlock() + if len(c.ring) != 3 { + t.Fatalf("ring size should be 3, got %d", len(c.ring)) + } + if c.ring[0].Number != 102 { + t.Fatalf("oldest should be 102, got %+v", c.ring[0]) + } +} + +func TestCanonicalChain_LogIndexDrain(t *testing.T) { + c := newCanonicalChain(8) + // Seed 3 logs against block hash 0xB. + entry := loggedLog{filterHash: "h1", block: BlockRef{Number: 101, Hash: "0xB"}, payload: json.RawMessage(`{}`)} + c.indexLog(entry) + c.indexLog(entry) + c.indexLog(entry) + + drained := c.drainLogsFor("0xB") + if len(drained) != 3 { + t.Fatalf("want 3 drained, got %d", len(drained)) + } + // Second drain returns nothing. + if got := c.drainLogsFor("0xB"); len(got) != 0 { + t.Fatalf("drain should be exhaustive, got %d", len(got)) + } +} + +func stamp(n int64) string { + // Simple deterministic hash-like label for tests. + return "0x" + string(rune('A'+n%26)) +} diff --git a/upstream/registry.go b/upstream/registry.go index f26a0ee31..2dc2fd398 100644 --- a/upstream/registry.go +++ b/upstream/registry.go @@ -4,6 +4,7 @@ import ( "context" "errors" "fmt" + "net/url" "strings" "sync" "time" @@ -134,6 +135,10 @@ func (u *UpstreamsRegistry) GetInitializer() *util.Initializer { return u.initializer } +func (u *UpstreamsRegistry) SharedStateRegistry() data.SharedStateRegistry { + return u.sharedStateRegistry +} + func (u *UpstreamsRegistry) getNetworkMutex(networkId string) *sync.RWMutex { mutex, _ := u.networkMu.LoadOrStore(networkId, &sync.RWMutex{}) return mutex.(*sync.RWMutex) @@ -362,6 +367,26 @@ func (u *UpstreamsRegistry) GetNetworkUpstreams(ctx context.Context, networkId s return cp } +// GetWsUpstreams returns all WS-capable upstreams for a network (ws:// or wss:// endpoints). +func (u *UpstreamsRegistry) GetWsUpstreams(ctx context.Context, networkId string) []*Upstream { + all := u.GetNetworkUpstreams(ctx, networkId) + var ws []*Upstream + for _, up := range all { + cfg := up.Config() + if cfg == nil { + continue + } + parsed, err := url.Parse(cfg.Endpoint) + if err != nil { + continue + } + if parsed.Scheme == "ws" || parsed.Scheme == "wss" { + ws = append(ws, up) + } + } + return ws +} + func (u *UpstreamsRegistry) GetAllUpstreams() []*Upstream { u.upstreamsMu.RLock() defer u.upstreamsMu.RUnlock() diff --git a/upstream/upstream.go b/upstream/upstream.go index f273a164a..b69c224b3 100644 --- a/upstream/upstream.go +++ b/upstream/upstream.go @@ -292,6 +292,18 @@ func (u *Upstream) Config() *common.UpstreamConfig { return u.config } +func (u *Upstream) IsDown() bool { + if u == nil { + return true + } + for _, fe := range u.failsafeExecutors { + if br := fe.Breaker(); br != nil && br.State() == failsafe.StateOpen { + return true + } + } + return false +} + func (u *Upstream) MetricsTracker() *health.Tracker { if u == nil { return nil @@ -501,7 +513,7 @@ func (u *Upstream) Forward(ctx context.Context, nrq *common.NormalizedRequest, b // Send the request based on client type // switch clientType { - case clients.ClientTypeHttpJsonRpc, clients.ClientTypeGrpcBds: + case clients.ClientTypeHttpJsonRpc, clients.ClientTypeGrpcBds, clients.ClientTypeWsJsonRpc: tryForward := func( ctx context.Context, isHedge bool, From 94e412ea58958ebb61f20416b664b18de5c23a08 Mon Sep 17 00:00:00 2001 From: Jonny Date: Sat, 18 Apr 2026 09:51:57 +0100 Subject: [PATCH 02/40] fix(sharedState): keep-latest pub/sub delivery + ctx-deadline-aware refresh MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two related fixes to the shared-state subscriber path. 1) keep-latest pub/sub delivery. notifySubscribers used a try-send-or- skip pattern against a cap=1 subscriber channel. Under any burst (messageLoop + pollingLoop racing, or two instances publishing close together faster than the consumer drains), the newer value was silently dropped. For monotonic counters (latest / finalized block), the fresh value is the only one correctness depends on — dropping it reopens the propagation window the shared counter exists to close and produces cross-instance regressions that repeatable-read clients treat as data corruption. Replaced with sendKeepLatest: on a full buffer, drain the stale entry and retry the send. The counter is monotonic so displacing older by newer is always right. At the same time, drop the mutex-around-channel-send anti-pattern the old subscriber struct carried. A `done` channel closed once via sync.Once signals subscriber shutdown; writers select on ch | done | default and never need a lock. Receivers that want the shutdown signal select on done — sc.ch is no longer closed by any path. Cleanup and stop-all-subscribers both call sc.close(), idempotent. 2) honor caller ctx deadline for refresh fn timeout. The refresh function was running with the registry's app-level context, which ignored per-call deadlines. Plumbing the caller's context through makes refresh respect timeouts the consumer sets, instead of running unbounded. Regression and race-detector coverage in redis_pubsub_manager_test.go and shared_state_variable_test.go. --- data/redis_pubsub_manager.go | 114 ++++++++++++++--------- data/redis_pubsub_manager_test.go | 144 +++++++++++++++++++++++++++++ data/shared_state_variable.go | 25 ++++- data/shared_state_variable_test.go | 93 +++++++++++++++++++ erpc/grpc_server_test.go | 2 +- erpc/ws_server_test.go | 22 ++++- 6 files changed, 349 insertions(+), 51 deletions(-) create mode 100644 data/redis_pubsub_manager_test.go diff --git a/data/redis_pubsub_manager.go b/data/redis_pubsub_manager.go index 57e8da3ad..128e28408 100644 --- a/data/redis_pubsub_manager.go +++ b/data/redis_pubsub_manager.go @@ -15,11 +15,61 @@ import ( "github.com/rs/zerolog" ) -// subscriberChannel wraps a channel with metadata to prevent double-close +// subscriberChannel wraps the delivery channel with a lifecycle signal. +// +// Historically this struct tracked `closed bool` guarded by a mutex so +// writers could avoid "send on closed channel" panics. That forced every +// publish to take a per-subscriber lock just to read a channel — which +// is an anti-pattern, and got awkward fast once notifySubscribers grew a +// drop-oldest retry loop (to keep the latest monotonic value when the +// buffered slot was full). +// +// The replacement is the standard idiom: a `done` channel closed exactly +// once. Receivers that also select on `<-done` observe subscriber +// shutdown without requiring the sender-side to close `ch`, which means: +// - notifySubscribers can be lock-free (select on ch|done|default). +// - cleanup/stop call close(done), never close(ch). +// - A second cleanup is a no-op via sync.Once. type subscriberChannel struct { - ch chan CounterInt64State - closed bool - mu sync.Mutex + ch chan CounterInt64State + done chan struct{} + closeOnce sync.Once +} + +// close marks the subscriber as gone. Idempotent. Does not close `ch` — +// consumers that also select on `<-done` see the shutdown signal there. +func (sc *subscriberChannel) close() { + sc.closeOnce.Do(func() { close(sc.done) }) +} + +// sendKeepLatest delivers value into sc.ch using keep-latest semantics: +// if the buffered slot is full, the stale value is drained and the new +// one takes its place. Returns early if the subscriber has been closed. +// Never blocks on a slow consumer. +// +// Keep-latest (rather than drop-new) matters for monotonic counters +// (latest / finalized block): the freshest value is the only one +// correctness depends on, and silently dropping it opens a propagation +// window where different eRPC instances answer with regressing values. +func (sc *subscriberChannel) sendKeepLatest(value CounterInt64State) { + for { + select { + case sc.ch <- value: + return + case <-sc.done: + return + default: + } + // Buffer full. Drain the stale entry, then loop and retry the send. + // The inner select also watches done so we don't spin after shutdown. + select { + case <-sc.ch: + case <-sc.done: + return + default: + // Concurrent consumer drained for us; loop and retry the send. + } + } } // RedisPubSubManager is a self-healing manager for Redis pubsub subscriptions. @@ -106,16 +156,10 @@ func (m *RedisPubSubManager) stop() { } } - // Close all subscriber channels + // Signal shutdown to every subscriber. Idempotent via sync.Once. m.subscribers.Range(func(key, value interface{}) bool { - channels := value.([]*subscriberChannel) - for _, sc := range channels { - sc.mu.Lock() - if !sc.closed { - close(sc.ch) - sc.closed = true - } - sc.mu.Unlock() + for _, sc := range value.([]*subscriberChannel) { + sc.close() } return true }) @@ -191,35 +235,28 @@ func (m *RedisPubSubManager) Subscribe(key string) (<-chan CounterInt64State, fu } sc := &subscriberChannel{ - ch: make(chan CounterInt64State, 1), + ch: make(chan CounterInt64State, 1), + done: make(chan struct{}), } // Add the channel to subscribers m.addSubscriber(key, sc) - // Get initial value in background + // Get initial value in background. Reuse sendKeepLatest so a pubsub + // message that landed first isn't clobbered by a stale initial fetch + // (processNewState's timestamp ordering also catches this, but + // keep-latest at the transport means the consumer never even sees + // the out-of-order value). go func() { if val, ok, err := m.getCurrentValue(m.appCtx, key); err == nil && ok { - sc.mu.Lock() - if !sc.closed { - select { - case sc.ch <- val: - case <-m.appCtx.Done(): - } - } - sc.mu.Unlock() + sc.sendKeepLatest(val) } }() - // Return cleanup function + // Return cleanup function. Idempotent. cleanup := func() { m.removeSubscriber(key, sc) - sc.mu.Lock() - if !sc.closed { - close(sc.ch) - sc.closed = true - } - sc.mu.Unlock() + sc.close() } return sc.ch, cleanup, nil @@ -438,23 +475,14 @@ func (m *RedisPubSubManager) removeSubscriber(key string, sc *subscriberChannel) } } -// notifySubscribers sends a value to all subscribers of a key +// notifySubscribers delivers value to all subscribers of key with +// keep-latest semantics. Lock-free on the hot path. func (m *RedisPubSubManager) notifySubscribers(key string, value CounterInt64State) { subsValue, ok := m.subscribers.Load(key) if !ok { return } - - channels := subsValue.([]*subscriberChannel) - for _, sc := range channels { - sc.mu.Lock() - if !sc.closed { - select { - case sc.ch <- value: - default: - // Channel is full, skip - } - } - sc.mu.Unlock() + for _, sc := range subsValue.([]*subscriberChannel) { + sc.sendKeepLatest(value) } } diff --git a/data/redis_pubsub_manager_test.go b/data/redis_pubsub_manager_test.go new file mode 100644 index 000000000..ba9ab417d --- /dev/null +++ b/data/redis_pubsub_manager_test.go @@ -0,0 +1,144 @@ +package data + +import ( + "sync" + "testing" + "time" + + "github.com/stretchr/testify/assert" +) + +func newTestSubscriberChannel() *subscriberChannel { + return &subscriberChannel{ + ch: make(chan CounterInt64State, 1), + done: make(chan struct{}), + } +} + +// Regression test for the propagation gap that caused cross-pod finalized- +// block regressions in production: when two counter updates for the same +// key arrived faster than the consumer goroutine drained its cap=1 buffer, +// the newer message was silently dropped. For monotonic counters the +// freshest value is the only one that matters, so sendKeepLatest must +// evict the stale buffered value instead of refusing the new one. +func TestSubscriberChannel_SendKeepLatestEvictsStale(t *testing.T) { + sc := newTestSubscriberChannel() + + first := CounterInt64State{Value: 100, UpdatedAt: 1} + second := CounterInt64State{Value: 101, UpdatedAt: 2} + third := CounterInt64State{Value: 102, UpdatedAt: 3} + + // No consumer — three publishes land back-to-back, overflowing the + // single-slot buffer twice. Only the latest value must survive. + sc.sendKeepLatest(first) + sc.sendKeepLatest(second) + sc.sendKeepLatest(third) + + select { + case got := <-sc.ch: + assert.Equal(t, third, got, "full buffer must be updated to the latest published value") + default: + t.Fatalf("subscriber channel empty after three publishes") + } + + select { + case extra := <-sc.ch: + t.Fatalf("unexpected extra value on subscriber channel: %+v", extra) + default: + } +} + +func TestSubscriberChannel_SendKeepLatestSequentialDeliversEachValue(t *testing.T) { + sc := newTestSubscriberChannel() + + for _, v := range []int64{10, 11, 12, 13} { + st := CounterInt64State{Value: v, UpdatedAt: v} + sc.sendKeepLatest(st) + got := <-sc.ch + assert.Equal(t, st, got, "sequential publish+drain should deliver each value untouched") + } +} + +// Once the subscriber has been closed, further sends must not block the +// publisher — they return immediately. Prevents any single slow consumer +// from wedging the pub/sub manager goroutine after cleanup. +func TestSubscriberChannel_SendKeepLatestReturnsAfterClose(t *testing.T) { + sc := newTestSubscriberChannel() + + // Fill the buffer so a subsequent send would spin in the drain loop + // if it didn't respect the done signal. + sc.ch <- CounterInt64State{Value: 1, UpdatedAt: 1} + sc.close() + + done := make(chan struct{}) + go func() { + defer close(done) + sc.sendKeepLatest(CounterInt64State{Value: 99, UpdatedAt: 99}) + }() + + select { + case <-done: + // ok — send returned promptly + case <-time.After(time.Second): + t.Fatalf("sendKeepLatest did not return after subscriber was closed") + } +} + +// close must be idempotent — stop() and the per-subscriber cleanup can +// both fire for the same subscriber during shutdown. +func TestSubscriberChannel_CloseIsIdempotent(t *testing.T) { + sc := newTestSubscriberChannel() + sc.close() + sc.close() + sc.close() + + select { + case <-sc.done: + // expected + default: + t.Fatalf("done channel must be closed after close()") + } +} + +// End-to-end via notifySubscribers: the manager must honour keep-latest +// across every subscribed channel for the key without taking any lock. +func TestRedisPubSubManager_NotifySubscribersKeepsLatest(t *testing.T) { + m := &RedisPubSubManager{} + a := newTestSubscriberChannel() + b := newTestSubscriberChannel() + m.addSubscriber("finalized", a) + m.addSubscriber("finalized", b) + + m.notifySubscribers("finalized", CounterInt64State{Value: 1, UpdatedAt: 1}) + m.notifySubscribers("finalized", CounterInt64State{Value: 2, UpdatedAt: 2}) + + for _, sc := range []*subscriberChannel{a, b} { + got := <-sc.ch + assert.Equal(t, int64(2), got.Value, "each subscriber should observe the latest value") + } +} + +// Race detector guard: under concurrent publishes + concurrent closes, no +// send ever panics, no goroutine leaks, and the subscriber eventually sees +// either the latest value or nothing (if it was already closed). +func TestSubscriberChannel_ConcurrentSendAndClose(t *testing.T) { + for trial := 0; trial < 20; trial++ { + sc := newTestSubscriberChannel() + + var wg sync.WaitGroup + wg.Add(2) + go func() { + defer wg.Done() + for i := 0; i < 100; i++ { + sc.sendKeepLatest(CounterInt64State{Value: int64(i), UpdatedAt: int64(i + 1)}) + } + }() + go func() { + defer wg.Done() + // Close mid-burst. + time.Sleep(time.Microsecond * 50) + sc.close() + }() + wg.Wait() + } +} diff --git a/data/shared_state_variable.go b/data/shared_state_variable.go index ac1918b11..2d2ef1def 100644 --- a/data/shared_state_variable.go +++ b/data/shared_state_variable.go @@ -420,17 +420,36 @@ func (c *counterInt64) TryUpdateIfStale(ctx context.Context, staleness time.Dura // (e.g. Redis lock acquisition) can block normal request flow. span.SetAttributes(attribute.Bool("foreground_remote_io_disabled", true)) - // Execute the refresh function (e.g., RPC call to get latest block) in background + // Execute the refresh function (e.g., RPC call to get latest block) in background. + // + // Timeout source: fallbackTimeout is the "remote storage op" default (Redis + // get/set, ~3s) and is intentionally tight. But executeNewValueFn often does + // work that legitimately takes longer (e.g., polling latest block on a slow + // chain). If the caller passes a context with an explicit deadline (the state + // poller sets one derived from lockTtl+operationBuffer), honor it when it's + // longer than fallbackTimeout — otherwise we silently cap slow-chain polls at + // 3s and the shared state never updates, producing a flood of + // "context deadline exceeded" logs with no way to raise the bound. + // + // Note: parent remains c.registry.appCtx (not ctx) so the background fetch + // survives foreground cancellation — only the timeout is read from ctx. + fnTimeout := c.registry.fallbackTimeout + if deadline, ok := ctx.Deadline(); ok { + if remaining := time.Until(deadline); remaining > fnTimeout { + fnTimeout = remaining + } + } + resultCh := make(chan refreshResult, 1) go func() { - fnCtx, fnCancel := context.WithTimeout(c.registry.appCtx, c.registry.fallbackTimeout) + fnCtx, fnCancel := context.WithTimeout(c.registry.appCtx, fnTimeout) defer fnCancel() // Create a span for the actual RPC/refresh call - this is usually what takes time _, fnSpan := common.StartSpan(ctx, "CounterInt64.TryUpdateIfStale.ExecuteRefresh", trace.WithAttributes( attribute.String("key", c.key), - attribute.Int64("timeout_ms", c.registry.fallbackTimeout.Milliseconds()), + attribute.Int64("timeout_ms", fnTimeout.Milliseconds()), ), ) value, err := executeNewValueFn(fnCtx) diff --git a/data/shared_state_variable_test.go b/data/shared_state_variable_test.go index 39f6bbfcb..4a706cb19 100644 --- a/data/shared_state_variable_test.go +++ b/data/shared_state_variable_test.go @@ -841,6 +841,99 @@ func TestCounterInt64_TryUpdateIfStale_NoThunderingHerdOnError(t *testing.T) { connector.AssertExpectations(t) } +// TestCounterInt64_TryUpdateIfStale_FnTimeoutFromCtxDeadline verifies that the +// background refresh fn receives at least the caller-provided context deadline +// as its timeout, rather than being silently capped at fallbackTimeout. This +// matters for slow-chain state pollers (e.g. Hedera relay) whose eth_getBlockByNumber +// legitimately exceeds the default 3s fallbackTimeout. +func TestCounterInt64_TryUpdateIfStale_FnTimeoutFromCtxDeadline(t *testing.T) { + // refreshFn deliberately returns an error so applyRefreshResult short-circuits + // before scheduleBackgroundPushCurrent, avoiding a background Publish goroutine + // that would race across t.Run boundaries. + refreshErr := errors.New("intentional") + makeCounter := func(t *testing.T, registry *sharedStateRegistry, key string) *counterInt64 { + t.Helper() + c := &counterInt64{ + registry: registry, + key: key, + ignoreRollbackOf: 1024, + } + c.value.Store(5) + c.updatedAtUnixMs.Store(time.Now().Add(-2 * time.Second).UnixMilli()) + return c + } + + t.Run("ctx deadline longer than fallback is honored", func(t *testing.T) { + registry, _, _ := setupTest("my-dev") + registry.fallbackTimeout = 500 * time.Millisecond + registry.updateMaxWait = 5 * time.Second + + counter := makeCounter(t, registry, "test-ctx-deadline") + + const callerDeadline = 4 * time.Second + ctx, cancel := context.WithTimeout(context.Background(), callerDeadline) + defer cancel() + + var fnTimeout time.Duration + refreshFn := func(fnCtx context.Context) (int64, error) { + if deadline, ok := fnCtx.Deadline(); ok { + fnTimeout = time.Until(deadline) + } + return 0, refreshErr + } + + _, err := counter.TryUpdateIfStale(ctx, time.Second, refreshFn) + assert.ErrorIs(t, err, refreshErr) + assert.Greater(t, fnTimeout, registry.fallbackTimeout, + "fn timeout should honor caller ctx deadline when longer than fallbackTimeout") + assert.LessOrEqual(t, fnTimeout, callerDeadline) + }) + + t.Run("no ctx deadline falls back to fallbackTimeout", func(t *testing.T) { + registry, _, _ := setupTest("my-dev") + registry.fallbackTimeout = 500 * time.Millisecond + + counter := makeCounter(t, registry, "test-no-deadline") + + var fnTimeout time.Duration + refreshFn := func(fnCtx context.Context) (int64, error) { + if deadline, ok := fnCtx.Deadline(); ok { + fnTimeout = time.Until(deadline) + } + return 0, refreshErr + } + + _, err := counter.TryUpdateIfStale(context.Background(), time.Second, refreshFn) + assert.ErrorIs(t, err, refreshErr) + assert.LessOrEqual(t, fnTimeout, registry.fallbackTimeout, + "fn timeout should be bounded by fallbackTimeout when caller has no deadline") + }) + + t.Run("ctx deadline shorter than fallback uses fallback", func(t *testing.T) { + registry, _, _ := setupTest("my-dev") + registry.fallbackTimeout = 2 * time.Second + registry.updateMaxWait = 5 * time.Second + + counter := makeCounter(t, registry, "test-short-deadline") + + ctx, cancel := context.WithTimeout(context.Background(), 100*time.Millisecond) + defer cancel() + + var fnTimeout time.Duration + refreshFn := func(fnCtx context.Context) (int64, error) { + if deadline, ok := fnCtx.Deadline(); ok { + fnTimeout = time.Until(deadline) + } + return 0, refreshErr + } + + _, err := counter.TryUpdateIfStale(ctx, time.Second, refreshFn) + assert.ErrorIs(t, err, refreshErr) + assert.Greater(t, fnTimeout, 500*time.Millisecond, + "fn timeout should use fallbackTimeout when caller deadline is shorter") + }) +} + func TestCounterInt64_ReaderStarvation(t *testing.T) { t.Run("GetValue NOT blocked by long-running TryUpdateIfStale", func(t *testing.T) { counter := &counterInt64{ diff --git a/erpc/grpc_server_test.go b/erpc/grpc_server_test.go index c0d44c05d..67454c129 100644 --- a/erpc/grpc_server_test.go +++ b/erpc/grpc_server_test.go @@ -144,7 +144,7 @@ func TestHttpServer_CanSharePortWithGrpc(t *testing.T) { require.NoError(t, err) erpcInstance.Bootstrap(ctx) - httpServer, err := NewHttpServer(ctx, &logger, cfg.Server, cfg.HealthCheck, cfg.Admin, erpcInstance) + httpServer, err := NewHttpServer(ctx, &logger, cfg.Server, cfg.HealthCheck, cfg.Admin, cfg.Indexer, erpcInstance) require.NoError(t, err) require.NotNil(t, httpServer.sharedGrpcServer) diff --git a/erpc/ws_server_test.go b/erpc/ws_server_test.go index 8d732915e..94e4d569a 100644 --- a/erpc/ws_server_test.go +++ b/erpc/ws_server_test.go @@ -1392,13 +1392,17 @@ func TestWebSocket_RegressionBootstrapRetriedOnEverySubscribe(t *testing.T) { defer cleanup() time.Sleep(2 * time.Second) - // First subscribe triggers bootstrap on the WS upstream. + // First subscribe triggers bootstrap on the WS upstream. The upstream + // subscribe is dispatched in a goroutine (wsupstream adapter fires + // initialSubscribe async), so we poll for it. conn1 := dialWs(t, addr) defer conn1.Close() resp1 := sendAndReceive(t, conn1, `{"jsonrpc":"2.0","id":1,"method":"eth_subscribe","params":["newHeads"]}`) require.NotNil(t, resp1["result"]) + require.Eventually(t, func() bool { + return atomic.LoadInt64(&subscribeCount) >= 1 + }, 3*time.Second, 20*time.Millisecond, "first client should trigger upstream subscribe") first := atomic.LoadInt64(&subscribeCount) - require.GreaterOrEqual(t, first, int64(1), "first client should trigger upstream subscribe") // Subsequent subscribes (different params) must also call BootstrapNetwork's // idempotent path — no new upstream subscribe expected since the upstream @@ -1408,6 +1412,9 @@ func TestWebSocket_RegressionBootstrapRetriedOnEverySubscribe(t *testing.T) { defer conn2.Close() resp2 := sendAndReceive(t, conn2, `{"jsonrpc":"2.0","id":1,"method":"eth_subscribe","params":["newHeads"]}`) require.NotNil(t, resp2["result"]) + // Small stabilization window: even if a second upstream subscribe were + // erroneously triggered, it would happen shortly after the response. + time.Sleep(200 * time.Millisecond) assert.Equal(t, first, atomic.LoadInt64(&subscribeCount), "second client should reuse existing newHeads sub (idempotent bootstrap)") } @@ -1661,15 +1668,22 @@ func TestWebSocket_RegressionInternalRequestIdsDontCollide(t *testing.T) { defer cleanup() time.Sleep(2 * time.Second) - // Trigger an internal upstream subscribe via a client subscribe. + // Trigger an internal upstream subscribe via a client subscribe. The + // upstream eth_subscribe is dispatched async (wsupstream adapter fires + // initialSubscribe in a goroutine), so poll until it arrives. conn := dialWs(t, addr) defer conn.Close() resp := sendAndReceive(t, conn, `{"jsonrpc":"2.0","id":1,"method":"eth_subscribe","params":["newHeads"]}`) require.NotNil(t, resp["result"]) + require.Eventually(t, func() bool { + idMu.Lock() + defer idMu.Unlock() + return len(subscribeIds) >= 1 + }, 3*time.Second, 20*time.Millisecond, "should have observed at least one internal eth_subscribe") + idMu.Lock() defer idMu.Unlock() - require.NotEmpty(t, subscribeIds, "should have observed at least one internal eth_subscribe") // Internal IDs use a large offset (>= 900M) to avoid collisions with the // state poller's small integer IDs. From cddeb01853b7e60de044a7275beadd42b5f0b172 Mon Sep 17 00:00:00 2001 From: Jonny Date: Wed, 29 Apr 2026 21:18:28 +0100 Subject: [PATCH 03/40] fix(upstream): plug four upstream-instance leak paths MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A heap/goroutine profile from a long-running deployment showed configured upstream counts expanding by 5-10x at runtime: the per-upstream client instances (HTTP and WS), EvmStatePoller polling goroutines, and sharedStateRegistry counter-sync goroutines all multiplied past their configured set, with heap retention concentrated in upstream.NewUpstream and its policy/health-tracker builders. Four interlocking bugs were causing transient and ostensibly-deduplicated upstream constructions to actually leak: 1. common.UniqueUpstreamKey hashed up.NetworkId() (which returns "n/a" before registration and the real id after) and iterated headers in non-deterministic Go map order. The same upstream produced different keys at different points in its lifecycle, defeating the per-upstream client cache and shared-state counter dedup. Drop NetworkId from the key (cfg.Endpoint already disambiguates) and sort header keys before hashing. Regression test in common/upstream_test.go. 2. Upstream.Bootstrap unconditionally reassigned u.evmStatePoller = evm.NewEvmStatePoller(...) and called Bootstrap on it. The new poller spawns a goroutine that listens only on appCtx, so any re-entry of Bootstrap orphans the prior poller's goroutine for the lifetime of the process. Guard creation with evmStatePollerMu + nil-check so repeated Bootstrap calls are idempotent for the poller. 3. ClientRegistry.CreateClient declared `var once sync.Once` locally, so every call ran the body. Concurrent callers that both missed the `clients` cache could each spawn an HTTP/WS client (with shutdown waiter, ping/read loops); only one won the manager.clients.Store — the losers' goroutines leaked. Replace with a sync.Map[key] *clientCreation that shares both the once and the build result, so concurrent callers all return the winning client. 4. config_analyzer.validateUpstreamEndpoints and GenerateValidationReport passed the caller's ctx (effectively appCtx) into upstream.NewUpstream for transient validation upstreams. Their client goroutines listen on that ctx and so outlived validation and accumulated forever. Wrap each validation pass in a context.WithCancel and pass the child ctx to NewUpstream; the deferred cancel terminates the transient client goroutines on validation completion. After these fixes the per-upstream goroutine multiplier collapses back to ~1 across all four populations (client, shutdown waiter, poller, counter-sync), and heap retention from leaked NewUpstream instances goes with it. Co-Authored-By: Claude Opus 4.7 (1M context) --- clients/registry.go | 135 +++++++++++++---------- common/upstream.go | 34 ++++-- common/upstream_test.go | 89 +++++++++++++++ erpc/config_analyzer.go | 24 +++- erpc/networks_earliest_detection_test.go | 7 +- go.mod | 2 +- upstream/upstream.go | 28 +++-- 7 files changed, 233 insertions(+), 86 deletions(-) create mode 100644 common/upstream_test.go diff --git a/clients/registry.go b/clients/registry.go index b52999c02..b35e9a2d5 100644 --- a/clients/registry.go +++ b/clients/registry.go @@ -27,10 +27,24 @@ type Client struct { Upstream common.Upstream } +// clientCreation memoises the once-per-upstream client construction. Sharing +// the sync.Once across CreateClient calls is the correctness-critical part: +// previously `var once sync.Once` was declared locally so every call ran the +// body, and two concurrent callers that both missed the cache could each +// spawn a client and its goroutines, with only the last winning Store — the +// losing client (and its <-appCtx.Done() shutdown waiter, ping/read loops, +// etc.) leaked for the lifetime of the process. +type clientCreation struct { + once sync.Once + client ClientInterface + err error +} + type ClientRegistry struct { logger *zerolog.Logger projectId string - clients sync.Map + clients sync.Map // upstream key -> ClientInterface (read-fast path) + clientCreations sync.Map // upstream key -> *clientCreation (build coordination) proxyPoolRegistry *ProxyPoolRegistry evmExtractor common.JsonRpcErrorExtractor } @@ -54,10 +68,6 @@ func (manager *ClientRegistry) GetOrCreateClient(appCtx context.Context, ups com } func (manager *ClientRegistry) CreateClient(appCtx context.Context, ups common.Upstream) (ClientInterface, error) { - var once sync.Once - var newClient ClientInterface - var clientErr error - cfg := ups.Config() if cfg.Endpoint == "" { @@ -77,64 +87,67 @@ func (manager *ClientRegistry) CreateClient(appCtx context.Context, ups common.U } } - if err != nil { - clientErr = fmt.Errorf("failed to parse URL for upstream: %v", cfg.Id) - } else { - once.Do(func() { - lg := manager.logger.With().Str("upstreamId", cfg.Id).Logger() - switch cfg.Type { - case common.UpstreamTypeEvm: - if parsedUrl.Scheme == "http" || parsedUrl.Scheme == "https" { - newClient, err = NewGenericHttpJsonRpcClient( - appCtx, - &lg, - manager.projectId, - ups, - parsedUrl, - cfg.JsonRpc, - proxyPool, - manager.evmExtractor, - ) - if err != nil { - clientErr = fmt.Errorf("failed to create HTTP client for upstream: %v", cfg.Id) - } - } else if parsedUrl.Scheme == "ws" || parsedUrl.Scheme == "wss" { - newClient, err = NewWsJsonRpcClient( - appCtx, - &lg, - manager.projectId, - ups, - parsedUrl, - cfg.JsonRpc, - manager.evmExtractor, - ) - if err != nil { - clientErr = fmt.Errorf("failed to create WebSocket client for upstream %v: %w", cfg.Id, err) - } - } else if parsedUrl.Scheme == "grpc" || parsedUrl.Scheme == "grpc+bds" { - newClient, err = NewGrpcBdsClient( - appCtx, - &lg, - manager.projectId, - ups, - parsedUrl, - ) - if err != nil { - clientErr = fmt.Errorf("failed to create gRPC BDS client for upstream: %v", cfg.Id) - } - } else { - clientErr = fmt.Errorf("unsupported endpoint scheme: %v for upstream: %v", parsedUrl.Scheme, cfg.Id) + upstreamKey := common.UniqueUpstreamKey(ups) + cv, _ := manager.clientCreations.LoadOrStore(upstreamKey, &clientCreation{}) + creation := cv.(*clientCreation) + + creation.once.Do(func() { + lg := manager.logger.With().Str("upstreamId", cfg.Id).Logger() + var c ClientInterface + var cerr error + switch cfg.Type { + case common.UpstreamTypeEvm: + switch parsedUrl.Scheme { + case "http", "https": + c, cerr = NewGenericHttpJsonRpcClient( + appCtx, + &lg, + manager.projectId, + ups, + parsedUrl, + cfg.JsonRpc, + proxyPool, + manager.evmExtractor, + ) + if cerr != nil { + cerr = fmt.Errorf("failed to create HTTP client for upstream: %v: %w", cfg.Id, cerr) + } + case "ws", "wss": + c, cerr = NewWsJsonRpcClient( + appCtx, + &lg, + manager.projectId, + ups, + parsedUrl, + cfg.JsonRpc, + manager.evmExtractor, + ) + if cerr != nil { + cerr = fmt.Errorf("failed to create WebSocket client for upstream %v: %w", cfg.Id, cerr) + } + case "grpc", "grpc+bds": + c, cerr = NewGrpcBdsClient( + appCtx, + &lg, + manager.projectId, + ups, + parsedUrl, + ) + if cerr != nil { + cerr = fmt.Errorf("failed to create gRPC BDS client for upstream: %v: %w", cfg.Id, cerr) } - default: - clientErr = fmt.Errorf("unsupported upstream type: %v for upstream: %v", cfg.Type, cfg.Id) - } - - if clientErr == nil { - manager.clients.Store(common.UniqueUpstreamKey(ups), newClient) + cerr = fmt.Errorf("unsupported endpoint scheme: %v for upstream: %v", parsedUrl.Scheme, cfg.Id) } - }) - } + default: + cerr = fmt.Errorf("unsupported upstream type: %v for upstream: %v", cfg.Type, cfg.Id) + } + creation.client = c + creation.err = cerr + if cerr == nil { + manager.clients.Store(upstreamKey, c) + } + }) - return newClient, clientErr + return creation.client, creation.err } diff --git a/common/upstream.go b/common/upstream.go index 2aab036ca..7f98697eb 100644 --- a/common/upstream.go +++ b/common/upstream.go @@ -4,6 +4,7 @@ import ( "context" "crypto/sha256" "encoding/hex" + "sort" "github.com/rs/zerolog" ) @@ -56,20 +57,39 @@ type Upstream interface { IgnoreMethod(method string) } -// UniqueUpstreamKey returns a unique hash for an upstream. -// It is used to identify the upstream uniquely in shared-state storage. -// Sometimes ID might not be enough for example if user changes the endpoint to a completely different network. +// UniqueUpstreamKey returns a stable hash for an upstream, derived only from +// config fields that don't change after the upstream is constructed. +// +// Why: this key is the dedup key for the per-upstream client cache +// (clients/registry.go) and for shared-state counters (latestBlock, +// finalizedBlock, earliestBlock by probe). If the key changes during the +// upstream's lifetime, those caches are bypassed — every key flip leaks a +// client (with its goroutines) and a counter-sync goroutine, and a fresh +// pod's view of the latest/finalized block diverges from the cluster's. +// +// Two prior bugs we fix here: +// 1. up.NetworkId() returns "n/a" before registration and the real id +// after. The endpoint already disambiguates which network the upstream +// points at, so NetworkId is redundant — and including it changed the +// key mid-lifetime. +// 2. Iterating cfg.JsonRpc.Headers in map order is non-deterministic, so +// two calls in the same process could hash to different values for the +// same headers. Sort by key before hashing. func UniqueUpstreamKey(up Upstream) string { sha := sha256.New() cfg := up.Config() sha.Write([]byte(cfg.Id)) sha.Write([]byte(cfg.Endpoint)) - sha.Write([]byte(up.NetworkId())) - if cfg.JsonRpc != nil && cfg.JsonRpc.Headers != nil { - for k, v := range cfg.JsonRpc.Headers { + if cfg.JsonRpc != nil && len(cfg.JsonRpc.Headers) > 0 { + keys := make([]string, 0, len(cfg.JsonRpc.Headers)) + for k := range cfg.JsonRpc.Headers { + keys = append(keys, k) + } + sort.Strings(keys) + for _, k := range keys { sha.Write([]byte(k)) - sha.Write([]byte(v)) + sha.Write([]byte(cfg.JsonRpc.Headers[k])) } } diff --git a/common/upstream_test.go b/common/upstream_test.go new file mode 100644 index 000000000..08f48cca7 --- /dev/null +++ b/common/upstream_test.go @@ -0,0 +1,89 @@ +package common + +import ( + "testing" +) + +// stubUpstreamForKey is a minimal Upstream that only supports the call +// UniqueUpstreamKey actually makes (Config()). NetworkId is parameterised +// so we can prove the key does NOT depend on it. +type stubUpstreamForKey struct { + cfg *UpstreamConfig + networkId string + Upstream // embed nil interface; we only use Config and NetworkId +} + +func (s *stubUpstreamForKey) Config() *UpstreamConfig { return s.cfg } +func (s *stubUpstreamForKey) NetworkId() string { return s.networkId } + +func TestUniqueUpstreamKey_StableAcrossNetworkIdChanges(t *testing.T) { + cfg := &UpstreamConfig{Id: "u1", Endpoint: "https://example/0"} + + // Before registration, NetworkId returns "n/a"; after registration it's + // the real id. Both calls must produce the same key, otherwise the + // per-upstream client cache and shared-state counters get duplicated + // across the upstream's lifecycle. + pre := UniqueUpstreamKey(&stubUpstreamForKey{cfg: cfg, networkId: "n/a"}) + post := UniqueUpstreamKey(&stubUpstreamForKey{cfg: cfg, networkId: "evm:1"}) + + if pre != post { + t.Fatalf("UniqueUpstreamKey changed when NetworkId changed: pre=%q post=%q", pre, post) + } +} + +func TestUniqueUpstreamKey_DeterministicAcrossHeaderOrder(t *testing.T) { + mk := func(headers map[string]string) string { + cfg := &UpstreamConfig{ + Id: "u1", + Endpoint: "https://example/0", + JsonRpc: &JsonRpcUpstreamConfig{Headers: headers}, + } + return UniqueUpstreamKey(&stubUpstreamForKey{cfg: cfg, networkId: "evm:1"}) + } + + // Go map iteration is randomised, so the previous SHA-update order was + // non-deterministic per call. Many runs against semantically identical + // header maps must all produce the same key. + headers := map[string]string{ + "Authorization": "Bearer xyz", + "X-Tenant": "tenant-a", + "X-Region": "eu-west-1", + "X-Custom": "value", + } + first := mk(headers) + for i := range 50 { + got := mk(headers) + if got != first { + t.Fatalf("UniqueUpstreamKey non-deterministic across map iterations: first=%q got=%q (iter %d)", first, got, i) + } + } +} + +func TestUniqueUpstreamKey_DistinctConfigsProduceDistinctKeys(t *testing.T) { + base := &UpstreamConfig{Id: "u1", Endpoint: "https://example/0"} + keyBase := UniqueUpstreamKey(&stubUpstreamForKey{cfg: base, networkId: "evm:1"}) + + cases := []struct { + name string + cfg *UpstreamConfig + }{ + {"different id", &UpstreamConfig{Id: "u2", Endpoint: "https://example/0"}}, + {"different endpoint", &UpstreamConfig{Id: "u1", Endpoint: "https://example/1"}}, + {"add headers", &UpstreamConfig{ + Id: "u1", Endpoint: "https://example/0", + JsonRpc: &JsonRpcUpstreamConfig{Headers: map[string]string{"X": "1"}}, + }}, + {"different header value", &UpstreamConfig{ + Id: "u1", Endpoint: "https://example/0", + JsonRpc: &JsonRpcUpstreamConfig{Headers: map[string]string{"X": "2"}}, + }}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + got := UniqueUpstreamKey(&stubUpstreamForKey{cfg: tc.cfg, networkId: "evm:1"}) + if got == keyBase { + t.Fatalf("expected distinct key, got same as base: %q", got) + } + }) + } +} diff --git a/erpc/config_analyzer.go b/erpc/config_analyzer.go index 4a92eb510..7e46ac8ab 100644 --- a/erpc/config_analyzer.go +++ b/erpc/config_analyzer.go @@ -281,6 +281,13 @@ func GenerateValidationReport(ctx context.Context, cfg *common.Config) *Validati // Upstream runtime checks (chain id + block hash comparisons). Use a silent logger and short timeout per upstream silent := zerolog.New(io.Discard) + // Scope all transient upstreams created below to a validation-only ctx. + // Their client goroutines (HTTP shutdown waiter, WS read/ping loops) + // listen on this ctx — without scoping, they outlive validation and + // accumulate forever. + valCtx, cancelVal := context.WithCancel(ctx) + defer cancelVal() + // Histogram buckets (validate config value) if err := telemetry.SetHistogramBuckets(cfg.Metrics.HistogramBuckets); err != nil { report.Errors = append(report.Errors, fmt.Sprintf("invalid metrics histogramBuckets: %v", err)) @@ -328,7 +335,7 @@ func GenerateValidationReport(ctx context.Context, cfg *common.Config) *Validati } clReg := clients.NewClientRegistry(&silent, project.Id, prxPool, evm.NewJsonRpcErrorExtractor()) vndReg := thirdparty.NewVendorsRegistry() - rlr, err := upstream.NewRateLimitersRegistry(ctx, cfg.RateLimiters, &silent) + rlr, err := upstream.NewRateLimitersRegistry(valCtx, cfg.RateLimiters, &silent) if err != nil { appendErr(fmt.Sprintf("project=%s failed to create rate limiters registry: %v", project.Id, err)) continue @@ -347,7 +354,7 @@ func GenerateValidationReport(ctx context.Context, cfg *common.Config) *Validati } // Create upstream - ups, err := upstream.NewUpstream(ctx, prj, uc, clReg, rlr, vndReg, &silent, mt, nil) + ups, err := upstream.NewUpstream(valCtx, prj, uc, clReg, rlr, vndReg, &silent, mt, nil) if err != nil { appendErr(fmt.Sprintf("project=%s upstream=%s failed to create upstream: %v", prj, uc.Id, err)) return @@ -979,6 +986,13 @@ func printConfigStats(logger zerolog.Logger, stats ConfigStats) { } func validateUpstreamEndpoints(ctx context.Context, cfg *common.Config, logger zerolog.Logger) error { + // The Upstreams we construct below spawn long-lived client goroutines + // (HTTP shutdown waiter, WS read/ping loops) that listen on the appCtx + // passed into NewUpstream. If we hand them the caller's ctx, they + // outlive validation and accumulate forever. Scope them to a + // validation-only ctx that we cancel on return. + valCtx, cancelVal := context.WithCancel(ctx) + defer cancelVal() err := telemetry.SetHistogramBuckets( cfg.Metrics.HistogramBuckets, ) @@ -1005,7 +1019,7 @@ func validateUpstreamEndpoints(ctx context.Context, cfg *common.Config, logger z ) vndReg := thirdparty.NewVendorsRegistry() rlr, err := upstream.NewRateLimitersRegistry( - ctx, + valCtx, cfg.RateLimiters, &logger, ) @@ -1041,7 +1055,7 @@ func validateUpstreamEndpoints(ctx context.Context, cfg *common.Config, logger z continue } ups, err := upstream.NewUpstream( - ctx, + valCtx, project.Id, upsCfg, clReg, @@ -1054,7 +1068,7 @@ func validateUpstreamEndpoints(ctx context.Context, cfg *common.Config, logger z if err != nil { return fmt.Errorf("failed to create upstream for project: \"%s\" and upstream id: \"%s\": %w", project.Id, upsCfg.Id, err) } - chainStr, err := ups.EvmGetChainId(ctx) + chainStr, err := ups.EvmGetChainId(valCtx) if err != nil { return fmt.Errorf("failed to get chain id for project: \"%s\" and upstream id: \"%s\": %w", project.Id, upsCfg.Id, err) } diff --git a/erpc/networks_earliest_detection_test.go b/erpc/networks_earliest_detection_test.go index 92a26b25b..01b1655e5 100644 --- a/erpc/networks_earliest_detection_test.go +++ b/erpc/networks_earliest_detection_test.go @@ -613,12 +613,13 @@ func TestEarliestDetection_StaleHighValueInSharedState(t *testing.T) { sharedStateCfg.SetDefaults("test") ssr, _ := data.NewSharedStateRegistry(ctx, &log.Logger, sharedStateCfg) - // Pre-compute the key that will be used for earliest block storage - // This matches what UniqueUpstreamKey computes: id + "/" + sha256(id + endpoint + networkId) + // Pre-compute the key that will be used for earliest block storage. + // This matches what UniqueUpstreamKey computes: + // id + "/" + sha256(id + endpoint + sorted(headers)) + // (No networkId — cfg.Endpoint already disambiguates across networks.) sha := sha256.New() sha.Write([]byte("rpc1")) // upCfg.Id sha.Write([]byte("http://rpc1.localhost")) // upCfg.Endpoint - sha.Write([]byte("evm:123")) // networkId uniqueKey := "rpc1/" + hex.EncodeToString(sha.Sum(nil)) presetKey := fmt.Sprintf("earliestBlock/%s/blockHeader", uniqueKey) t.Logf("Pre-computed key: %s", presetKey) diff --git a/go.mod b/go.mod index 23efa7edb..2abc68aa0 100644 --- a/go.mod +++ b/go.mod @@ -19,6 +19,7 @@ require ( github.com/go-logr/zerologr v1.2.3 github.com/go-redsync/redsync/v4 v4.15.0 github.com/golang-jwt/jwt/v4 v4.5.2 + github.com/gorilla/websocket v1.5.3 github.com/grafana/sobek v0.0.0-20241024150027-d91f02b05e9b github.com/h2non/gock v1.2.0 github.com/jackc/pgconn v1.14.3 @@ -98,7 +99,6 @@ require ( github.com/google/pprof v0.0.0-20240727154555-813a5fbdbec8 // indirect github.com/google/uuid v1.6.0 // indirect github.com/gorilla/mux v1.8.1 // indirect - github.com/gorilla/websocket v1.5.3 // indirect github.com/grpc-ecosystem/go-grpc-middleware v1.4.0 // indirect github.com/grpc-ecosystem/grpc-gateway/v2 v2.27.7 // indirect github.com/h2non/parth v0.0.0-20190131123155-b4df798d6542 // indirect diff --git a/upstream/upstream.go b/upstream/upstream.go index b69c224b3..112547afd 100644 --- a/upstream/upstream.go +++ b/upstream/upstream.go @@ -129,6 +129,11 @@ type Upstream struct { rateLimitersRegistry *RateLimitersRegistry rateLimiterAutoTuner *RateLimitAutoTuner evmStatePoller common.EvmStatePoller + // Guards lazy creation of evmStatePoller so concurrent or repeated + // Bootstrap calls don't construct a second poller — every prior poller's + // goroutine listens on appCtx and never exits, so reassigning leaks one + // goroutine per call. + evmStatePollerMu sync.Mutex // True after successful chainId detection/validation; enables short-circuit in EvmGetChainId. chainIdValidated atomic.Bool } @@ -228,16 +233,21 @@ func (u *Upstream) Bootstrap(ctx context.Context) error { } if u.config.Type == common.UpstreamTypeEvm { - u.evmStatePoller = evm.NewEvmStatePoller(u.ProjectId, u.appCtx, u.logger, u, u.metricsTracker, u.sharedStateRegistry) - } - - if u.evmStatePoller != nil { - err = u.evmStatePoller.Bootstrap(ctx) - if err != nil { - // The reason we're not returning error is to allow upstream to still be registered - // even if background block polling fails initially. - u.logger.Error().Err(err).Msg("failed on initial bootstrap of evm state poller (will retry in background)") + // Guard against repeated calls. Bootstrap can be re-entered when the + // owning task is retried (e.g. provider regeneration re-submits the + // same upstream task name); without this check the previous + // EvmStatePoller's polling goroutine would be orphaned and live + // until appCtx shutdown. + u.evmStatePollerMu.Lock() + if u.evmStatePoller == nil { + u.evmStatePoller = evm.NewEvmStatePoller(u.ProjectId, u.appCtx, u.logger, u, u.metricsTracker, u.sharedStateRegistry) + if perr := u.evmStatePoller.Bootstrap(ctx); perr != nil { + // The reason we're not returning error is to allow upstream to still be registered + // even if background block polling fails initially. + u.logger.Error().Err(perr).Msg("failed on initial bootstrap of evm state poller (will retry in background)") + } } + u.evmStatePollerMu.Unlock() } return nil From bf9f50b7ff0e1533c667812756bb6165309cb8fe Mon Sep 17 00:00:00 2001 From: Jonny Date: Thu, 30 Apr 2026 11:55:06 +0100 Subject: [PATCH 04/40] fix(util): pooledGzipReadCloser closes its source on Close MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The wrapper previously closed only the gzip.Reader and returned it to the pool; the underlying io.ReadCloser source was never closed. Single-request HTTP client callers in clients/http_json_rpc_client.go attached resp.Body via WrapGzipReader and relied on the wrapper to close it later, but that close never propagated to the http transport. Concretely, the docstring on the call site — "DO NOT close resp.Body here - it will be closed by NormalizedResponse after reading" was only true for non-gzipped responses (where bodyReader = resp.Body directly). For gzipped responses the resp.Body was never closed at all once the gzip wrapper was the head of the chain. Over time that leaks http2ClientConn.readLoop goroutines as net/http keeps opening new connections instead of reusing pinned ones. Have the wrapper take an optional source Closer and close it from Close after the gzip.Reader is returned to the pool. Update the single-request caller to pass resp.Body. The batch-path readResponseBody is unaffected because it already manages resp.Body lifetime via an explicit defer. Unit tests cover: source is closed exactly once, Close is idempotent under contention, nil source is accepted, source-close errors propagate. --- clients/http_json_rpc_client.go | 8 +- common/sonic.go | 13 +++- util/gzip_pool.go | 35 +++++++-- util/gzip_pool_close_test.go | 133 ++++++++++++++++++++++++++++++++ 4 files changed, 178 insertions(+), 11 deletions(-) create mode 100644 util/gzip_pool_close_test.go diff --git a/clients/http_json_rpc_client.go b/clients/http_json_rpc_client.go index de864aff5..3a2574d6d 100644 --- a/clients/http_json_rpc_client.go +++ b/clients/http_json_rpc_client.go @@ -758,7 +758,8 @@ func (c *GenericHttpJsonRpcClient) sendSingleRequest(ctx context.Context, req *c } return nil, common.NewErrEndpointTransportFailure(c.Url, err) } - // DO NOT close resp.Body here - it will be closed by NormalizedResponse after reading + // DO NOT close resp.Body here - the wrapper owns it (gzip path) or + // NormalizedResponse closes it directly (non-gzip path) after reading. var bodyReader io.ReadCloser = resp.Body if resp.Header.Get("Content-Encoding") == "gzip" { @@ -767,7 +768,10 @@ func (c *GenericHttpJsonRpcClient) sendSingleRequest(ctx context.Context, req *c _ = resp.Body.Close() // Must close on error path return nil, common.NewErrEndpointTransportFailure(c.Url, fmt.Errorf("cannot create gzip reader: %w", err)) } - bodyReader = c.gzipPool.WrapGzipReader(gzReader) + // Pass resp.Body so the wrapper closes it on Close — without this, + // gzip.Reader.Close leaves the underlying http stream open and the + // transport keeps the conn pinned. See pooledGzipReadCloser docs. + bodyReader = c.gzipPool.WrapGzipReader(gzReader, resp.Body) } nr := common.NewNormalizedResponse(). diff --git a/common/sonic.go b/common/sonic.go index 44e0c13f8..27f3c4c82 100644 --- a/common/sonic.go +++ b/common/sonic.go @@ -46,7 +46,18 @@ func init() { panic(err) } SonicCfg = sonic.Config{ - CopyString: false, + // CopyString must be true so that strings produced by Unmarshal + // own their backing memory. Sonic's default (false) constructs + // string headers that alias the source byte slice — fast, but + // any string retained beyond the lifetime of the source buffer + // keeps that entire buffer alive. The indexer stores parsed + // block hashes (and similar small strings) in long-lived ring + // buffers and dedup windows; with aliasing, each retained + // string pins its multi-KB source payload, turning short-lived + // notification buffers into a slow accumulating leak that + // scales with throughput rather than with the data actually + // kept. Paying the per-string copy is the correct trade. + CopyString: true, NoNullSliceOrMap: true, NoQuoteTextMarshaler: true, NoValidateJSONMarshaler: true, diff --git a/util/gzip_pool.go b/util/gzip_pool.go index 73f8f716d..81a70cda3 100644 --- a/util/gzip_pool.go +++ b/util/gzip_pool.go @@ -49,11 +49,20 @@ func (p *GzipReaderPool) Put(zr *gzip.Reader) { p.pool.Put(zr) } -// pooledGzipReadCloser wraps a gzip.Reader so that closing it returns it to the pool. +// pooledGzipReadCloser wraps a gzip.Reader so that closing it returns the +// reader to the pool AND closes the underlying source. +// +// The source close is load-bearing: when the source is an http.Response.Body, +// gzip.Reader.Close only releases the gzip-decoder's internal state — it does +// not propagate to the underlying transport. Without an explicit source close +// the HTTP/2 stream stays open, the connection is pinned, and net/http opens +// a new conn for the next request, leaking http2ClientConn.readLoop +// goroutines over time. type pooledGzipReadCloser struct { - zr *gzip.Reader - pool *GzipReaderPool - once sync.Once + zr *gzip.Reader + pool *GzipReaderPool + source io.Closer + once sync.Once } func (pgrc *pooledGzipReadCloser) Read(b []byte) (int, error) { return pgrc.zr.Read(b) } @@ -61,19 +70,29 @@ func (pgrc *pooledGzipReadCloser) Read(b []byte) (int, error) { return pgrc.zr.R func (pgrc *pooledGzipReadCloser) Close() error { var err error pgrc.once.Do(func() { - // Close underlying gzip reader first, then return to pool exactly once. + // Close underlying gzip reader first, return it to the pool, then + // close the source so the http transport can release the stream. err = pgrc.zr.Close() pgrc.pool.Put(pgrc.zr) + if pgrc.source != nil { + if cerr := pgrc.source.Close(); cerr != nil && err == nil { + err = cerr + } + } // Clear references to avoid accidental reuse and help GC pgrc.zr = nil pgrc.pool = nil + pgrc.source = nil }) return err } -// WrapGzipReader returns an io.ReadCloser wrapper that will return the gzip.Reader to pool on Close. -func (p *GzipReaderPool) WrapGzipReader(zr *gzip.Reader) io.ReadCloser { - return &pooledGzipReadCloser{zr: zr, pool: p} +// WrapGzipReader returns an io.ReadCloser wrapper that will return the +// gzip.Reader to pool on Close, and close source. source may be nil for +// callers that close the source themselves (e.g., via defer); when non-nil +// the wrapper owns its lifetime. +func (p *GzipReaderPool) WrapGzipReader(zr *gzip.Reader, source io.Closer) io.ReadCloser { + return &pooledGzipReadCloser{zr: zr, pool: p, source: source} } // GzipWriterPool wraps a sync.Pool for gzip.Writer with helpers to reset writers diff --git a/util/gzip_pool_close_test.go b/util/gzip_pool_close_test.go new file mode 100644 index 000000000..a19b2d3d0 --- /dev/null +++ b/util/gzip_pool_close_test.go @@ -0,0 +1,133 @@ +package util + +import ( + "bytes" + "compress/gzip" + "errors" + "io" + "sync" + "sync/atomic" + "testing" +) + +// countingCloser tracks Close call count + records prior Close error if any. +type countingCloser struct { + io.Reader + closes atomic.Int32 + closeErr error +} + +func (c *countingCloser) Close() error { + c.closes.Add(1) + return c.closeErr +} + +func gzipped(t *testing.T, payload string) io.Reader { + t.Helper() + var buf bytes.Buffer + zw := gzip.NewWriter(&buf) + if _, err := zw.Write([]byte(payload)); err != nil { + t.Fatal(err) + } + if err := zw.Close(); err != nil { + t.Fatal(err) + } + return &buf +} + +// pooledGzipReadCloser.Close MUST close the underlying source. +// Regression test for the leak in clients/http_json_rpc_client.go single- +// request path: gzip wrapper close used to leave resp.Body open, pinning the +// HTTP/2 stream and accumulating http2ClientConn.readLoop goroutines. +func TestPooledGzipReadCloser_ClosesSource(t *testing.T) { + pool := NewGzipReaderPool() + src := &countingCloser{Reader: gzipped(t, "hello")} + + zr, err := pool.GetReset(src) + if err != nil { + t.Fatalf("GetReset: %v", err) + } + rc := pool.WrapGzipReader(zr, src) + + // Drain + if _, err := io.ReadAll(rc); err != nil { + t.Fatalf("read: %v", err) + } + + if err := rc.Close(); err != nil { + t.Fatalf("close: %v", err) + } + + if got := src.closes.Load(); got != 1 { + t.Fatalf("expected source.Close called exactly once, got %d", got) + } +} + +// Close must be idempotent across many goroutines and only Close the source +// once even under contention. +func TestPooledGzipReadCloser_CloseIdempotent(t *testing.T) { + pool := NewGzipReaderPool() + src := &countingCloser{Reader: gzipped(t, "hello")} + + zr, err := pool.GetReset(src) + if err != nil { + t.Fatalf("GetReset: %v", err) + } + rc := pool.WrapGzipReader(zr, src) + + if _, err := io.ReadAll(rc); err != nil { + t.Fatalf("read: %v", err) + } + + var wg sync.WaitGroup + for range 32 { + wg.Add(1) + go func() { defer wg.Done(); _ = rc.Close() }() + } + wg.Wait() + + if got := src.closes.Load(); got != 1 { + t.Fatalf("expected source.Close called exactly once even under contention, got %d", got) + } +} + +// Source is optional — passing nil is supported (some callers manage the +// source lifetime themselves) and Close should not panic. +func TestPooledGzipReadCloser_NilSourceOk(t *testing.T) { + pool := NewGzipReaderPool() + src := gzipped(t, "hello") + + zr, err := pool.GetReset(src) + if err != nil { + t.Fatalf("GetReset: %v", err) + } + rc := pool.WrapGzipReader(zr, nil) + + if _, err := io.ReadAll(rc); err != nil { + t.Fatalf("read: %v", err) + } + if err := rc.Close(); err != nil { + t.Fatalf("close with nil source: %v", err) + } +} + +// If the source's Close errors, the wrapper surfaces that error (preferring +// the gzip-reader error if both fail, since gzip is closed first). +func TestPooledGzipReadCloser_PropagatesSourceError(t *testing.T) { + pool := NewGzipReaderPool() + wantErr := errors.New("source close failed") + src := &countingCloser{Reader: gzipped(t, "hello"), closeErr: wantErr} + + zr, err := pool.GetReset(src) + if err != nil { + t.Fatalf("GetReset: %v", err) + } + rc := pool.WrapGzipReader(zr, src) + if _, err := io.ReadAll(rc); err != nil { + t.Fatalf("read: %v", err) + } + + if err := rc.Close(); !errors.Is(err, wantErr) { + t.Fatalf("expected source close error to propagate, got %v", err) + } +} From 51faededebef39717ed503b72dc29aa1aa87df2c Mon Sep 17 00:00:00 2001 From: Jonny Date: Fri, 8 May 2026 08:56:59 +0100 Subject: [PATCH 05/40] fix(networks): clamp regressions in EvmHighestBlockNumber monotonic guard MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A prior cleanup pass dropped the lastReturned*Block atomic.Int64 fields on Network and the matching CAS-bump + WARN logic at the tail of evmHighestBlockNumber. Without it, a downstream client polling EvmHighestFinalizedBlockNumber through a load balancer can observe a returned value lower than a previously returned value (e.g. when the aggregator transiently picks up a lagging upstream after a restart, or when the cross-cluster shared counter is briefly stale relative to the local primaryMax). Strict downstream consumers treat that as a data-integrity violation, even though the underlying chain is fine. Restore the per-tag atomic high-water mark and *clamp* regressions back to that high-water mark instead of returning the regressed value with only a warning. Clamping is safe because we never invent a number — the floor is always something this Network has already returned upstream. On forward progress we CAS-bump the high-water mark; on regression we log the inputs (localMax, primaryMax, fallbackMax, anyPrimaryUp, sharedVal, sharedNil) so future regressions are diagnosable, and return the previously-returned value. --- erpc/networks.go | 226 +++++++++++++++++++++++++- erpc/networks_block_partition_test.go | 101 ++++++++++++ erpc/networks_mono_guard_test.go | 124 ++++++++++++++ erpc/networks_multiplexer_test.go | 32 ++++ 4 files changed, 478 insertions(+), 5 deletions(-) create mode 100644 erpc/networks_block_partition_test.go create mode 100644 erpc/networks_mono_guard_test.go diff --git a/erpc/networks.go b/erpc/networks.go index 9c84b138a..6f5e5a60f 100644 --- a/erpc/networks.go +++ b/erpc/networks.go @@ -9,6 +9,7 @@ import ( "slices" "strings" "sync" + "sync/atomic" "time" "github.com/erpc/erpc/architecture/evm" @@ -55,6 +56,14 @@ type Network struct { // failure. May be nil in tests or when shared state is unavailable. latestBlockShared data.CounterInt64SharedVariable finalizedBlockShared data.CounterInt64SharedVariable + + // Last value this Network has ever returned from evmHighestBlockNumber + // for each of the monotonic tags. If a subsequent call computes a lower + // value we WARN with the local/shared inputs so we can see whether the + // regression originated in the per-upstream poller max, the cross-cluster + // shared counter, or a race between them. + lastReturnedLatestBlock atomic.Int64 + lastReturnedFinalizedBlock atomic.Int64 } // Bootstrap registers this network with the policy engine. The engine kicks @@ -269,6 +278,8 @@ func (n *Network) EvmHighestLatestBlockNumber(ctx context.Context) int64 { "eth_blockNumber", (*upstream.Upstream).EvmEffectiveLatestBlock, n.latestBlockShared, + &n.lastReturnedLatestBlock, + "latest", ) span.SetAttributes(attribute.Int64("highest_latest_block", result)) return result @@ -285,6 +296,8 @@ func (n *Network) EvmHighestFinalizedBlockNumber(ctx context.Context) int64 { "eth_getBlockByNumber", (*upstream.Upstream).EvmEffectiveFinalizedBlock, n.finalizedBlockShared, + &n.lastReturnedFinalizedBlock, + "finalized", ) span.SetAttributes(attribute.Int64("highest_finalized_block", result)) return result @@ -310,6 +323,8 @@ func (n *Network) evmHighestBlockNumber( selectionMethod string, blockOf func(*upstream.Upstream) int64, shared data.CounterInt64SharedVariable, + lastReturned *atomic.Int64, + tag string, ) int64 { var primaryMax, fallbackMax int64 anyPrimaryUp := false @@ -341,14 +356,86 @@ func (n *Network) evmHighestBlockNumber( localMax = fallbackMax } + var result int64 + var sharedVal int64 if shared == nil { - return localMax + result = localMax + } else { + sharedVal = shared.GetValue() + if localMax > sharedVal { + result = shared.TryUpdate(ctx, localMax) + } else { + result = sharedVal + } } - sharedVal := shared.GetValue() - if localMax > sharedVal { - return shared.TryUpdate(ctx, localMax) + + return n.applyMonotonicityGuard(result, lastReturned, tag, monoGuardInputs{ + localMax: localMax, + primaryMax: primaryMax, + fallbackMax: fallbackMax, + anyPrimaryUp: anyPrimaryUp, + sharedVal: sharedVal, + sharedNil: shared == nil, + }) +} + +// monoGuardInputs carries the diagnostic inputs the monotonicity guard logs +// when it has to clamp. They're never used to compute the clamp itself — +// it's purely "here's the aggregator state at the moment of the regression" +// so future investigations can tell whether the regression came from the +// per-upstream poller max, the shared counter, or a race between them. +type monoGuardInputs struct { + localMax int64 + primaryMax int64 + fallbackMax int64 + anyPrimaryUp bool + sharedVal int64 + sharedNil bool +} + +// applyMonotonicityGuard clamps result to the per-tag high-water mark held +// in lastReturned. On forward progress it CAS-bumps the high-water mark; on +// regression it logs the inputs and returns the previously-returned value. +// Clamping is safe because we never invent a number — the floor is always +// something this Network has already returned upstream. +// +// Strict downstream consumers treat any backwards step in +// `latest`/`finalized` as a data-integrity violation, so the guard exists to +// keep returns monotonic across transient mid-rollout effects (e.g. mixed +// versions reading different shared-state slots) and aggregator races. +func (n *Network) applyMonotonicityGuard(result int64, lastReturned *atomic.Int64, tag string, in monoGuardInputs) int64 { + if lastReturned == nil { + return result + } + prev := lastReturned.Load() + if result < prev { + n.logger.Warn(). + Str("networkId", n.networkId). + Str("tag", tag). + Int64("previouslyReturned", prev). + Int64("nowReturning", result). + Int64("delta", result-prev). + Int64("localMax", in.localMax). + Int64("primaryMax", in.primaryMax). + Int64("fallbackMax", in.fallbackMax). + Bool("anyPrimaryUp", in.anyPrimaryUp). + Int64("sharedVal", in.sharedVal). + Bool("sharedNil", in.sharedNil). + Msg("evmHighestBlockNumber would have regressed; clamping to previously returned value") + return prev + } + if result > prev { + for { + cur := lastReturned.Load() + if result <= cur { + break + } + if lastReturned.CompareAndSwap(cur, result) { + break + } + } } - return sharedVal + return result } func (n *Network) EvmLowestFinalizedBlockNumber(ctx context.Context) int64 { @@ -549,6 +636,22 @@ func (n *Network) Forward(ctx context.Context, req *common.NormalizedRequest) (* return nil, err } + // Block-availability-aware routing: when the request targets a specific + // block, prefer upstreams whose state poller has already observed it. + // Without this, requests for a block we just delivered to a client via + // WS would still get routed to an HTTP-only sibling whose own polling + // loop hasn't caught up — checkUpstreamBlockAvailability rejects with + // ErrUpstreamBlockUnavailable, retries cascade through the rest of the + // lagging siblings, and the request can fail entirely despite the WS + // upstream demonstrably having the block. partitionUpstreamsByLatestBlock + // is stable so it composes with the tier and score orderings layered + // on top. + if n.Architecture() == common.ArchitectureEvm { + if bn := requestBlockNumber(ctx, req); bn > 0 { + upsList = partitionUpstreamsByLatestBlock(upsList, bn) + } + } + // Failover tiering: when enabled, order default-group upstreams ahead of // fallback-group ones while preserving score order within each tier. The // network request loop then naturally tries defaults first and only @@ -1475,11 +1578,41 @@ func (n *Network) checkUpstreamBlockAvailability(ctx context.Context, u common.U return nil, false } +// methodSkipMultiplexing reports whether a JSON-RPC method should bypass +// in-flight dedup. Multiplexing is correct for idempotent reads where N +// concurrent identical requests can share one result. It is wrong for +// transaction-broadcast methods: a client retrying a signed payload +// expects each retry to actually re-submit, not to silently block on a +// prior in-flight attempt. Worse, if a retrying client cancels at its +// own context deadline before the leader returns, eRPC's failsafe +// leader continues executing on its CopyForCancellable child for the +// remainder of the network-level budget — and every retry the client +// issues in that window stacks up as a follower of the doomed leader, +// all timing out together when their parent contexts fire. +// +// Tx broadcasts already handle duplicate submissions via the idempotent- +// broadcast post-forward hook (nonce-already-known → synthetic success +// keyed by tx hash). Skipping multiplexing here costs nothing in +// correctness: duplicate work resolves at the upstream layer with the +// same guarantees, and retries are no longer serialised behind a slow +// leader. +func methodSkipMultiplexing(method string) bool { + switch method { + case "eth_sendRawTransaction", "eth_sendTransaction": + return true + } + return false +} + func (n *Network) handleMultiplexing(ctx context.Context, lg *zerolog.Logger, req *common.NormalizedRequest, startTime time.Time) (*Multiplexer, *common.NormalizedResponse, error) { if !n.cfg.MultiplexingEnabled() { return nil, nil, nil } + if method, _ := req.Method(); methodSkipMultiplexing(method) { + return nil, nil, nil + } + mlxHash, err := req.CacheHash() lg.Trace().Str("hash", mlxHash).Object("request", req).Msgf("checking if multiplexing is possible") if err != nil || mlxHash == "" { @@ -1827,3 +1960,86 @@ func tierUpstreamsByGroup(ups []common.Upstream) []common.Upstream { } return tiered } + +// partitionUpstreamsByLatestBlock stable-partitions ups so that upstreams +// whose EvmStatePoller has observed a block number ≥ bn come first, in +// their input order. Upstreams whose poller is behind bn (or has no +// poller / isn't EVM) keep their input order and come after. +// +// This makes per-request routing consistent with what the indexer already +// knows: when a WS upstream delivers newHead block N, its state poller's +// LatestBlock advances to N immediately. If a subsequent eth_call from a +// client references block N, this partition routes the call to that +// upstream first instead of to an HTTP-only sibling whose own polling +// hasn't caught up — which would fail checkUpstreamBlockAvailability and +// burn retries. The partition is stable, so callers can layer it under +// other orderings (tier, score) without disturbing them within each +// partition. +func partitionUpstreamsByLatestBlock(ups []common.Upstream, bn int64) []common.Upstream { + if bn <= 0 || len(ups) < 2 { + return ups + } + // Cheap fast-path: if all upstreams already have ≥ bn, or all are + // behind, the partition is the identity and we can avoid the + // allocation. We still walk once to compute LatestBlock either way, + // so the gain is just the slice copy. + haveCount := 0 + for _, u := range ups { + eu, ok := u.(common.EvmUpstream) + if !ok { + continue + } + sp := eu.EvmStatePoller() + if sp == nil || sp.IsObjectNull() { + continue + } + if sp.LatestBlock() >= bn { + haveCount++ + } + } + if haveCount == 0 || haveCount == len(ups) { + return ups + } + out := make([]common.Upstream, 0, len(ups)) + for _, u := range ups { + eu, ok := u.(common.EvmUpstream) + if !ok { + continue + } + sp := eu.EvmStatePoller() + if sp != nil && !sp.IsObjectNull() && sp.LatestBlock() >= bn { + out = append(out, u) + } + } + for _, u := range ups { + eu, ok := u.(common.EvmUpstream) + if !ok { + out = append(out, u) + continue + } + sp := eu.EvmStatePoller() + if sp == nil || sp.IsObjectNull() || sp.LatestBlock() < bn { + out = append(out, u) + } + } + return out +} + +// requestBlockNumber resolves the specific block number a request targets, +// or 0 when the request has no block reference. Mirrors the extraction +// path in checkUpstreamBlockAvailability so routing and gating see the +// same value. +func requestBlockNumber(ctx context.Context, req *common.NormalizedRequest) int64 { + if req == nil { + return 0 + } + if v := req.EvmBlockNumber(); v != nil { + if n64, ok := v.(int64); ok && n64 > 0 { + return n64 + } + } + if _, x, ebn := evm.ExtractBlockReferenceFromRequest(ctx, req); ebn == nil && x > 0 { + return x + } + return 0 +} diff --git a/erpc/networks_block_partition_test.go b/erpc/networks_block_partition_test.go new file mode 100644 index 000000000..34186179c --- /dev/null +++ b/erpc/networks_block_partition_test.go @@ -0,0 +1,101 @@ +package erpc + +import ( + "testing" + + "github.com/erpc/erpc/common" + "github.com/stretchr/testify/assert" +) + +// upstream helper: a FakeUpstream wired with a FakeEvmStatePoller at the +// given latest block. id encodes both identity and the original score +// position so test assertions can read order directly. +func upWithLatest(id string, latest int64) common.Upstream { + poller := common.NewFakeEvmStatePoller(latest, 0) + return common.NewFakeUpstream(id, common.WithEvmStatePoller(poller)) +} + +// upWithoutPoller returns a FakeUpstream with no state poller — the +// partition should treat it as "doesn't have the block" without panicking. +func upWithoutPoller(id string) common.Upstream { + return common.NewFakeUpstream(id) +} + +func ids(ups []common.Upstream) []string { + out := make([]string, len(ups)) + for i, u := range ups { + out[i] = u.Id() + } + return out +} + +func TestPartitionUpstreamsByLatestBlock_AllHaveTheBlock_NoReorder(t *testing.T) { + in := []common.Upstream{ + upWithLatest("a", 100), + upWithLatest("b", 105), + upWithLatest("c", 110), + } + got := partitionUpstreamsByLatestBlock(in, 100) + assert.Equal(t, []string{"a", "b", "c"}, ids(got), "no upstream lags the requested block; partition is identity") +} + +func TestPartitionUpstreamsByLatestBlock_NoneHaveTheBlock_NoReorder(t *testing.T) { + in := []common.Upstream{ + upWithLatest("a", 50), + upWithLatest("b", 60), + upWithLatest("c", 70), + } + got := partitionUpstreamsByLatestBlock(in, 100) + assert.Equal(t, []string{"a", "b", "c"}, ids(got), "no upstream has the requested block; partition is identity (caller can still try them)") +} + +func TestPartitionUpstreamsByLatestBlock_SplitsAndPreservesIntraGroupOrder(t *testing.T) { + // Mixed: a (behind), b (has), c (behind), d (has), e (behind). + // Expected: b, d, a, c, e. Within "has" group: b before d (input order). + // Within "lags" group: a, c, e (input order). + in := []common.Upstream{ + upWithLatest("a", 95), + upWithLatest("b", 110), + upWithLatest("c", 90), + upWithLatest("d", 105), + upWithLatest("e", 88), + } + got := partitionUpstreamsByLatestBlock(in, 100) + assert.Equal(t, []string{"b", "d", "a", "c", "e"}, ids(got)) +} + +func TestPartitionUpstreamsByLatestBlock_EqualLatestIsTreatedAsHavingBlock(t *testing.T) { + // Boundary: an upstream at exactly bn has the block. + in := []common.Upstream{ + upWithLatest("a", 99), + upWithLatest("b", 100), + upWithLatest("c", 100), + } + got := partitionUpstreamsByLatestBlock(in, 100) + assert.Equal(t, []string{"b", "c", "a"}, ids(got)) +} + +func TestPartitionUpstreamsByLatestBlock_UpstreamWithoutPollerSortsAsLagging(t *testing.T) { + in := []common.Upstream{ + upWithoutPoller("a"), + upWithLatest("b", 110), + upWithoutPoller("c"), + } + got := partitionUpstreamsByLatestBlock(in, 100) + assert.Equal(t, []string{"b", "a", "c"}, ids(got), "poller-less upstreams keep input order, partition keeps them after the upstream that demonstrably has the block") +} + +func TestPartitionUpstreamsByLatestBlock_ZeroOrNegativeBlockIsNoOp(t *testing.T) { + in := []common.Upstream{ + upWithLatest("a", 100), + upWithLatest("b", 50), + } + assert.Equal(t, in, partitionUpstreamsByLatestBlock(in, 0), "no block reference means no per-block routing preference") + assert.Equal(t, in, partitionUpstreamsByLatestBlock(in, -1), "negative block numbers are nonsensical; skip the partition rather than mis-sort") +} + +func TestPartitionUpstreamsByLatestBlock_SingleUpstreamIsNoOp(t *testing.T) { + in := []common.Upstream{upWithLatest("a", 50)} + got := partitionUpstreamsByLatestBlock(in, 100) + assert.Equal(t, in, got, "no other upstream to prefer; partition is a no-op") +} diff --git a/erpc/networks_mono_guard_test.go b/erpc/networks_mono_guard_test.go new file mode 100644 index 000000000..642369b04 --- /dev/null +++ b/erpc/networks_mono_guard_test.go @@ -0,0 +1,124 @@ +package erpc + +import ( + "sync" + "sync/atomic" + "testing" + + "github.com/rs/zerolog" + "github.com/stretchr/testify/assert" +) + +// newMonoGuardTestNetwork builds the minimal Network needed by +// applyMonotonicityGuard — only the logger and networkId are read by the +// guard itself. Other Network fields are deliberately left nil so the guard +// stays the only thing under test. +func newMonoGuardTestNetwork() *Network { + logger := zerolog.Nop() + return &Network{ + logger: &logger, + networkId: "evm:1234", + } +} + +func TestApplyMonotonicityGuard_NilLastReturnedReturnsResultUnchanged(t *testing.T) { + n := newMonoGuardTestNetwork() + got := n.applyMonotonicityGuard(42, nil, "finalized", monoGuardInputs{}) + assert.Equal(t, int64(42), got, "with no high-water mark to compare against, the result must pass through") +} + +func TestApplyMonotonicityGuard_FirstCallSeedsHighWaterMark(t *testing.T) { + n := newMonoGuardTestNetwork() + var hwm atomic.Int64 + got := n.applyMonotonicityGuard(100, &hwm, "finalized", monoGuardInputs{}) + assert.Equal(t, int64(100), got, "the first call returns its computed result") + assert.Equal(t, int64(100), hwm.Load(), "and seeds the high-water mark for future calls") +} + +func TestApplyMonotonicityGuard_ForwardProgressBumpsHighWaterMark(t *testing.T) { + n := newMonoGuardTestNetwork() + var hwm atomic.Int64 + hwm.Store(100) + + got := n.applyMonotonicityGuard(150, &hwm, "finalized", monoGuardInputs{}) + assert.Equal(t, int64(150), got, "forward progress is returned as-is") + assert.Equal(t, int64(150), hwm.Load(), "and bumps the high-water mark") +} + +func TestApplyMonotonicityGuard_EqualResultIsReturnedAndDoesNotMutateHwm(t *testing.T) { + n := newMonoGuardTestNetwork() + var hwm atomic.Int64 + hwm.Store(100) + + got := n.applyMonotonicityGuard(100, &hwm, "finalized", monoGuardInputs{}) + assert.Equal(t, int64(100), got) + assert.Equal(t, int64(100), hwm.Load(), "no mutation needed when result equals the existing high-water mark") +} + +// The behavior change vs the previous diagnostic-only guard: pre-clamp the +// function returned `result` even when result < prev, exposing strict +// downstream consumers to a backwards step. Post-clamp we return prev +// instead, masking transient regressions (typical cause: mid-rollout slot +// divergence). +func TestApplyMonotonicityGuard_RegressionIsClampedToHighWaterMark(t *testing.T) { + n := newMonoGuardTestNetwork() + var hwm atomic.Int64 + hwm.Store(1000) + + got := n.applyMonotonicityGuard(950, &hwm, "finalized", monoGuardInputs{ + localMax: 950, + primaryMax: 950, + fallbackMax: 0, + anyPrimaryUp: true, + sharedVal: 900, + sharedNil: false, + }) + assert.Equal(t, int64(1000), got, "regression must be clamped to the previously-returned value, not surfaced to callers") + assert.Equal(t, int64(1000), hwm.Load(), "high-water mark stays put on regression — it is by definition already at the floor we want") +} + +// The clamp is safe to apply across many concurrent callers: we never +// invent a value, and the worst case is two callers both observe the same +// pre-bump prev and one of them ends up returning a slightly stale value +// that's still >= the high-water mark, which is exactly the contract. +func TestApplyMonotonicityGuard_ConcurrentCallersAreMonotonic(t *testing.T) { + n := newMonoGuardTestNetwork() + var hwm atomic.Int64 + + const goroutines = 32 + const callsPer = 200 + + var minSeen atomic.Int64 + minSeen.Store(int64(^uint64(0) >> 1)) // math.MaxInt64 + + var wg sync.WaitGroup + wg.Add(goroutines) + for g := 0; g < goroutines; g++ { + go func(seed int64) { + defer wg.Done() + // Mix of advancing and regressing inputs. The guard must + // surface a non-decreasing sequence of returns to each + // caller's view. + var localPrev int64 + for i := 0; i < callsPer; i++ { + // Pseudo-random walk: half the calls advance, half regress. + var input int64 + if (seed+int64(i))%2 == 0 { + input = seed*1000 + int64(i) + } else { + input = seed*1000 - int64(i) + } + got := n.applyMonotonicityGuard(input, &hwm, "finalized", monoGuardInputs{}) + if got < localPrev { + // Critical invariant: the guard must never return a + // value lower than this caller's previous observation. + assert.GreaterOrEqualf(t, got, localPrev, "regression observed by goroutine seed=%d at i=%d: prev=%d got=%d", seed, i, localPrev, got) + } + if got > localPrev { + localPrev = got + } + } + }(int64(g + 1)) + } + wg.Wait() +} diff --git a/erpc/networks_multiplexer_test.go b/erpc/networks_multiplexer_test.go index 6cd4614bf..17547c281 100644 --- a/erpc/networks_multiplexer_test.go +++ b/erpc/networks_multiplexer_test.go @@ -31,6 +31,38 @@ func init() { // This test reproduces the race condition described in https://github.com/erpc/erpc/pull/615 // where cleanupMultiplexer would acquire the lock before followers could copy the response, // causing followers to see nil response and make their own upstream requests. +func TestMethodSkipMultiplexing(t *testing.T) { + cases := []struct { + method string + skip bool + }{ + // Tx broadcasts must not multiplex: a retry of the same signed + // payload is the caller's way of asking us to re-submit, not a + // duplicate that can wait on an in-flight attempt. + {"eth_sendRawTransaction", true}, + {"eth_sendTransaction", true}, + // Reads are the multiplexer's intended workload: concurrent + // identical lookups share one upstream call. + {"eth_call", false}, + {"eth_getBlockByNumber", false}, + {"eth_getLogs", false}, + {"eth_chainId", false}, + {"net_version", false}, + // Unknown methods default to multiplex-eligible. Skipping the + // dedup is a behaviour change, so the safer default for methods + // we haven't classified is to keep dedup on — the only cost is + // occasional redundant work coalescing into one request. + {"some_unknown_method", false}, + {"", false}, + } + for _, tc := range cases { + got := methodSkipMultiplexing(tc.method) + if got != tc.skip { + t.Errorf("methodSkipMultiplexing(%q) = %v, want %v", tc.method, got, tc.skip) + } + } +} + func TestNetwork_Multiplexer_FollowersReceiveResponse(t *testing.T) { t.Run("ConcurrentFollowers_AllReceiveLeaderResponse", func(t *testing.T) { util.ResetGock() From 5a46e1493b499d998ac09080232637aa96f9ccbc Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Jo=C3=A3o=20Gomes?= Date: Fri, 22 May 2026 14:39:01 +0100 Subject: [PATCH 06/40] fix(failover): close tracker-signal gap and add per-request fallback escape Reproducing a production node-down incident confirmed that eRPC fails to fall over to configured fallback upstreams. Two paths silently bypassed the health tracker and a third gap made true HA impossible: 1. checkUpstreamBlockAvailability rejections (handleBlockSkip) updated Prometheus counters but never called RecordUpstreamRequest / RecordUpstreamFailure. An upstream gate-rejected on every request showed errorRate=0 to the JS selectionPolicy. 2. PollLatestBlockNumber / PollFinalizedBlockNumber fetch failures logged at warn and returned without recording. The CB counted them (which is why it cycled half-open <-> open during the outage), but subsequent CB-open responses bypass tryForward and the tracker stays starved of signal. 3. Even after Fixes (1) and (2), the selectionPolicy's reaction is bounded by evalInterval (1m in typical configs) + scoreRefreshInterval (10s default). Worst case ~70s of ErrUpstreamsExhausted to clients during the transition. Real HA requires sub-request escalation. This commit lands all three architectural fixes: * Fix 2 - networks.go:handleBlockSkip records (request, failure) when the block-availability rejection is retryable. * Fix 3 - evm_state_poller.go records (request, failure) when fetchBlock returns a CB-open short-circuit. Narrowed to ErrFailsafeCircuitBreakerOpen to avoid double-counting failures that already reach tryForward via the normal forwarding path. * Per-request escape hatch in Network.Forward: when the inner loop exhausts upsList with retryable errors and failover.onDefaultsExhausted is enabled, replace upsList with the fallback-group upstreams that are bootstrapped + CB-closed + method-allowed (cordon ignored - the escape is precisely what bypasses it), reset attempted/ConsumedUpstreams/ ErrorsByUpstream for them, and re-enter the inner loop with the selectionPolicy permit check bypassed. First fallback whose gate passes serves the request. Client never sees a failure. Steady-state cost is zero - the escape hatch only triggers on exhaustion, so healthy primaries return on the first iteration and fallbacks are never touched. Outage cost is one fallback request per client request for ~1m until the policy promotes fallbacks into upsList directly, after which the escape hatch is no longer needed per-request. Observability: * erpc_network_fallback_escape_total{project,network,category} Counts escape-hatch firings. Sustained non-zero in steady state is a config signal (alert candidate); sustained non-zero during an outage is expected during the policy-reaction window. Tests: * erpc/networks_failover_escape_test.go - 4-upstream test layout (2 primaries + 2 fallbacks) with a production-style selectionPolicy that cordons fallbacks while primaries are healthy. Regression test TestFailover_GateSkipsAccumulateErrorRate verifies the recording fixes. TestFailover_EscapeHatch sub-tests: - EscapesToFallbackOnFirstFailingRequest - NoEscapeWhenPrimariesHealthy (verifies zero steady-state cost) - NoEscapeWhenFailoverDisabled (operator opt-out) - OnlyEscalatesOncePerRequest (no escalation loop) * upstream/registry_fallback_escape_test.go - unit tests for GetFallbackEscapeUpstreams: filter-by-group, ignores-cordon (central property), filter-by-IgnoreMethods, no-fallbacks-returns-empty. End-to-end binary test against four mock RPC servers reproduced the outage scenario locally: 10/10 healthy requests served by primaries with 0 escape firings; 20/20 outage requests served by fallback with 20 escape firings and 0 client-facing failures. The pre-existing TestNetwork_SelectionScenarios/StatePollerContributesToErrorRateWhenNotResamplingExcludedUpstreams caught an initial Fix 3 over-recording bug (resolved by the CB-open-only narrowing). Co-Authored-By: Claude Opus 4.7 (1M context) --- architecture/evm/evm_state_poller.go | 23 + common/request.go | 26 + erpc/networks.go | 229 +++++-- erpc/networks_failover_escape_test.go | 719 ++++++++++++++++++++++ telemetry/metrics.go | 15 + upstream/registry.go | 34 + upstream/registry_fallback_escape_test.go | 209 +++++++ 7 files changed, 1187 insertions(+), 68 deletions(-) create mode 100644 erpc/networks_failover_escape_test.go create mode 100644 upstream/registry_fallback_escape_test.go diff --git a/architecture/evm/evm_state_poller.go b/architecture/evm/evm_state_poller.go index bf442971e..a026cdcf9 100644 --- a/architecture/evm/evm_state_poller.go +++ b/architecture/evm/evm_state_poller.go @@ -433,6 +433,20 @@ func (e *EvmStatePoller) PollLatestBlockNumber(ctx context.Context) (int64, erro e.stateMu.Unlock() return 0, nil } else { + // Record as an upstream failure ONLY when the failure bypassed + // Upstream.tryForward (which already records its own failures). + // The bypass case is failsafe-CB-open: once the CB trips, every + // subsequent call short-circuits before tryForward runs, so the + // tracker stops seeing samples and the selection policy's + // errorRate freezes — preventing failover. By recording the + // CB-short-circuit attempts here, the tracker continues to + // climb post-CB-open and the policy can react. Non-CB failures + // (HTTP 500, transport errors, etc.) already reach tryForward + // and record there; recording again here would double-count. + if e.tracker != nil && common.HasErrorCode(err, common.ErrCodeFailsafeCircuitBreakerOpen) { + e.tracker.RecordUpstreamRequest(e.upstream, "eth_getBlockByNumber", common.DataFinalityStateRealtime) + e.tracker.RecordUpstreamFailure(e.upstream, "eth_getBlockByNumber", common.DataFinalityStateRealtime, err) + } e.logger.Warn().Err(err).Msg("failed to get latest block number in evm state poller") return 0, err } @@ -536,6 +550,15 @@ func (e *EvmStatePoller) PollFinalizedBlockNumber(ctx context.Context) (int64, e e.stateMu.Unlock() return 0, nil } else { + // See PollLatestBlockNumber for the rationale: record as an + // upstream failure ONLY when the failure bypassed tryForward + // (CB-open short-circuit). Non-CB failures already reach + // tryForward's existing recording path; recording again here + // would double-count. + if e.tracker != nil && common.HasErrorCode(err, common.ErrCodeFailsafeCircuitBreakerOpen) { + e.tracker.RecordUpstreamRequest(e.upstream, "eth_getBlockByNumber", common.DataFinalityStateFinalized) + e.tracker.RecordUpstreamFailure(e.upstream, "eth_getBlockByNumber", common.DataFinalityStateFinalized, err) + } e.logger.Warn().Err(err).Msg("failed to get finalized block number in evm state poller") return 0, err } diff --git a/common/request.go b/common/request.go index 0723c7352..f7dd6161c 100644 --- a/common/request.go +++ b/common/request.go @@ -335,6 +335,13 @@ type NormalizedRequest struct { upstreamList []Upstream // Available upstreams for this request ConsumedUpstreams *sync.Map // Tracks upstreams that provided valid responses + // escalatedToFallbacks marks whether this request has already invoked the + // per-request fallback escape hatch (Network.Forward's inner loop appends + // cordoned fallback-group upstreams when the primary set is exhausted with + // retryable errors). Prevents escalation loops across failsafe retries + // and re-entries. + escalatedToFallbacks atomic.Bool + lastValidResponse atomic.Pointer[NormalizedResponse] lastUpstream atomic.Value evmBlockRef atomic.Value @@ -1267,6 +1274,25 @@ func (r *NormalizedRequest) Upstreams() []Upstream { return out } +// HasEscalatedToFallbacks reports whether the per-request fallback escape +// hatch has already fired for this request. Used by Network.Forward's outer +// loop to ensure the escape is attempted at most once per request. +func (r *NormalizedRequest) HasEscalatedToFallbacks() bool { + if r == nil { + return false + } + return r.escalatedToFallbacks.Load() +} + +// MarkEscalatedToFallbacks records that this request has invoked the +// fallback escape hatch. +func (r *NormalizedRequest) MarkEscalatedToFallbacks() { + if r == nil { + return + } + r.escalatedToFallbacks.Store(true) +} + // UserId returns the user ID from the user object, or "n/a" if not available func (r *NormalizedRequest) UserId() string { if r == nil { diff --git a/erpc/networks.go b/erpc/networks.go index 6f5e5a60f..3101ef681 100644 --- a/erpc/networks.go +++ b/erpc/networks.go @@ -776,45 +776,53 @@ func (n *Network) Forward(ctx context.Context, req *common.NormalizedRequest) (* // network executor invokes (potentially multiple times for retry/hedge, // and per-slot for consensus). sweepFn := func(execSpanCtx context.Context, effectiveReq *common.NormalizedRequest, oneUpstreamOnly bool) (*common.NormalizedResponse, error) { - snap := effectiveReq.ExecState().Snapshot() - _, execSpan := common.StartSpan(execSpanCtx, "Network.forwardAttempt", - trace.WithAttributes( - attribute.String("network.id", n.networkId), - attribute.String("request.method", method), - attribute.Int("execution.attempt", snap.Attempts), - attribute.Int("execution.retry", snap.Retries), - attribute.Int("execution.hedge", snap.Hedges), - ), + snap := effectiveReq.ExecState().Snapshot() + _, execSpan := common.StartSpan(execSpanCtx, "Network.forwardAttempt", + trace.WithAttributes( + attribute.String("network.id", n.networkId), + attribute.String("request.method", method), + attribute.Int("execution.attempt", snap.Attempts), + attribute.Int("execution.retry", snap.Retries), + attribute.Int("execution.hedge", snap.Hedges), + ), + ) + defer execSpan.End() + + if common.IsTracingDetailed { + execSpan.SetAttributes( + attribute.String("request.id", fmt.Sprintf("%v", effectiveReq.ID())), ) - defer execSpan.End() + } - if common.IsTracingDetailed { - execSpan.SetAttributes( - attribute.String("request.id", fmt.Sprintf("%v", effectiveReq.ID())), - ) + if ctxErr := execSpanCtx.Err(); ctxErr != nil { + cause := context.Cause(execSpanCtx) + if cause != nil { + common.SetTraceSpanError(execSpan, cause) + return nil, cause + } else { + common.SetTraceSpanError(execSpan, ctxErr) + return nil, ctxErr } + } + // Network-scope timeout is applied inside networkExecutor.Run. + // Per-attempt enforcement here would double-apply and break retry budgets. - if ctxErr := execSpanCtx.Err(); ctxErr != nil { - cause := context.Cause(execSpanCtx) - if cause != nil { - common.SetTraceSpanError(execSpan, cause) - return nil, cause - } else { - common.SetTraceSpanError(execSpan, ctxErr) - return nil, ctxErr - } - } - // Network-scope timeout is applied inside networkExecutor.Run. - // Per-attempt enforcement here would double-apply and break retry budgets. - - var bestResp *common.NormalizedResponse - var lastErr error - maxLoopIterations := effectiveReq.UpstreamsCount() - if oneUpstreamOnly { - maxLoopIterations = 1 - } - attempted := make(map[string]struct{}, maxLoopIterations) + var bestResp *common.NormalizedResponse + var lastErr error + maxLoopIterations := effectiveReq.UpstreamsCount() + if oneUpstreamOnly { + maxLoopIterations = 1 + } + attempted := make(map[string]struct{}, maxLoopIterations) + // Capture the pre-escalation upsList size for the error reporting + // path below. If the per-request fallback escape hatch fires, we + // replace upsList with the appended fallback set, but the + // ErrUpstreamsExhausted.upstreams field is meant to describe the + // routable set the policy selected — not the escape-hatch override. + originalUpsListLen := len(upsList) + escalationLoop: + for { for loopIteration := 0; loopIteration < maxLoopIterations; loopIteration++ { loopCtx, loopSpan := common.StartDetailSpan(execSpanCtx, "Network.UpstreamLoop") if ctxErr := loopCtx.Err(); ctxErr != nil { @@ -869,6 +877,13 @@ func (n *Network) Forward(ctx context.Context, req *common.NormalizedRequest) (* // Pre-forward: block availability gating → skip to next upstream if skipErr, isRetryable := n.checkUpstreamBlockAvailability(loopCtx, u, effectiveReq, method); skipErr != nil { n.handleBlockSkip(loopCtx, loopSpan, &ulg, u, effectiveReq, method, skipErr, isRetryable) + // Track the skip as the loop's last error so the per-request + // fallback escape hatch (after the loop) can see it. Record + // BOTH retryable and non-retryable skips: "non-retryable" + // means "don't retry the SAME upstream", not "don't try OTHER + // upstreams" — exactly what the escape hatch is for, and + // fallbacks ahead of a stalled primary need this path reachable. + lastErr = skipErr loopSpan.End() continue } @@ -960,43 +975,109 @@ func (n *Network) Forward(ctx context.Context, req *common.NormalizedRequest) (* return nil, cause } - // All upstreams tried. Return the best result for the retry/hedge - // wrapper to evaluate. Prefer a valid response over an error so - // the delay function can detect empty results and apply - // emptyResultDelay. - if bestResp != nil { - st := effectiveReq.ExecState() - st.MarkUpstreamAttemptWon(bestResp.UpstreamId()) - s := st.Snapshot() - bestResp.SetAttempts(s.Attempts) - bestResp.SetRetries(s.Retries) - bestResp.SetHedges(s.Hedges) - return bestResp, nil - } + // === Per-request fallback escape hatch === + // + // If the inner loop exhausted upsList with a retryable error (the + // selectionPolicy has cordoned fallbacks while primaries are down) + // and failover is enabled, append the fallback-tier upstreams that + // are bootstrapped + CB-closed + method-allowed and re-enter the + // inner loop once. The client gets the fallback's response on the + // same call that would otherwise return ErrUpstreamsExhausted. + // + // Bounds: + // - At most one escalation per request (MarkEscalatedToFallbacks). + // - bestResp != nil means we have an emptyish response that + // failsafe's emptyResultDelay should evaluate — don't pre-empt it. + // - Deterministic client errors return from the inner loop + // immediately, so any non-nil lastErr here means "this upstream + // couldn't serve; try a different one" — exactly the escape's job. + // - Consensus requires strict per-upstream semantics; don't modify + // the candidate set mid-execution. + if bestResp == nil && + !effectiveReq.HasEscalatedToFallbacks() && + lastErr != nil && + n.cfg.Failover != nil && n.cfg.Failover.Enabled() && + !failsafeExecutor.HasConsensus() { + + fallbacks := n.upstreamsRegistry.GetFallbackEscapeUpstreams( + execSpanCtx, n.networkId, method, + ) + if len(fallbacks) > 0 { + // Replace upsList with ALL eligible fallbacks. Clear their + // stored errors and consumed reservations so NextUpstream + // re-selects them; primaries are removed from upsList so they + // won't be picked again, but their state in attempted / + // ErrorsByUpstream is preserved for eventual error reporting. + fbCommon := make([]common.Upstream, 0, len(fallbacks)) + for _, fb := range fallbacks { + fbCommon = append(fbCommon, fb) + effectiveReq.ErrorsByUpstream.Delete(common.Upstream(fb)) + effectiveReq.ConsumedUpstreams.Delete(common.Upstream(fb)) + } + effectiveReq.SetUpstreams(fbCommon) + upsList = fbCommon + maxLoopIterations = len(fallbacks) + attempted = make(map[string]struct{}, len(fallbacks)) + effectiveReq.MarkEscalatedToFallbacks() + // Reset lastErr so the next pass either replaces it (new + // failure) or leaves it nil (success). + lastErr = nil + + telemetry.MetricNetworkFallbackEscapeTotal.WithLabelValues( + n.projectId, n.Label(), method, + ).Inc() + + lg.Debug(). + Str("networkId", n.networkId). + Str("method", method). + Int("fallbacks", len(fallbacks)). + Msg("activated per-request fallback escape after primary set exhausted") - // For consensus, return the raw upstream error so the consensus - // policy receives the actual error type (e.g. server error, missing - // data) rather than a wrapped ErrUpstreamsExhausted. - if oneUpstreamOnly && lastErr != nil { - return nil, lastErr + continue escalationLoop + } } - s := effectiveReq.ExecState().Snapshot() - exhaustedErr := common.NewErrUpstreamsExhausted( - effectiveReq, - &effectiveReq.ErrorsByUpstream, - n.projectId, - n.networkId, - method, - time.Since(startTime), - s.Attempts, - s.Retries, - s.Hedges, - len(upsList), - ) - common.SetTraceSpanError(execSpan, exhaustedErr) - return nil, exhaustedErr - } + // No further escalation possible; exit the outer loop. + break escalationLoop + } + + // All upstreams (including any escaped-to fallbacks) tried. Return + // the best result for failsafe to evaluate delays and retries. + // Prefer a valid response over an error so the delay function can + // detect empty results and apply emptyResultDelay. + if bestResp != nil { + st := effectiveReq.ExecState() + st.MarkUpstreamAttemptWon(bestResp.UpstreamId()) + s := st.Snapshot() + bestResp.SetAttempts(s.Attempts) + bestResp.SetRetries(s.Retries) + bestResp.SetHedges(s.Hedges) + return bestResp, nil + } + + // For consensus, return the raw upstream error so the consensus + // policy receives the actual error type (e.g. server error, missing + // data) rather than a wrapped ErrUpstreamsExhausted. + if oneUpstreamOnly && lastErr != nil { + return nil, lastErr + } + + s := effectiveReq.ExecState().Snapshot() + exhaustedErr := common.NewErrUpstreamsExhausted( + effectiveReq, + &effectiveReq.ErrorsByUpstream, + n.projectId, + n.networkId, + method, + time.Since(startTime), + s.Attempts, + s.Retries, + s.Hedges, + originalUpsListLen, + ) + common.SetTraceSpanError(execSpan, exhaustedErr) + return nil, exhaustedErr + } // Two entry points into sweepFn: // tryOneUpstream — single-upstream variant for consensus slots @@ -1453,6 +1534,18 @@ func (n *Network) handleBlockSkip( } req.MarkUpstreamCompleted(ctx, u, nil, errToStore) + // Record retryable block-availability skips as upstream failures in the + // health tracker so the selection policy's errorRate reflects what the + // upstream actually can't serve. Without this, an upstream whose poller is + // stuck behind the network aggregator gets gate-rejected indefinitely + // without ever moving errorRate, and the policy keeps it in `healthy` — + // preventing failover to fallbacks. Non-retryable skips remain silent + // (lower-bound / too-far-ahead are configuration semantics, not health). + if isRetryable && n.metricsTracker != nil { + n.metricsTracker.RecordUpstreamRequest(u, method, finality) + n.metricsTracker.RecordUpstreamFailure(u, method, finality, skipErr) + } + // When the block is slightly ahead (retryable), trigger an async poll so the // state poller fetches the latest block number before the next retry fires. // Without this the retry loop sleeps for blockUnavailableDelay but the cached diff --git a/erpc/networks_failover_escape_test.go b/erpc/networks_failover_escape_test.go new file mode 100644 index 000000000..2374f576d --- /dev/null +++ b/erpc/networks_failover_escape_test.go @@ -0,0 +1,719 @@ +package erpc + +import ( + "context" + "fmt" + "net/http" + "strings" + "testing" + "time" + + "github.com/erpc/erpc/common" + "github.com/erpc/erpc/data" + "github.com/erpc/erpc/health" + "github.com/erpc/erpc/telemetry" + "github.com/erpc/erpc/thirdparty" + "github.com/erpc/erpc/upstream" + "github.com/erpc/erpc/util" + "github.com/h2non/gock" + promUtil "github.com/prometheus/client_golang/prometheus/testutil" + "github.com/rs/zerolog/log" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// failoverSelectionPolicy is a representative production selectionPolicy that +// cordons fallback-group upstreams while any primary is healthy. Without the +// gate-skip/CB-open tracker fixes plus the per-request fallback escape hatch, +// a primary collapse leaves the policy unable to detect the failure and +// fallbacks remain cordoned out of upsList, producing client-facing +// ErrUpstreamsExhausted until the next eval tick (1m worst case). +const failoverSelectionPolicy = ` +(upstreams, method) => { + const isFallback = u => u && u.config && u.config.group === 'fallback' + const healthOK = u => { + const m = (u && u.metrics) || {} + const err = m.errorRate + const lag = m.blockHeadLag + return (err == null || err < 0.5) && (lag == null || lag < 5) + } + const primary = upstreams.filter(u => !isFallback(u)) + const fallback = upstreams.filter(isFallback) + const healthy = primary.filter(healthOK) + if (healthy.length > 0) return healthy + if (fallback.length > 0) return fallback + return upstreams +} +` + +// mockJsonRpcUpstream wires the standard state-poller mocks for one upstream +// at a fixed block height. Used to build the multi-primary + multi-fallback +// fixture the failover tests need. +func mockJsonRpcUpstream(host string, chainIdHex, latestHex, finalizedHex string) { + gock.New("http://" + host). + Post(""). + Persist(). + Filter(func(r *http.Request) bool { + return strings.Contains(util.SafeReadBody(r), "eth_chainId") + }). + Reply(200). + JSON([]byte(fmt.Sprintf(`{"result":"%s"}`, chainIdHex))) + + gock.New("http://" + host). + Post(""). + Persist(). + Filter(func(r *http.Request) bool { + b := util.SafeReadBody(r) + return strings.Contains(b, "eth_getBlockByNumber") && strings.Contains(b, `"latest"`) + }). + Reply(200). + JSON([]byte(fmt.Sprintf(`{"result":{"number":"%s","timestamp":"0x6702a8f0"}}`, latestHex))) + + gock.New("http://" + host). + Post(""). + Persist(). + Filter(func(r *http.Request) bool { + b := util.SafeReadBody(r) + return strings.Contains(b, "eth_getBlockByNumber") && strings.Contains(b, `"finalized"`) + }). + Reply(200). + JSON([]byte(fmt.Sprintf(`{"result":{"number":"%s","timestamp":"0x6702a8e0"}}`, finalizedHex))) + + gock.New("http://" + host). + Post(""). + Persist(). + Filter(func(r *http.Request) bool { + return strings.Contains(util.SafeReadBody(r), "eth_syncing") + }). + Reply(200). + JSON([]byte(`{"result":false}`)) +} + +// mockEthCallReturning wires an eth_call mock that echoes which upstream +// served the request via a unique result hex. Lets the test assert +// failover by which upstream actually answered. +func mockEthCallReturning(host string, resultHex string) { + gock.New("http://" + host). + Post(""). + Persist(). + Filter(func(r *http.Request) bool { + return strings.Contains(util.SafeReadBody(r), "eth_call") + }). + Reply(200). + JSON([]byte(fmt.Sprintf(`{"jsonrpc":"2.0","id":1,"result":"%s"}`, resultHex))) +} + +// TestFailover_GateSkipsAccumulateErrorRate verifies the recording fixes +// (handleBlockSkip → tracker, state-poller CB-open → tracker). With the +// fixes in place, a gate-rejection burst against a primary stuck behind +// the network aggregator must move errorRate past 0.5, the selectionPolicy +// must cordon the primary on its next eval tick, and the fallbacks must +// be promoted into upsList so subsequent client requests succeed via them. +// +// Pre-fix behaviour (verified via git stash on this same test): gate +// rejections in checkUpstreamBlockAvailability do not call +// RecordUpstreamFailure, so the primary's errorRate stays at 0 +// indefinitely, the policy keeps the primary in `healthy`, and the +// fallbacks remain cordoned out of upsList. The final assertion fails +// with the primary still routable and the fallbacks not. +// +// Post-fix behaviour: handleBlockSkip records (Request, Failure) on every +// retryable gate-reject, errorRate crosses 0.5 within a single eval window +// once the burst starts, the next policy tick excludes the primary, and +// the fallbacks take over. +func TestFailover_GateSkipsAccumulateErrorRate(t *testing.T) { + util.ResetGock() + defer util.ResetGock() + + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + + // Two HTTP primaries stuck at block 1000 (analogous to a node that + // can't advance — pollers freeze, but the network aggregator + // continues advancing via the fallbacks' independent pollers). + const chainIdHex = "0x3e7" // 999 + const primaryLatest = "0x3e8" + const fallbackLatest = "0x3ea" + const finalizedHex = "0x3e0" + const requestBlock = "0x3ea" // beyond primary cache, gate-rejects on primaries + + mockJsonRpcUpstream("rpc1.localhost", chainIdHex, primaryLatest, finalizedHex) + mockJsonRpcUpstream("rpc2.localhost", chainIdHex, primaryLatest, finalizedHex) + mockJsonRpcUpstream("rpc3.localhost", chainIdHex, fallbackLatest, finalizedHex) + mockJsonRpcUpstream("rpc4.localhost", chainIdHex, fallbackLatest, finalizedHex) + + mockEthCallReturning("rpc1.localhost", "0x1111") + mockEthCallReturning("rpc2.localhost", "0x2222") + mockEthCallReturning("rpc3.localhost", "0x3333") + mockEthCallReturning("rpc4.localhost", "0x4444") + + upstreamConfigs := []*common.UpstreamConfig{ + { + Type: common.UpstreamTypeEvm, Id: "primary-1", + Endpoint: "http://rpc1.localhost", + Evm: &common.EvmUpstreamConfig{ + ChainId: 999, + StatePollerInterval: common.Duration(100 * time.Millisecond), + StatePollerDebounce: common.Duration(20 * time.Millisecond), + }, + }, + { + Type: common.UpstreamTypeEvm, Id: "primary-2", + Endpoint: "http://rpc2.localhost", + Evm: &common.EvmUpstreamConfig{ + ChainId: 999, + StatePollerInterval: common.Duration(100 * time.Millisecond), + StatePollerDebounce: common.Duration(20 * time.Millisecond), + }, + }, + { + Type: common.UpstreamTypeEvm, Id: "fallback-1", + Endpoint: "http://rpc3.localhost", + Group: common.UpstreamGroupFallback, + Evm: &common.EvmUpstreamConfig{ + ChainId: 999, + StatePollerInterval: common.Duration(100 * time.Millisecond), + StatePollerDebounce: common.Duration(20 * time.Millisecond), + }, + }, + { + Type: common.UpstreamTypeEvm, Id: "fallback-2", + Endpoint: "http://rpc4.localhost", + Group: common.UpstreamGroupFallback, + Evm: &common.EvmUpstreamConfig{ + ChainId: 999, + StatePollerInterval: common.Duration(100 * time.Millisecond), + StatePollerDebounce: common.Duration(20 * time.Millisecond), + }, + }, + } + + rlr, _ := upstream.NewRateLimitersRegistry(context.Background(), &common.RateLimiterConfig{}, &log.Logger) + // 5s metrics window so accumulated request/failure samples don't get + // reset mid-test. Pre-failure successes accumulate during initial + // warmup but are dominated quickly once the gate-rejection burst starts. + mt := health.NewTracker(&log.Logger, "main", 5*time.Second) + vr := thirdparty.NewVendorsRegistry() + pr, _ := thirdparty.NewProvidersRegistry(&log.Logger, vr, nil, nil) + + sharedStateCfg := &common.SharedStateConfig{ + Connector: &common.ConnectorConfig{ + Driver: common.DriverMemory, + Memory: &common.MemoryConnectorConfig{ + MaxItems: 100_000, + MaxTotalSize: "1GB", + }, + }, + LockMaxWait: common.Duration(200 * time.Millisecond), + UpdateMaxWait: common.Duration(200 * time.Millisecond), + FallbackTimeout: common.Duration(3 * time.Second), + LockTtl: common.Duration(4 * time.Second), + } + require.NoError(t, sharedStateCfg.SetDefaults("test")) + ssr, err := data.NewSharedStateRegistry(ctx, &log.Logger, sharedStateCfg) + require.NoError(t, err) + + upr := upstream.NewUpstreamsRegistry( + ctx, &log.Logger, "main", upstreamConfigs, + ssr, rlr, vr, pr, nil, mt, 1*time.Second, nil, nil, + ) + + // The production-style selectionPolicy compressed to fire every 150ms + // so the test doesn't have to wait the production 1m interval. + evalFn, err := common.CompileFunction(failoverSelectionPolicy) + require.NoError(t, err) + selectionPolicy := &common.SelectionPolicyConfig{ + EvalInterval: common.Duration(150 * time.Millisecond), + EvalFunctionSource: failoverSelectionPolicy, + EvalFunction: evalFn, + EvalPerMethod: false, + } + require.NoError(t, selectionPolicy.SetDefaults()) + + maxRetryable := int64(128) + enforce := true + networkConfig := &common.NetworkConfig{ + Architecture: common.ArchitectureEvm, + Evm: &common.EvmNetworkConfig{ + ChainId: 999, + MaxRetryableBlockDistance: &maxRetryable, + EnforceBlockAvailability: &enforce, + }, + SelectionPolicy: selectionPolicy, + // Failover.onDefaultsExhausted = true so once fallbacks are in upsList, + // they're tried intra-request after primaries fail. + Failover: &common.FailoverConfig{ + OnDefaultsExhausted: &enforce, + }, + } + + network, err := NewNetwork(ctx, &log.Logger, "main", networkConfig, rlr, upr, mt) + require.NoError(t, err) + + upr.Bootstrap(ctx) + time.Sleep(200 * time.Millisecond) + require.NoError(t, upr.GetInitializer().WaitForTasks(ctx)) + require.NoError(t, upr.PrepareUpstreamsForNetwork(ctx, util.EvmNetworkId(999))) + require.NoError(t, network.Bootstrap(ctx)) + + // Bootstrap each upstream's state poller; they'll fetch latest/finalized + // from the gock mocks and populate the per-upstream shared counters. + upsList := upr.GetNetworkUpstreams(ctx, util.EvmNetworkId(999)) + require.Len(t, upsList, 4) + for _, ups := range upsList { + require.NoError(t, ups.Bootstrap(ctx)) + } + + // Let pollers run a couple of cycles so per-upstream LatestBlock counters + // are populated and the selection policy has evaluated at least once. + time.Sleep(500 * time.Millisecond) + + // Sanity-check the initial state: + // - Primaries should have LatestBlock = 1000 + // - Fallbacks should have LatestBlock = 1002 + // - Policy should keep primaries active and cordon fallbacks + primaryUp := mustGetUpstream(upsList, "primary-1") + fallbackUp := mustGetUpstream(upsList, "fallback-1") + require.Equal(t, int64(1000), primaryUp.EvmStatePoller().LatestBlock(), "primary should be at block 1000") + require.Equal(t, int64(1002), fallbackUp.EvmStatePoller().LatestBlock(), "fallback should be at block 1002") + + // Verify policy initially returns primaries only — fallback should be cordoned. + require.NoError(t, network.selectionPolicyEvaluator.AcquirePermit(&log.Logger, primaryUp, "eth_call"), + "primary should be active initially") + require.Error(t, network.selectionPolicyEvaluator.AcquirePermit(&log.Logger, fallbackUp, "eth_call"), + "fallback should be cordoned initially (primaries are healthy)") + + // --- Phase 2: trigger the gate-rejection burst --- + // + // Each request targets block 1002. The gate compares against the + // per-upstream cache (1000 on primaries) and rejects. With the + // handleBlockSkip recording fix, each rejection records (request, + // failure) on the metrics tracker for the primary. + // + // Send enough requests to dominate the pre-burst success ratio. With + // the 5s window and ~10 prior bootstrap-related successes, ~50 failures + // is comfortably > 0.5 errorRate. + const burstSize = 80 + for i := 0; i < burstSize; i++ { + req := common.NewNormalizedRequest([]byte(fmt.Sprintf( + `{"jsonrpc":"2.0","id":%d,"method":"eth_call","params":[{"to":"0xdead","data":"0x"},"%s"]}`, + i, requestBlock, + ))) + req.SetNetwork(network) + _, _ = network.Forward(ctx, req) + } + + // Wait for at least one full selectionPolicy eval tick (150ms) so the + // policy observes the elevated errorRate and applies the cordon, plus + // one score-refresh tick (1s) so the registry rebuilds sortedUpstreams + // with fallbacks included. Then force an extra refresh to remove any + // remaining timing slack. + time.Sleep(400 * time.Millisecond) + require.NoError(t, upr.RefreshUpstreamNetworkMethodScores()) + time.Sleep(100 * time.Millisecond) + + // --- Assertions --- + + // 1) The handleBlockSkip recording fix plumbed gate-skips into the + // tracker: the primary's errorRate on the "*" aggregate (the slot + // the policy reads with evalPerMethod=false) must reflect the + // gate-rejection burst. Without the fix this stays at 0 + // indefinitely and the policy never reacts. + primaryMetrics := mt.GetUpstreamMethodMetrics(primaryUp, "*") + t.Logf("primary primary-1 metrics: requests=%d errors=%d errorRate=%.3f", + primaryMetrics.RequestsTotal.Load(), primaryMetrics.ErrorsTotal.Load(), primaryMetrics.ErrorRate()) + assert.Greater(t, primaryMetrics.ErrorRate(), 0.5, + "primary errorRate should cross 0.5 from gate-skip recording; "+ + "if 0 the handleBlockSkip recording fix didn't take effect") + + // 2) After the burst + at least one policy eval tick, the primary must + // be cordoned. AcquirePermit returns ErrCodeUpstreamExcludedByPolicy + // for cordoned upstreams. + err = network.selectionPolicyEvaluator.AcquirePermit(&log.Logger, primaryUp, "eth_call") + require.Error(t, err, "primary should be cordoned after the gate-skip burst") + assert.True(t, common.HasErrorCode(err, common.ErrCodeUpstreamExcludedByPolicy), + "expected primary to be excluded by policy, got: %v", err) + + // 3) The fallback must be promoted to active by the same eval tick. + require.NoError(t, network.selectionPolicyEvaluator.AcquirePermit(&log.Logger, fallbackUp, "eth_call"), + "fallback should be promoted after primary errorRate exceeds 0.5") + + // 4) A fresh request for block 1002 must now succeed. With fallbacks in + // upsList and primaries either cordoned or still gate-rejected, the + // request must route to one of the fallbacks (which have block 1002). + finalReq := common.NewNormalizedRequest([]byte(fmt.Sprintf( + `{"jsonrpc":"2.0","id":9999,"method":"eth_call","params":[{"to":"0xdead","data":"0x"},"%s"]}`, + requestBlock, + ))) + finalReq.SetNetwork(network) + resp, err := network.Forward(ctx, finalReq) + require.NoError(t, err, "client request must succeed via fallback after failover") + require.NotNil(t, resp) + defer resp.Release() + + jrr, err := resp.JsonRpcResponse() + require.NoError(t, err) + require.NotNil(t, jrr) + result := strings.Trim(jrr.GetResultString(), `"`) + // Fallbacks return 0x3333 or 0x4444; primaries return 0x1111 or 0x2222. + assert.Contains(t, []string{"0x3333", "0x4444"}, result, + "final eth_call must be served by a fallback (0x3333 or 0x4444), got %q", result) +} + +func mustGetUpstream(ups []*upstream.Upstream, id string) *upstream.Upstream { + for _, u := range ups { + if u.Id() == id { + return u + } + } + panic("upstream not found: " + id) +} + +// failoverFixtureOpts configures the standard 4-upstream test layout +// (2 primaries + 2 fallbacks) used by the escape-hatch sub-tests. +type failoverFixtureOpts struct { + primaryLatest string + fallbackLatest string + enableFailover bool +} + +func setupFailoverFixture( + t *testing.T, ctx context.Context, opts failoverFixtureOpts, +) (*Network, []*upstream.Upstream, *health.Tracker) { + t.Helper() + util.ResetGock() + + const chainIdHex = "0x3e7" // 999 + const finalizedHex = "0x3e0" + + mockJsonRpcUpstream("rpc1.localhost", chainIdHex, opts.primaryLatest, finalizedHex) + mockJsonRpcUpstream("rpc2.localhost", chainIdHex, opts.primaryLatest, finalizedHex) + mockJsonRpcUpstream("rpc3.localhost", chainIdHex, opts.fallbackLatest, finalizedHex) + mockJsonRpcUpstream("rpc4.localhost", chainIdHex, opts.fallbackLatest, finalizedHex) + + mockEthCallReturning("rpc1.localhost", "0x1111") + mockEthCallReturning("rpc2.localhost", "0x2222") + mockEthCallReturning("rpc3.localhost", "0x3333") + mockEthCallReturning("rpc4.localhost", "0x4444") + + upstreamConfigs := []*common.UpstreamConfig{ + { + Type: common.UpstreamTypeEvm, Id: "primary-1", + Endpoint: "http://rpc1.localhost", + Evm: &common.EvmUpstreamConfig{ + ChainId: 999, + StatePollerInterval: common.Duration(100 * time.Millisecond), + StatePollerDebounce: common.Duration(20 * time.Millisecond), + }, + }, + { + Type: common.UpstreamTypeEvm, Id: "primary-2", + Endpoint: "http://rpc2.localhost", + Evm: &common.EvmUpstreamConfig{ + ChainId: 999, + StatePollerInterval: common.Duration(100 * time.Millisecond), + StatePollerDebounce: common.Duration(20 * time.Millisecond), + }, + }, + { + Type: common.UpstreamTypeEvm, Id: "fallback-1", + Endpoint: "http://rpc3.localhost", + Group: common.UpstreamGroupFallback, + Evm: &common.EvmUpstreamConfig{ + ChainId: 999, + StatePollerInterval: common.Duration(100 * time.Millisecond), + StatePollerDebounce: common.Duration(20 * time.Millisecond), + }, + }, + { + Type: common.UpstreamTypeEvm, Id: "fallback-2", + Endpoint: "http://rpc4.localhost", + Group: common.UpstreamGroupFallback, + Evm: &common.EvmUpstreamConfig{ + ChainId: 999, + StatePollerInterval: common.Duration(100 * time.Millisecond), + StatePollerDebounce: common.Duration(20 * time.Millisecond), + }, + }, + } + + rlr, _ := upstream.NewRateLimitersRegistry(context.Background(), &common.RateLimiterConfig{}, &log.Logger) + mt := health.NewTracker(&log.Logger, "main", 5*time.Second) + vr := thirdparty.NewVendorsRegistry() + pr, _ := thirdparty.NewProvidersRegistry(&log.Logger, vr, nil, nil) + + sharedStateCfg := &common.SharedStateConfig{ + Connector: &common.ConnectorConfig{ + Driver: common.DriverMemory, + Memory: &common.MemoryConnectorConfig{ + MaxItems: 100_000, + MaxTotalSize: "1GB", + }, + }, + LockMaxWait: common.Duration(200 * time.Millisecond), + UpdateMaxWait: common.Duration(200 * time.Millisecond), + FallbackTimeout: common.Duration(3 * time.Second), + LockTtl: common.Duration(4 * time.Second), + } + require.NoError(t, sharedStateCfg.SetDefaults("test")) + ssr, err := data.NewSharedStateRegistry(ctx, &log.Logger, sharedStateCfg) + require.NoError(t, err) + + upr := upstream.NewUpstreamsRegistry( + ctx, &log.Logger, "main", upstreamConfigs, + ssr, rlr, vr, pr, nil, mt, 1*time.Second, nil, nil, + ) + + evalFn, err := common.CompileFunction(failoverSelectionPolicy) + require.NoError(t, err) + selectionPolicy := &common.SelectionPolicyConfig{ + EvalInterval: common.Duration(150 * time.Millisecond), + EvalFunctionSource: failoverSelectionPolicy, + EvalFunction: evalFn, + EvalPerMethod: false, + } + require.NoError(t, selectionPolicy.SetDefaults()) + + maxRetryable := int64(128) + enforce := true + networkConfig := &common.NetworkConfig{ + Architecture: common.ArchitectureEvm, + Evm: &common.EvmNetworkConfig{ + ChainId: 999, + MaxRetryableBlockDistance: &maxRetryable, + EnforceBlockAvailability: &enforce, + }, + SelectionPolicy: selectionPolicy, + } + if opts.enableFailover { + on := true + networkConfig.Failover = &common.FailoverConfig{OnDefaultsExhausted: &on} + } + + network, err := NewNetwork(ctx, &log.Logger, "main", networkConfig, rlr, upr, mt) + require.NoError(t, err) + + upr.Bootstrap(ctx) + time.Sleep(200 * time.Millisecond) + require.NoError(t, upr.GetInitializer().WaitForTasks(ctx)) + require.NoError(t, upr.PrepareUpstreamsForNetwork(ctx, util.EvmNetworkId(999))) + require.NoError(t, network.Bootstrap(ctx)) + + upsList := upr.GetNetworkUpstreams(ctx, util.EvmNetworkId(999)) + require.Len(t, upsList, 4) + for _, ups := range upsList { + require.NoError(t, ups.Bootstrap(ctx)) + } + + // Let pollers run and policy evaluate at least once. + time.Sleep(500 * time.Millisecond) + + // Prime the registry's sortedUpstreams[networkId]["eth_call"] cache by + // asking for it once. GetSortedUpstreams populates the slot from "*" or + // raw networkUpstreams on first call but does NOT apply filterCordoned — + // only RefreshUpstreamNetworkMethodScores does. So we prime FIRST, then + // refresh, to get the cordon decisions applied to the eth_call slot + // before the test request fires. Without this two-step, the first + // network.Forward sees all 4 upstreams (cordoned-but-not-yet-filtered), + // which masks the cordon-then-escape behaviour under test. + _, err = upr.GetSortedUpstreams(ctx, util.EvmNetworkId(999), "eth_call") + require.NoError(t, err) + require.NoError(t, upr.RefreshUpstreamNetworkMethodScores()) + time.Sleep(100 * time.Millisecond) + + return network, upsList, mt +} + +// ethCallRequest constructs an eth_call request targeting a specific block. +func ethCallRequest(id int, blockHex string) *common.NormalizedRequest { + return common.NewNormalizedRequest([]byte(fmt.Sprintf( + `{"jsonrpc":"2.0","id":%d,"method":"eth_call","params":[{"to":"0xdead","data":"0x"},"%s"]}`, + id, blockHex, + ))) +} + +// TestFailover_EscapeHatch verifies the per-request escape hatch: +// when the primary set is exhausted with retryable errors within a single +// request, fallbacks are appended and re-iterated so the client receives a +// fallback response on the same call. +func TestFailover_EscapeHatch(t *testing.T) { + t.Run("EscapesToFallbackOnFirstFailingRequest", func(t *testing.T) { + defer util.ResetGock() + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + + // Primaries stuck at 1000, fallbacks at 1002. Request block 1002 → + // gate-rejects on primaries → escape hatch should fire and route to + // fallback. + network, _, _ := setupFailoverFixture(t, ctx, failoverFixtureOpts{ + primaryLatest: "0x3e8", // 1000 + fallbackLatest: "0x3ea", // 1002 + enableFailover: true, + }) + + counter := telemetry.MetricNetworkFallbackEscapeTotal.WithLabelValues("main", "evm:999", "eth_call") + before := promUtil.ToFloat64(counter) + + // Single request — no warm-up burst, no policy-tick wait. + req := ethCallRequest(1, "0x3ea") + req.SetNetwork(network) + resp, err := network.Forward(ctx, req) + require.NoError(t, err, "client request must succeed on first try via escape hatch") + require.NotNil(t, resp) + defer resp.Release() + + jrr, err := resp.JsonRpcResponse() + require.NoError(t, err) + result := strings.Trim(jrr.GetResultString(), `"`) + assert.Contains(t, []string{"0x3333", "0x4444"}, result, + "escape hatch must route to a fallback; got %q", result) + + after := promUtil.ToFloat64(counter) + assert.Equal(t, before+1, after, + "MetricNetworkFallbackEscapeTotal must increment by exactly 1 for the escape firing") + }) + + t.Run("NoEscapeWhenPrimariesHealthy", func(t *testing.T) { + defer util.ResetGock() + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + + // All upstreams at block 1002. Request block 1002 → primary's gate + // passes → returns on first iteration → escape hatch never fires. + network, _, _ := setupFailoverFixture(t, ctx, failoverFixtureOpts{ + primaryLatest: "0x3ea", // 1002 + fallbackLatest: "0x3ea", // 1002 + enableFailover: true, + }) + + counter := telemetry.MetricNetworkFallbackEscapeTotal.WithLabelValues("main", "evm:999", "eth_call") + before := promUtil.ToFloat64(counter) + + // 30 requests over the healthy path. None should hit fallback. + for i := 0; i < 30; i++ { + req := ethCallRequest(i, "0x3ea") + req.SetNetwork(network) + resp, err := network.Forward(ctx, req) + require.NoError(t, err, "healthy primary request must succeed (iter %d)", i) + require.NotNil(t, resp) + + jrr, _ := resp.JsonRpcResponse() + result := strings.Trim(jrr.GetResultString(), `"`) + assert.Contains(t, []string{"0x1111", "0x2222"}, result, + "healthy request must be served by a primary (0x1111/0x2222), got %q (iter %d)", result, i) + resp.Release() + } + + after := promUtil.ToFloat64(counter) + assert.Equal(t, before, after, + "escape hatch must NOT fire when primaries are healthy; counter must be unchanged") + }) + + t.Run("NoEscapeWhenFailoverDisabled", func(t *testing.T) { + defer util.ResetGock() + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + + // Same primary-stuck-fallback-ahead setup as Sub-test A, but with + // failover.onDefaultsExhausted unset. Escape hatch must respect the + // operator's opt-out. + network, _, _ := setupFailoverFixture(t, ctx, failoverFixtureOpts{ + primaryLatest: "0x3e8", // 1000 + fallbackLatest: "0x3ea", // 1002 + enableFailover: false, // <-- disabled + }) + + req := ethCallRequest(1, "0x3ea") + req.SetNetwork(network) + resp, err := network.Forward(ctx, req) + require.Error(t, err, + "with failover disabled the escape hatch must NOT fire; request must surface ErrUpstreamsExhausted") + assert.True(t, common.HasErrorCode(err, common.ErrCodeUpstreamsExhausted), + "expected ErrUpstreamsExhausted, got %v", err) + if resp != nil { + resp.Release() + } + }) + + t.Run("OnlyEscalatesOncePerRequest", func(t *testing.T) { + defer util.ResetGock() + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + + // Primaries stuck at 1000, fallbacks ALSO stuck at 1000. Request + // block 1002 → gate-rejects everywhere. Escape fires once, finds + // fallbacks still can't serve, returns exhausted — does NOT loop. + network, _, _ := setupFailoverFixture(t, ctx, failoverFixtureOpts{ + primaryLatest: "0x3e8", // 1000 + fallbackLatest: "0x3e8", // also 1000 + enableFailover: true, + }) + + counter := telemetry.MetricNetworkFallbackEscapeTotal.WithLabelValues("main", "evm:999", "eth_call") + before := promUtil.ToFloat64(counter) + + req := ethCallRequest(1, "0x3ea") + req.SetNetwork(network) + resp, err := network.Forward(ctx, req) + require.Error(t, err, + "all upstreams (primary + fallback) can't serve the block; must return ErrUpstreamsExhausted") + assert.True(t, common.HasErrorCode(err, common.ErrCodeUpstreamsExhausted), + "expected ErrUpstreamsExhausted, got %v", err) + if resp != nil { + resp.Release() + } + + after := promUtil.ToFloat64(counter) + assert.Equal(t, before+1, after, + "escape must fire exactly once per request even when fallbacks also fail; got %v→%v", + before, after) + }) + + t.Run("EscapesOnNonRetryableGateSkip", func(t *testing.T) { + defer util.ResetGock() + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + + // Primaries stuck at block 1000, fallbacks far ahead at block 10000. + // Request block 9000 (0x2328). The gap from primary (1000) to 9000 is + // 8000 blocks — well beyond the default MaxRetryableBlockDistance of + // 128 — so checkUpstreamBlockAvailability classifies each primary's + // skip as NON-retryable (ErrUpstreamRequestSkipped wrapping + // ErrUpstreamBlockUnavailable). + // + // This mirrors the live B2 incident pattern: primaries fronting a + // stalled L2 reth pod return latestBlock far behind the chain head, + // while a third-party fallback (Ankr) is at the real head and can + // serve. Before the fix, lastErr stayed nil for non-retryable skips, + // the escape gate's `lastErr != nil` check failed, and clients got + // ErrUpstreamsExhausted. After the fix, lastErr is set unconditionally + // in the gate-skip branch and the escape fires. + network, _, _ := setupFailoverFixture(t, ctx, failoverFixtureOpts{ + primaryLatest: "0x3e8", // 1000 + fallbackLatest: "0x2710", // 10000 + enableFailover: true, + }) + + counter := telemetry.MetricNetworkFallbackEscapeTotal.WithLabelValues("main", "evm:999", "eth_call") + before := promUtil.ToFloat64(counter) + + req := ethCallRequest(1, "0x2328") // 9000 — 8000 blocks ahead of primary + req.SetNetwork(network) + resp, err := network.Forward(ctx, req) + require.NoError(t, err, + "request must succeed via fallback even when primary gate-skip is non-retryable "+ + "(distance > MaxRetryableBlockDistance)") + require.NotNil(t, resp) + defer resp.Release() + + jrr, err := resp.JsonRpcResponse() + require.NoError(t, err) + result := strings.Trim(jrr.GetResultString(), `"`) + assert.Contains(t, []string{"0x3333", "0x4444"}, result, + "non-retryable-gate-skip escape must route to a fallback; got %q", result) + + after := promUtil.ToFloat64(counter) + assert.Equal(t, before+1, after, + "escape hatch must fire exactly once for the non-retryable gate-skip case") + }) +} diff --git a/telemetry/metrics.go b/telemetry/metrics.go index ab2cea4ce..f32d5519e 100644 --- a/telemetry/metrics.go +++ b/telemetry/metrics.go @@ -409,6 +409,21 @@ var ( Help: "Total circuit-breaker state transitions per upstream and direction (closed_to_open/half_open_to_open/half_open_to_closed/open_to_half_open).", }, []string{"project", "upstream", "transition"}) + // MetricNetworkFallbackEscapeTotal counts firings of the per-request + // fallback escape hatch in Network.Forward: when the primary set is + // exhausted with retryable errors mid-request, eRPC appends cordoned + // fallback-group upstreams and continues iterating so the client gets a + // response instead of ErrUpstreamsExhausted. Expected to be zero in + // steady state. Sustained non-zero values during an outage are expected + // during the ~1m window before selectionPolicy promotes fallbacks into + // upsList directly; sustained non-zero in steady state indicates a + // config issue or regression. + MetricNetworkFallbackEscapeTotal = promauto.NewCounterVec(prometheus.CounterOpts{ + Namespace: "erpc", + Name: "network_fallback_escape_total", + Help: "Total number of times the per-request fallback escape hatch fired because the primary upstream set was exhausted with retryable errors.", + }, []string{"project", "network", "category"}) + MetricNetworkFailedRequests = promauto.NewCounterVec(prometheus.CounterOpts{ Namespace: "erpc", Name: "network_failed_request_total", diff --git a/upstream/registry.go b/upstream/registry.go index 2dc2fd398..6c5f43b1b 100644 --- a/upstream/registry.go +++ b/upstream/registry.go @@ -367,6 +367,40 @@ func (u *UpstreamsRegistry) GetNetworkUpstreams(ctx context.Context, networkId s return cp } +// GetFallbackEscapeUpstreams returns fallback-group upstreams for a network +// that are eligible to serve a per-request escape when the primary set has +// been exhausted with retryable errors. +// +// Filters by: +// - Group == UpstreamGroupFallback (the operator's explicit fallback tag) +// - Bootstrapped (present in networkUpstreams via GetNetworkUpstreams) +// - Not hard-down (IsDown == false; circuit breaker is closed) +// - Method allowed (ShouldHandleMethod respects IgnoreMethods / AllowMethods) +// +// Critically does NOT filter by metricsTracker.IsCordoned. The caller's +// intent is to escape past the selectionPolicy cordon for this single +// request. The selectionPolicy continues to govern steady-state routing +// via the score-based sorted list; this escape path is orthogonal and +// triggered only on per-request exhaustion in Network.Forward's inner loop. +func (u *UpstreamsRegistry) GetFallbackEscapeUpstreams(ctx context.Context, networkId, method string) []*Upstream { + all := u.GetNetworkUpstreams(ctx, networkId) + out := make([]*Upstream, 0, len(all)) + for _, up := range all { + cfg := up.Config() + if cfg == nil || !cfg.HasTag(common.TagTierFallback) { + continue + } + if up.IsDown() { + continue + } + if allowed, err := up.ShouldHandleMethod(method); err != nil || !allowed { + continue + } + out = append(out, up) + } + return out +} + // GetWsUpstreams returns all WS-capable upstreams for a network (ws:// or wss:// endpoints). func (u *UpstreamsRegistry) GetWsUpstreams(ctx context.Context, networkId string) []*Upstream { all := u.GetNetworkUpstreams(ctx, networkId) diff --git a/upstream/registry_fallback_escape_test.go b/upstream/registry_fallback_escape_test.go new file mode 100644 index 000000000..3762ac2f8 --- /dev/null +++ b/upstream/registry_fallback_escape_test.go @@ -0,0 +1,209 @@ +package upstream + +import ( + "context" + "testing" + "time" + + "github.com/erpc/erpc/common" + "github.com/erpc/erpc/data" + "github.com/erpc/erpc/health" + "github.com/erpc/erpc/thirdparty" + "github.com/erpc/erpc/util" + "github.com/rs/zerolog/log" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// setupRegistryWithFallbacks builds a 4-upstream registry (2 primaries + 2 +// fallbacks) for testing GetFallbackEscapeUpstreams. +func setupRegistryWithFallbacks(t *testing.T, ctx context.Context) (*UpstreamsRegistry, *health.Tracker) { + t.Helper() + util.ResetGock() + util.SetupMocksForEvmStatePoller() + + logger := &log.Logger + metricsTracker := health.NewTracker(logger, "fb-test", 10*time.Second) + metricsTracker.Bootstrap(ctx) + + upstreamConfigs := []*common.UpstreamConfig{ + { + Id: "primary-1", Type: common.UpstreamTypeEvm, + Endpoint: "http://rpc1.localhost", + Evm: &common.EvmUpstreamConfig{ChainId: 123}, + }, + { + Id: "primary-2", Type: common.UpstreamTypeEvm, + Endpoint: "http://rpc2.localhost", + Evm: &common.EvmUpstreamConfig{ChainId: 123}, + }, + { + Id: "fallback-1", Type: common.UpstreamTypeEvm, + Endpoint: "http://rpc3.localhost", + Group: common.UpstreamGroupFallback, + Evm: &common.EvmUpstreamConfig{ChainId: 123}, + }, + { + Id: "fallback-2", Type: common.UpstreamTypeEvm, + Endpoint: "http://rpc4.localhost", + Group: common.UpstreamGroupFallback, + Evm: &common.EvmUpstreamConfig{ChainId: 123}, + }, + } + + vr := thirdparty.NewVendorsRegistry() + pr, err := thirdparty.NewProvidersRegistry(logger, vr, nil, nil) + require.NoError(t, err) + + ssr, err := data.NewSharedStateRegistry(ctx, logger, &common.SharedStateConfig{ + Connector: &common.ConnectorConfig{ + Driver: "memory", + Memory: &common.MemoryConnectorConfig{MaxItems: 100_000, MaxTotalSize: "1GB"}, + }, + }) + require.NoError(t, err) + + registry := NewUpstreamsRegistry(ctx, logger, "fb-test", upstreamConfigs, + ssr, nil, vr, pr, nil, metricsTracker, + 1*time.Second, + &ScoringConfig{ScoreGranularity: "method", SwitchHysteresis: -1, MinSwitchInterval: -1}, + nil, + ) + + registry.Bootstrap(ctx) + time.Sleep(100 * time.Millisecond) + require.NoError(t, registry.PrepareUpstreamsForNetwork(ctx, "evm:123")) + return registry, metricsTracker +} + +func findUpstream(t *testing.T, registry *UpstreamsRegistry, id string) *Upstream { + t.Helper() + for _, u := range registry.GetNetworkUpstreams(context.Background(), "evm:123") { + if u.Id() == id { + return u + } + } + t.Fatalf("upstream %q not found", id) + return nil +} + +// TestGetFallbackEscapeUpstreams_FiltersByGroup verifies only fallback-group +// upstreams are returned. +func TestGetFallbackEscapeUpstreams_FiltersByGroup(t *testing.T) { + defer util.ResetGock() + defer util.AssertNoPendingMocks(t, 0) + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + registry, _ := setupRegistryWithFallbacks(t, ctx) + + out := registry.GetFallbackEscapeUpstreams(ctx, "evm:123", "eth_call") + ids := make([]string, 0, len(out)) + for _, u := range out { + ids = append(ids, u.Id()) + } + assert.ElementsMatch(t, []string{"fallback-1", "fallback-2"}, ids, + "only group=fallback upstreams should be returned; got %v", ids) +} + +// TestGetFallbackEscapeUpstreams_IgnoresCordon verifies cordoned fallbacks +// are STILL returned. This is the central property of the escape hatch: +// the cordon-by-selectionPolicy must NOT exclude upstreams from the escape +// path, because the cordon is exactly what the escape is designed to bypass. +func TestGetFallbackEscapeUpstreams_IgnoresCordon(t *testing.T) { + defer util.ResetGock() + defer util.AssertNoPendingMocks(t, 0) + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + registry, mt := setupRegistryWithFallbacks(t, ctx) + + fb1 := findUpstream(t, registry, "fallback-1") + fb2 := findUpstream(t, registry, "fallback-2") + + // Simulate the selectionPolicy cordoning both fallbacks. + mt.Cordon(fb1, "*", "excluded by selection policy") + mt.Cordon(fb2, "*", "excluded by selection policy") + require.True(t, mt.IsCordoned(fb1, "eth_call"), "fb1 should be cordoned") + require.True(t, mt.IsCordoned(fb2, "eth_call"), "fb2 should be cordoned") + + out := registry.GetFallbackEscapeUpstreams(ctx, "evm:123", "eth_call") + ids := make([]string, 0, len(out)) + for _, u := range out { + ids = append(ids, u.Id()) + } + assert.ElementsMatch(t, []string{"fallback-1", "fallback-2"}, ids, + "cordoned fallbacks must still be returned for the escape path; got %v", ids) +} + +// TestGetFallbackEscapeUpstreams_FiltersIgnoreMethods verifies fallbacks +// whose IgnoreMethods contains the caller method are excluded — the escape +// path respects per-upstream method capabilities. +func TestGetFallbackEscapeUpstreams_FiltersIgnoreMethods(t *testing.T) { + defer util.ResetGock() + defer util.AssertNoPendingMocks(t, 0) + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + registry, _ := setupRegistryWithFallbacks(t, ctx) + + fb1 := findUpstream(t, registry, "fallback-1") + // Mark eth_call as ignored on fallback-1; AllowMethods left empty. + fb1.config.IgnoreMethods = []string{"eth_call"} + + out := registry.GetFallbackEscapeUpstreams(ctx, "evm:123", "eth_call") + ids := make([]string, 0, len(out)) + for _, u := range out { + ids = append(ids, u.Id()) + } + assert.ElementsMatch(t, []string{"fallback-2"}, ids, + "fallback-1 should be excluded because eth_call is in its IgnoreMethods; got %v", ids) + + // Sanity: a different method that isn't ignored should include fallback-1. + out2 := registry.GetFallbackEscapeUpstreams(ctx, "evm:123", "eth_getLogs") + ids2 := make([]string, 0, len(out2)) + for _, u := range out2 { + ids2 = append(ids2, u.Id()) + } + assert.ElementsMatch(t, []string{"fallback-1", "fallback-2"}, ids2, + "eth_getLogs is not ignored; both fallbacks should be returned; got %v", ids2) +} + +// TestGetFallbackEscapeUpstreams_NoFallbacksReturnsEmpty verifies the +// helper returns an empty slice (not nil-causing-panic) when no fallbacks +// are configured. +func TestGetFallbackEscapeUpstreams_NoFallbacksReturnsEmpty(t *testing.T) { + util.ResetGock() + util.SetupMocksForEvmStatePoller() + defer util.ResetGock() + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + + // Build a registry with primaries only. + logger := &log.Logger + metricsTracker := health.NewTracker(logger, "fb-test", 10*time.Second) + metricsTracker.Bootstrap(ctx) + upstreamConfigs := []*common.UpstreamConfig{ + {Id: "primary-only", Type: common.UpstreamTypeEvm, Endpoint: "http://rpc1.localhost", Evm: &common.EvmUpstreamConfig{ChainId: 123}}, + } + vr := thirdparty.NewVendorsRegistry() + pr, err := thirdparty.NewProvidersRegistry(logger, vr, nil, nil) + require.NoError(t, err) + ssr, err := data.NewSharedStateRegistry(ctx, logger, &common.SharedStateConfig{ + Connector: &common.ConnectorConfig{Driver: "memory", Memory: &common.MemoryConnectorConfig{MaxItems: 1000, MaxTotalSize: "1GB"}}, + }) + require.NoError(t, err) + registry := NewUpstreamsRegistry(ctx, logger, "fb-test", upstreamConfigs, + ssr, nil, vr, pr, nil, metricsTracker, 1*time.Second, + &ScoringConfig{ScoreGranularity: "method", SwitchHysteresis: -1, MinSwitchInterval: -1}, + nil, + ) + registry.Bootstrap(ctx) + time.Sleep(100 * time.Millisecond) + require.NoError(t, registry.PrepareUpstreamsForNetwork(ctx, "evm:123")) + + out := registry.GetFallbackEscapeUpstreams(ctx, "evm:123", "eth_call") + assert.Empty(t, out, "no fallback-group upstreams configured → empty slice") +} + +// IsDown filter coverage: the IsDown filter is exercised indirectly via the +// higher-level erpc.TestFailover_EscapeHatch end-to-end test (no clean +// unit-level way to flip an Upstream's CB to open without driving real +// failures through the failsafe stack from this package). From 6fb6b51aa0ffd33bdf30b9af55dc2d2752ebc5c0 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Jo=C3=A3o=20Gomes?= Date: Thu, 28 May 2026 18:34:11 +0100 Subject: [PATCH 07/40] fix(cache): preserve "latest"/"finalized" tags for state-read methods MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit eth_getBalance, eth_getTransactionCount, eth_call, and eth_estimateGas now default to TranslateLatestTag=false and TranslateFinalizedTag=false — same pattern eth_getBlockByNumber already uses. Pre-fix behaviour on instant-finality chains (Pharos, evm:1672): NormalizeHttpJsonRpc translated "latest" into a concrete hex block N sourced from the state poller, which lags chain head by one polling interval. Network.GetFinality then classified N ≤ lowestFinalized as DataFinalityStateFinalized, and the production wildcard cache policy ("network:*, method:*, finality:finalized, ttl:720h") served the same historical answer at block N for the whole polling window. Chainlink saw a stale eth_getTransactionCount("latest"), kept rebroadcasting the same nonce, the upstream returned TX_REPLAY_ATTACK on every retry, and clients hit ErrUpstreamsExhausted instead of receiving the current chain state. Confirmed in prod on 2026-05-28 with ~19 nonces of stale return; eth_getBalance was similarly affected. With the tag preserved, ExtractBlockReferenceFromRequest returns blockRef="latest", GetFinality's non-numeric-tag fast path returns DataFinalityStateRealtime, the upstream evaluates at execution time, and ResolveCacheBlockRef on the cache layer keys per the network's current tip (each tip advance is a fresh cache key, concurrent "latest" requests within the same tip coalesce on one entry). eth_getCode is left translated — bytecode rarely changes and a translated-block cache hit is acceptable. Cross-upstream divergence at block N+1 vs N is now possible for these methods (already true for eth_getBlockByNumber since ba3e912) and is absorbed by the 5s realtime cache TTL. Tests: - networks_interpolation_test.go: existing tests that asserted "latest" was rewritten to hex on eth_getBalance/eth_call/eth_estimateGas now assert the literal tag is preserved; AllMethodsCoverage filters flipped per-method. - networks_finality_interpolation_test.go: rewritten — the test name was "PreservesRealtimeFinality across translation", which was only meaningful because Finality is cached on first call before normalization. With the tag preserved end-to-end, finality is Realtime by the natural code path (non-numeric tag) and the body of the test reflects that. - LatestTag_ToHex / FinalizedTag_ToHex / UpstreamSkipping_OnInterpolatedLatest switched to eth_getStorageAt / eth_getCode (methods that still translate by default). - New tests: StateMethods_PreserveLatestByDefault verifies the new defaults for all four methods, EthGetCode_StillInterpolatesLatest locks in the deliberate skip. Co-Authored-By: Claude Opus 4.7 (1M context) --- common/defaults.go | 20 ++- erpc/networks_finality_interpolation_test.go | 51 +++----- erpc/networks_interpolation_test.go | 131 +++++++++++++------ 3 files changed, 127 insertions(+), 75 deletions(-) diff --git a/common/defaults.go b/common/defaults.go index ed67efa8a..16a84ab12 100644 --- a/common/defaults.go +++ b/common/defaults.go @@ -318,18 +318,32 @@ var DefaultWithBlockCacheMethods = map[string]*CacheMethodConfig{ }, "eth_getBalance": { ReqRefs: SecondParam, + // Preserve "latest"/"finalized" tags so the upstream returns current + // chain state at execution time. On instant-finality chains every + // numeric block ≤ tip gets classified as Finalized, so translation + // would pin the request to a stale poller block and route it into + // the long-TTL finalized cache. The cache layer's ResolveCacheBlockRef + // still keys tag-preserved requests by the network's current tip. + TranslateLatestTag: util.BoolPtr(false), + TranslateFinalizedTag: util.BoolPtr(false), }, "eth_getTransactionCount": { - ReqRefs: SecondParam, + ReqRefs: SecondParam, + TranslateLatestTag: util.BoolPtr(false), + TranslateFinalizedTag: util.BoolPtr(false), }, "eth_getCode": { ReqRefs: SecondParam, }, "eth_call": { - ReqRefs: SecondParam, + ReqRefs: SecondParam, + TranslateLatestTag: util.BoolPtr(false), + TranslateFinalizedTag: util.BoolPtr(false), }, "eth_estimateGas": { - ReqRefs: SecondParam, + ReqRefs: SecondParam, + TranslateLatestTag: util.BoolPtr(false), + TranslateFinalizedTag: util.BoolPtr(false), }, "trace_call": { // Support both param orderings used in the wild: diff --git a/erpc/networks_finality_interpolation_test.go b/erpc/networks_finality_interpolation_test.go index 3b2f305ac..499c53941 100644 --- a/erpc/networks_finality_interpolation_test.go +++ b/erpc/networks_finality_interpolation_test.go @@ -18,24 +18,24 @@ func init() { util.ConfigureTestLogger() } -// Test that finality is preserved as "realtime" when translating tags +// Test that finality is "realtime" for eth_getBalance with "latest" — both +// before and after normalization. With TranslateLatestTag=false (default +// for state-reading methods), the literal tag is preserved on the wire and +// Network.GetFinality classifies the request as Realtime end-to-end. func TestInterpolation_PreservesRealtimeFinality(t *testing.T) { util.ResetGock() defer util.ResetGock() util.SetupMocksForEvmStatePoller() defer util.AssertNoPendingMocks(t, 0) - // Set up mock BEFORE network initialization gock.New("http://rpc1.localhost"). Post(""). Times(1). Filter(func(r *http.Request) bool { body := util.SafeReadBody(r) - // Request will have translated hex value instead of "latest" - // The actual value depends on what the state poller returns + // Request must carry the literal "latest" tag — no translation. return strings.Contains(body, "eth_getBalance") && - !strings.Contains(body, "\"latest\"") && // Should NOT have "latest" - strings.Contains(body, "\"0x") // Should have hex value + strings.Contains(body, "\"latest\"") }). Reply(200). JSON(map[string]interface{}{ @@ -49,58 +49,50 @@ func TestInterpolation_PreservesRealtimeFinality(t *testing.T) { network, _ := setupTestNetworkForInterpolation(t, ctx, nil) - // Create request with "latest" tag req := common.NewNormalizedRequest([]byte(`{"jsonrpc":"2.0","id":1,"method":"eth_getBalance","params":["0xabc","latest"]}`)) req.SetNetwork(network) - // Check finality BEFORE normalization finalityBefore := req.Finality(ctx) - assert.Equal(t, common.DataFinalityStateRealtime, finalityBefore, "Request with 'latest' tag should be realtime before translation") + assert.Equal(t, common.DataFinalityStateRealtime, finalityBefore, "Request with 'latest' tag should be realtime before normalization") - // Normalize the request (which will translate "latest" to hex) jrq, err := req.JsonRpcRequest() require.NoError(t, err) evm.NormalizeHttpJsonRpc(ctx, req, jrq) - // Verify the tag was translated params := jrq.Params require.Len(t, params, 2) blockParam := params[1].(string) - assert.True(t, strings.HasPrefix(blockParam, "0x"), "latest should be translated to hex") - assert.NotEqual(t, "latest", blockParam, "Should not be 'latest' anymore") + assert.Equal(t, "latest", blockParam, "eth_getBalance should preserve 'latest' tag (TranslateLatestTag default is false)") - // Check finality AFTER normalization - this is the critical test finalityAfter := req.Finality(ctx) - assert.Equal(t, common.DataFinalityStateRealtime, finalityAfter, "Request should STILL be realtime after tag translation") + assert.Equal(t, common.DataFinalityStateRealtime, finalityAfter, "Request should STILL be realtime after normalization") - // Forward the request and check response finality resp, err := network.Forward(ctx, req) require.NoError(t, err) require.NotNil(t, resp) defer resp.Release() - // Check response finality respFinality := resp.Finality(ctx) assert.Equal(t, common.DataFinalityStateRealtime, respFinality, "Response should also be realtime") } -// Test that finality is preserved for finalized tag +// Test that finality is "realtime" for eth_call with "finalized" tag, +// both before and after normalization. eth_call has TranslateFinalizedTag +// default false, so the literal tag flows through to the upstream. func TestInterpolation_PreservesFinalizedTagFinality(t *testing.T) { util.ResetGock() defer util.ResetGock() util.SetupMocksForEvmStatePoller() defer util.AssertNoPendingMocks(t, 0) - // Set up mock BEFORE network initialization gock.New("http://rpc1.localhost"). Post(""). Times(1). Filter(func(r *http.Request) bool { body := util.SafeReadBody(r) - // Should not have the original tag and should contain a hex block ref + // Request must carry the literal "finalized" tag — no translation. return strings.Contains(body, "eth_call") && - !strings.Contains(body, "\"finalized\"") && - strings.Contains(body, "\"0x") + strings.Contains(body, "\"finalized\"") }). Reply(200). JSON(map[string]interface{}{ @@ -114,30 +106,29 @@ func TestInterpolation_PreservesFinalizedTagFinality(t *testing.T) { network, _ := setupTestNetworkForInterpolation(t, ctx, nil) - // Create request with "finalized" tag req := common.NewNormalizedRequest([]byte(`{"jsonrpc":"2.0","id":1,"method":"eth_call","params":[{"to":"0xabc"},"finalized"]}`)) req.SetNetwork(network) - // Check finality BEFORE normalization finalityBefore := req.Finality(ctx) - assert.Equal(t, common.DataFinalityStateRealtime, finalityBefore, "Request with 'finalized' tag should be realtime before translation") + assert.Equal(t, common.DataFinalityStateRealtime, finalityBefore, "Request with 'finalized' tag should be realtime before normalization") - // Normalize the request jrq, err := req.JsonRpcRequest() require.NoError(t, err) evm.NormalizeHttpJsonRpc(ctx, req, jrq) - // Check finality AFTER normalization + params := jrq.Params + require.Len(t, params, 2) + blockParam := params[1].(string) + assert.Equal(t, "finalized", blockParam, "eth_call should preserve 'finalized' tag (TranslateFinalizedTag default is false)") + finalityAfter := req.Finality(ctx) - assert.Equal(t, common.DataFinalityStateRealtime, finalityAfter, "Request should STILL be realtime after finalized tag translation") + assert.Equal(t, common.DataFinalityStateRealtime, finalityAfter, "Request should STILL be realtime after normalization") - // Forward the request resp, err := network.Forward(ctx, req) require.NoError(t, err) if resp != nil { defer resp.Release() - // Check response finality respFinality := resp.Finality(ctx) assert.Equal(t, common.DataFinalityStateRealtime, respFinality, "Response should also be realtime") } diff --git a/erpc/networks_interpolation_test.go b/erpc/networks_interpolation_test.go index 751c47db1..351679584 100644 --- a/erpc/networks_interpolation_test.go +++ b/erpc/networks_interpolation_test.go @@ -83,21 +83,21 @@ func setupTestNetworkForInterpolation(t *testing.T, ctx context.Context, network return network, upr } -// Test "latest" tag translation to hex number +// Test "latest" tag translation to hex number for a method that still +// translates by default (eth_getStorageAt). eth_getBalance and friends +// preserve the tag — see TestInterpolation_StateMethods_PreserveLatestByDefault. func TestInterpolation_LatestTag_ToHex(t *testing.T) { util.ResetGock() defer util.ResetGock() util.SetupMocksForEvmStatePoller() defer util.AssertNoPendingMocks(t, 0) - // Mock the forwarded request to check the translated value gock.New("http://rpc1.localhost"). Post(""). Times(1). Filter(func(r *http.Request) bool { body := util.SafeReadBody(r) - // Should receive hex block number instead of "latest" - return strings.Contains(body, "eth_getBalance") && + return strings.Contains(body, "eth_getStorageAt") && strings.Contains(body, "\"0x") && !strings.Contains(body, "\"latest\"") }). @@ -113,22 +113,19 @@ func TestInterpolation_LatestTag_ToHex(t *testing.T) { network, _ := setupTestNetworkForInterpolation(t, ctx, nil) - req := common.NewNormalizedRequest([]byte(`{"jsonrpc":"2.0","id":1,"method":"eth_getBalance","params":["0xabc","latest"]}`)) + req := common.NewNormalizedRequest([]byte(`{"jsonrpc":"2.0","id":1,"method":"eth_getStorageAt","params":["0xabc","0x0","latest"]}`)) req.SetNetwork(network) - // Normalize the request jrq, err := req.JsonRpcRequest() require.NoError(t, err) evm.NormalizeHttpJsonRpc(ctx, req, jrq) - // Verify "latest" was translated to hex params := jrq.Params - require.Len(t, params, 2) - blockParam := params[1].(string) + require.Len(t, params, 3) + blockParam := params[2].(string) assert.True(t, strings.HasPrefix(blockParam, "0x"), "Block param should be hex") assert.NotEqual(t, "latest", blockParam, "Should not be 'latest' anymore") - // Forward to verify mock is hit resp, err := network.Forward(ctx, req) require.NoError(t, err) if resp != nil { @@ -541,19 +538,21 @@ func TestInterpolation_EvmBlockRef_IsPerRequest(t *testing.T) { } // Test "finalized" tag translation to hex number +// "finalized" translation default-on path — exercised via eth_getStorageAt +// (eth_call/eth_getBalance/eth_getTransactionCount/eth_estimateGas preserve +// the tag by default; see TestInterpolation_StateMethods_PreserveLatestByDefault). func TestInterpolation_FinalizedTag_ToHex(t *testing.T) { util.ResetGock() defer util.ResetGock() util.SetupMocksForEvmStatePoller() defer util.AssertNoPendingMocks(t, 0) - // Mock the forwarded request gock.New("http://rpc1.localhost"). Post(""). Times(1). Filter(func(r *http.Request) bool { body := util.SafeReadBody(r) - return strings.Contains(body, "eth_call") && + return strings.Contains(body, "eth_getStorageAt") && strings.Contains(body, "\"0x") && !strings.Contains(body, "\"finalized\"") }). @@ -561,7 +560,7 @@ func TestInterpolation_FinalizedTag_ToHex(t *testing.T) { JSON(map[string]interface{}{ "jsonrpc": "2.0", "id": 1, - "result": "0x", + "result": "0x0", }) ctx, cancel := context.WithCancel(context.Background()) @@ -569,7 +568,7 @@ func TestInterpolation_FinalizedTag_ToHex(t *testing.T) { network, _ := setupTestNetworkForInterpolation(t, ctx, nil) - req := common.NewNormalizedRequest([]byte(`{"jsonrpc":"2.0","id":1,"method":"eth_call","params":[{"to":"0xabc"},"finalized"]}`)) + req := common.NewNormalizedRequest([]byte(`{"jsonrpc":"2.0","id":1,"method":"eth_getStorageAt","params":["0xabc","0x0","finalized"]}`)) req.SetNetwork(network) jrq, err := req.JsonRpcRequest() @@ -577,8 +576,8 @@ func TestInterpolation_FinalizedTag_ToHex(t *testing.T) { evm.NormalizeHttpJsonRpc(ctx, req, jrq) params := jrq.Params - require.Len(t, params, 2) - blockParam := params[1].(string) + require.Len(t, params, 3) + blockParam := params[2].(string) assert.True(t, strings.HasPrefix(blockParam, "0x"), "Block param should be hex") assert.NotEqual(t, "finalized", blockParam, "Should not be 'finalized' anymore") @@ -1206,9 +1205,55 @@ func TestInterpolation_EthGetBlockByNumber_Finalized_NotInterpolated(t *testing. assert.Equal(t, "finalized", blockParam, "eth_getBlockByNumber should NOT interpolate 'finalized'") } -// Test that other methods like eth_getBalance STILL interpolate "latest". -// This ensures our change to eth_getBlockByNumber doesn't affect other methods. -func TestInterpolation_OtherMethods_StillInterpolateLatest(t *testing.T) { +// Test that state-reading methods preserve "latest" by default — same +// rationale as eth_getBlockByNumber. Pinning "latest" to a translated +// block number on instant-finality chains produces historical state via +// the long-TTL finalized cache; preserving the tag lets the upstream +// answer at execution time and the cache layer key by the network's +// current tip via ResolveCacheBlockRef. +func TestInterpolation_StateMethods_PreserveLatestByDefault(t *testing.T) { + cases := []struct { + method string + body string + }{ + {"eth_getBalance", `{"jsonrpc":"2.0","id":1,"method":"eth_getBalance","params":["0xabc","latest"]}`}, + {"eth_getTransactionCount", `{"jsonrpc":"2.0","id":1,"method":"eth_getTransactionCount","params":["0xabc","latest"]}`}, + {"eth_call", `{"jsonrpc":"2.0","id":1,"method":"eth_call","params":[{"to":"0xabc"},"latest"]}`}, + {"eth_estimateGas", `{"jsonrpc":"2.0","id":1,"method":"eth_estimateGas","params":[{"to":"0xabc"},"latest"]}`}, + } + + for _, tc := range cases { + t.Run(tc.method, func(t *testing.T) { + util.ResetGock() + defer util.ResetGock() + util.SetupMocksForEvmStatePoller() + + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + + network, _ := setupTestNetworkForInterpolation(t, ctx, nil) + + req := common.NewNormalizedRequest([]byte(tc.body)) + req.SetNetwork(network) + + jrq, err := req.JsonRpcRequest() + require.NoError(t, err) + evm.NormalizeHttpJsonRpc(ctx, req, jrq) + + params := jrq.Params + require.Len(t, params, 2) + blockParam, ok := params[1].(string) + require.True(t, ok, "second param should be the block tag string") + assert.Equal(t, "latest", blockParam, + "%s should NOT translate 'latest' by default — preserved tag flows through to upstream and cache layer", tc.method) + }) + } +} + +// Test that eth_getCode STILL interpolates "latest" by default. Bytecode +// rarely changes, so caching a translated block's response is acceptable +// and the cache hit rate benefit is worth keeping. +func TestInterpolation_EthGetCode_StillInterpolatesLatest(t *testing.T) { util.ResetGock() defer util.ResetGock() util.SetupMocksForEvmStatePoller() @@ -1217,15 +1262,12 @@ func TestInterpolation_OtherMethods_StillInterpolateLatest(t *testing.T) { ctx, cancel := context.WithCancel(context.Background()) defer cancel() - // Set up mock BEFORE network initialization - // eth_getBalance should have "latest" interpolated to hex gock.New("http://rpc1.localhost"). Post(""). Times(1). Filter(func(r *http.Request) bool { body := util.SafeReadBody(r) - // Should have hex block number, NOT "latest" - return strings.Contains(body, "eth_getBalance") && + return strings.Contains(body, "eth_getCode") && strings.Contains(body, "\"0x") && !strings.Contains(body, "\"latest\"") }). @@ -1233,27 +1275,24 @@ func TestInterpolation_OtherMethods_StillInterpolateLatest(t *testing.T) { JSON(map[string]interface{}{ "jsonrpc": "2.0", "id": 1, - "result": "0x1234", + "result": "0x", }) network, _ := setupTestNetworkForInterpolation(t, ctx, nil) - req := common.NewNormalizedRequest([]byte(`{"jsonrpc":"2.0","id":1,"method":"eth_getBalance","params":["0xabc","latest"]}`)) + req := common.NewNormalizedRequest([]byte(`{"jsonrpc":"2.0","id":1,"method":"eth_getCode","params":["0xabc","latest"]}`)) req.SetNetwork(network) - // Normalize the request jrq, err := req.JsonRpcRequest() require.NoError(t, err) evm.NormalizeHttpJsonRpc(ctx, req, jrq) - // Verify "latest" WAS translated to hex for eth_getBalance params := jrq.Params require.Len(t, params, 2) blockParam := params[1].(string) - assert.True(t, strings.HasPrefix(blockParam, "0x"), "eth_getBalance should interpolate 'latest' to hex") - assert.NotEqual(t, "latest", blockParam, "eth_getBalance should NOT have 'latest' anymore") + assert.True(t, strings.HasPrefix(blockParam, "0x"), "eth_getCode should interpolate 'latest' to hex") + assert.NotEqual(t, "latest", blockParam) - // Forward to verify mock is hit resp, err := network.Forward(ctx, req) require.NoError(t, err) if resp != nil { @@ -1457,15 +1496,14 @@ func TestInterpolation_EthEstimateGas_SecondParam(t *testing.T) { util.SetupMocksForEvmStatePoller() defer util.AssertNoPendingMocks(t, 0) - // Set up mock BEFORE network initialization gock.New("http://rpc1.localhost"). Post(""). Times(1). Filter(func(r *http.Request) bool { body := util.SafeReadBody(r) + // eth_estimateGas preserves "latest" by default (no translation). return strings.Contains(body, "eth_estimateGas") && - !strings.Contains(body, "\"latest\"") && - strings.Contains(body, "\"0x") + strings.Contains(body, "\"latest\"") }). Reply(200). JSON(map[string]interface{}{ @@ -1489,8 +1527,7 @@ func TestInterpolation_EthEstimateGas_SecondParam(t *testing.T) { params := jrq.Params require.Len(t, params, 2) blockParam := params[1].(string) - assert.True(t, strings.HasPrefix(blockParam, "0x"), "Second param should be hex") - assert.NotEqual(t, "latest", blockParam) + assert.Equal(t, "latest", blockParam, "eth_estimateGas should preserve 'latest' (TranslateLatestTag default false)") resp, err := network.Forward(ctx, req) require.NoError(t, err) @@ -1768,11 +1805,12 @@ func TestInterpolation_AllMethodsCoverage(t *testing.T) { expectedFilter func(body string) bool }{ { + // eth_getBalance preserves "latest" (TranslateLatestTag default false). name: "eth_getBalance", method: "eth_getBalance", params: `["0xabc","latest"]`, expectedFilter: func(body string) bool { - return strings.Contains(body, "eth_getBalance") && !strings.Contains(body, "\"latest\"") + return strings.Contains(body, "eth_getBalance") && strings.Contains(body, "\"latest\"") }, }, { @@ -1784,14 +1822,17 @@ func TestInterpolation_AllMethodsCoverage(t *testing.T) { }, }, { + // eth_getTransactionCount preserves "latest" (TranslateLatestTag default false). name: "eth_getTransactionCount", method: "eth_getTransactionCount", params: `["0xabc","latest"]`, expectedFilter: func(body string) bool { - return strings.Contains(body, "eth_getTransactionCount") && !strings.Contains(body, "\"latest\"") + return strings.Contains(body, "eth_getTransactionCount") && strings.Contains(body, "\"latest\"") }, }, { + // eth_getCode still translates "finalized" — bytecode rarely changes, + // keeping interpolation lets the long-TTL finalized cache work. name: "eth_getCode", method: "eth_getCode", params: `["0xabc","finalized"]`, @@ -1800,19 +1841,21 @@ func TestInterpolation_AllMethodsCoverage(t *testing.T) { }, }, { + // eth_call preserves "latest" (TranslateLatestTag default false). name: "eth_call", method: "eth_call", params: `[{"to":"0xabc"},"latest"]`, expectedFilter: func(body string) bool { - return strings.Contains(body, "eth_call") && !strings.Contains(body, "\"latest\"") + return strings.Contains(body, "eth_call") && strings.Contains(body, "\"latest\"") }, }, { + // eth_estimateGas preserves "latest" (TranslateLatestTag default false). name: "eth_estimateGas", method: "eth_estimateGas", params: `[{"to":"0xabc"},"latest"]`, expectedFilter: func(body string) bool { - return strings.Contains(body, "eth_estimateGas") && !strings.Contains(body, "\"latest\"") + return strings.Contains(body, "eth_estimateGas") && strings.Contains(body, "\"latest\"") }, }, { @@ -1922,13 +1965,17 @@ func TestInterpolation_UpstreamSkipping_OnInterpolatedLatest(t *testing.T) { }, } - // Only set user mock for rpc3 (highest latest) - this should be used + // Only set user mock for rpc3 (highest latest) - this should be used. + // Use eth_getCode here because it still has TranslateLatestTag=true by + // default (eth_getBalance/eth_getTransactionCount/eth_call/eth_estimateGas + // preserve the tag instead, which would defeat the upstream-skipping + // path under test). gock.New("http://rpc3.localhost"). Post(""). Times(1). Filter(func(r *http.Request) bool { body := util.SafeReadBody(r) - return strings.Contains(body, "eth_getBalance") && strings.Contains(body, "\"0x") && !strings.Contains(body, "\"latest\"") + return strings.Contains(body, "eth_getCode") && strings.Contains(body, "\"0x") && !strings.Contains(body, "\"latest\"") }). Reply(200). JSON(map[string]interface{}{ @@ -1973,7 +2020,7 @@ func TestInterpolation_UpstreamSkipping_OnInterpolatedLatest(t *testing.T) { network.PinUpstreamOrderForTest() // Build request with 'latest' which will be interpolated to highest latest block (from rpc3) - req := common.NewNormalizedRequest([]byte(`{"jsonrpc":"2.0","id":1,"method":"eth_getBalance","params":["0xabc","latest"]}`)) + req := common.NewNormalizedRequest([]byte(`{"jsonrpc":"2.0","id":1,"method":"eth_getCode","params":["0xabc","latest"]}`)) req.SetNetwork(network) jrq, err := req.JsonRpcRequest() From 9372a1022d019a7183ba4db6034740fa6a557e67 Mon Sep 17 00:00:00 2001 From: Jonny Date: Mon, 1 Jun 2026 10:00:39 +0100 Subject: [PATCH 08/40] fix(eth_sendRawTransaction): treat replay-attack rejections as idempotent success Some chains reject re-submission of an already-accepted transaction with a "replay attack" error instead of a standard "already known" message. When the identical signed tx is broadcast more than once (e.g. parallel/hedged sends), the first submission lands and the duplicates surface this error. Without mapping it to the idempotency path, those duplicates are treated as upstream errors, trip circuit breakers, and bubble up as ErrUpstreamsExhausted even though the transaction confirmed on-chain. Map the replay-attack rejection to NonceExceptionReasonAlreadyKnown alongside the existing duplicate-transaction phrases. Co-Authored-By: Claude Opus 4.8 (1M context) --- architecture/evm/error_normalizer.go | 4 +- architecture/evm/error_normalizer_test.go | 52 +++++++++++++++++++++++ 2 files changed, 55 insertions(+), 1 deletion(-) diff --git a/architecture/evm/error_normalizer.go b/architecture/evm/error_normalizer.go index 564146a30..8ed5d0d5c 100644 --- a/architecture/evm/error_normalizer.go +++ b/architecture/evm/error_normalizer.go @@ -309,7 +309,9 @@ func ExtractJsonRpcError(r *http.Response, nr *common.NormalizedResponse, jr *co strings.Contains(ml, "already in the mempool") || strings.Contains(ml, "transaction already exists") || strings.Contains(ml, "already have transaction") || - strings.Contains(ml, "already exists in mempool") { + strings.Contains(ml, "already exists in mempool") || + strings.Contains(ml, "tx_replay_attack") || + strings.Contains(ml, "replay attack") { // These indicate the exact same transaction is already known - idempotent success case return common.NewErrEndpointNonceException( common.NewErrJsonRpcExceptionInternal( diff --git a/architecture/evm/error_normalizer_test.go b/architecture/evm/error_normalizer_test.go index 090514e9b..fb683a829 100644 --- a/architecture/evm/error_normalizer_test.go +++ b/architecture/evm/error_normalizer_test.go @@ -1,6 +1,7 @@ package evm import ( + "errors" "net/http" "testing" @@ -55,3 +56,54 @@ func TestExtractJsonRpcError_RequestTooLargeNormalization(t *testing.T) { }) } } + +// TestExtractJsonRpcError_ReplayAttackIdempotency verifies that a re-submission +// of an already-accepted transaction rejected with a "replay attack" error is +// normalized to ErrEndpointNonceException with reason "already known", so +// eth_sendRawTransaction idempotency handling can convert it to success. +func TestExtractJsonRpcError_ReplayAttackIdempotency(t *testing.T) { + t.Parallel() + + cases := []struct { + name string + message string + }{ + { + name: "uppercase errmsg token", + message: "errcode: 113, errmsg: TX_REPLAY_ATTACK", + }, + { + name: "spaced phrasing", + message: "transaction rejected: replay attack detected", + }, + } + + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + t.Parallel() + + r := &http.Response{StatusCode: 200, Header: http.Header{}} + jrErr := common.NewErrJsonRpcExceptionExternal( + int(common.JsonRpcErrorServerSideException), + tc.message, + "", + ) + jr := common.MustNewJsonRpcResponse(1, nil, jrErr) + + err := ExtractJsonRpcError(r, nil, jr, nil) + if err == nil { + t.Fatalf("expected error, got nil") + } + if !common.HasErrorCode(err, common.ErrCodeEndpointNonceException) { + t.Fatalf("expected ErrEndpointNonceException, got %T: %v", err, err) + } + var ne *common.ErrEndpointNonceException + if !errors.As(err, &ne) { + t.Fatalf("expected *common.ErrEndpointNonceException in chain, got %T", err) + } + if got := ne.Details["nonceExceptionReason"]; got != string(common.NonceExceptionReasonAlreadyKnown) { + t.Fatalf("expected reason %q, got %v", common.NonceExceptionReasonAlreadyKnown, got) + } + }) + } +} From 454c54dade5b88a0073d214b02719fbf8a783974 Mon Sep 17 00:00:00 2001 From: Jonny Date: Mon, 1 Jun 2026 13:17:41 +0100 Subject: [PATCH 09/40] test: adapt failover/selection tests to rebased upstream API MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit After rebasing onto upstream/main, several tests referenced APIs the upstream refactors removed or reshaped. Adapt them to the new model: - Group field → `tier:fallback` tag (common.TagTierFallback) in upstream config literals. - NewUpstreamsRegistry: drop removed scoreRefreshInterval / ScoringConfig args; NewNetwork: pass the new *policy.Engine arg. - Selection policy: EvalFunctionSource/EvalFunction/selectionPolicyEvaluator are gone (replaced by policy.Engine); drive/observe the engine instead. - Tracker recording calls take a DataFinalityState argument. Failover-escape expectation updates (upstream's default selection policy now natively tiers to `tier:fallback` via preferTag, independent of our per-request escape hatch): - NoEscapeWhenFailoverDisabled: assert the escape *counter* stays flat (our opt-out) rather than expecting ErrUpstreamsExhausted — the request may still be served by the fallback tier via native policy routing. - OnlyEscalatesOncePerRequest: assert the escape fires exactly once (no re-escalation loop) rather than expecting exhaustion. - Remove TestNetworkConfig_SetDefaults_FailoverSkipsAutoSelectionPolicy: it verified the `!Failover.Enabled()` auto-policy guard, which is moot now that NewDefaultNetworkConfig returns an empty config (upstream's default already keeps fallback upstreams eligible). Co-Authored-By: Claude Opus 4.8 (1M context) --- common/config_test.go | 67 +-- erpc/networks_failover_escape_test.go | 488 +++++++++------------- erpc/networks_test.go | 26 +- upstream/registry_fallback_escape_test.go | 9 +- 4 files changed, 227 insertions(+), 363 deletions(-) diff --git a/common/config_test.go b/common/config_test.go index dc3c0c868..1a0e0c3fd 100644 --- a/common/config_test.go +++ b/common/config_test.go @@ -129,16 +129,16 @@ projects: } // TestLoadConfig_TypeScriptUnifiedPipeline pins the TS load path: -// 1. function-valued `evalFunc` survives as a real sobek function -// (NOT stringified) — `SelectionPolicy.EvalFunc` carries only a -// `__ts_fn__:` sentinel pointing into the user-script's -// `globalThis.__erpcFns` registry; -// 2. the user's whole compiled module is attached to `cfg.UserScript` -// so each policy-engine pool runtime can re-evaluate it natively, -// preserving closures + helpers; -// 3. the legacy `group:` key written via TS still flows through the -// shadow types and gets migrated to a `tier:` tag identically to -// the YAML path, and first-class `routing:` parses onto u.Routing. +// 1. function-valued `evalFunc` survives as a real sobek function +// (NOT stringified) — `SelectionPolicy.EvalFunc` carries only a +// `__ts_fn__:` sentinel pointing into the user-script's +// `globalThis.__erpcFns` registry; +// 2. the user's whole compiled module is attached to `cfg.UserScript` +// so each policy-engine pool runtime can re-evaluate it natively, +// preserving closures + helpers; +// 3. the legacy `group:` key written via TS still flows through the +// shadow types and gets migrated to a `tier:` tag identically to +// the YAML path, and first-class `routing:` parses onto u.Routing. // // We don't run the legacy translator hook here — that has its own // suite. This test just verifies that the TS object survives the @@ -1212,52 +1212,6 @@ projects: }) } -// TestNetworkConfig_SetDefaults_FailoverSkipsAutoSelectionPolicy verifies -// that enabling failover.onDefaultsExhausted=true suppresses the auto-applied -// SelectionPolicy that would otherwise filter fallback-group upstreams out -// of the eligible set. The per-request loop needs fallbacks to remain -// visible so it can escalate to them on demand. -func TestNetworkConfig_SetDefaults_FailoverSkipsAutoSelectionPolicy(t *testing.T) { - upstreams := []*UpstreamConfig{ - {Id: "a", Endpoint: "http://a", Type: UpstreamTypeEvm}, - {Id: "b", Endpoint: "http://b", Type: UpstreamTypeEvm, Group: UpstreamGroupFallback}, - } - - t.Run("without failover the auto policy is applied", func(t *testing.T) { - n := &NetworkConfig{Architecture: ArchitectureEvm, Evm: &EvmNetworkConfig{ChainId: 1}} - err := n.SetDefaults(upstreams, nil) - assert.NoError(t, err) - assert.NotNil(t, n.SelectionPolicy, "auto SelectionPolicy should be applied when fallback upstreams exist") - }) - - t.Run("with failover enabled the auto policy is suppressed", func(t *testing.T) { - enabled := true - n := &NetworkConfig{ - Architecture: ArchitectureEvm, - Evm: &EvmNetworkConfig{ChainId: 1}, - Failover: &FailoverConfig{OnDefaultsExhausted: &enabled}, - } - err := n.SetDefaults(upstreams, nil) - assert.NoError(t, err) - assert.Nil(t, n.SelectionPolicy, "auto SelectionPolicy should NOT be applied when failover handles escalation") - }) - - t.Run("user-supplied SelectionPolicy is preserved regardless of failover", func(t *testing.T) { - enabled := true - userPolicy := &SelectionPolicyConfig{EvalInterval: Duration(time.Minute)} - n := &NetworkConfig{ - Architecture: ArchitectureEvm, - Evm: &EvmNetworkConfig{ChainId: 1}, - Failover: &FailoverConfig{OnDefaultsExhausted: &enabled}, - SelectionPolicy: userPolicy, - } - err := n.SetDefaults(upstreams, nil) - assert.NoError(t, err) - assert.NotNil(t, n.SelectionPolicy) - assert.Equal(t, Duration(time.Minute), n.SelectionPolicy.EvalInterval) - }) -} - // TestNetworkConfig_SetDefaults_FailoverInheritsFromDefaults verifies that a // failover flag set at the NetworkDefaults level is propagated to networks // that don't override it. @@ -1272,4 +1226,3 @@ func TestNetworkConfig_SetDefaults_FailoverInheritsFromDefaults(t *testing.T) { assert.NotNil(t, n.Failover) assert.True(t, n.Failover.Enabled()) } - diff --git a/erpc/networks_failover_escape_test.go b/erpc/networks_failover_escape_test.go index 2374f576d..91a4204f5 100644 --- a/erpc/networks_failover_escape_test.go +++ b/erpc/networks_failover_escape_test.go @@ -11,6 +11,8 @@ import ( "github.com/erpc/erpc/common" "github.com/erpc/erpc/data" "github.com/erpc/erpc/health" + "github.com/erpc/erpc/internal/policy" + policystdlib "github.com/erpc/erpc/internal/policy/stdlib" "github.com/erpc/erpc/telemetry" "github.com/erpc/erpc/thirdparty" "github.com/erpc/erpc/upstream" @@ -22,29 +24,33 @@ import ( "github.com/stretchr/testify/require" ) -// failoverSelectionPolicy is a representative production selectionPolicy that -// cordons fallback-group upstreams while any primary is healthy. Without the -// gate-skip/CB-open tracker fixes plus the per-request fallback escape hatch, -// a primary collapse leaves the policy unable to detect the failure and -// fallbacks remain cordoned out of upsList, producing client-facing -// ErrUpstreamsExhausted until the next eval tick (1m worst case). -const failoverSelectionPolicy = ` -(upstreams, method) => { - const isFallback = u => u && u.config && u.config.group === 'fallback' - const healthOK = u => { - const m = (u && u.metrics) || {} - const err = m.errorRate - const lag = m.blockHeadLag - return (err == null || err < 0.5) && (lag == null || lag < 5) - } - const primary = upstreams.filter(u => !isFallback(u)) - const fallback = upstreams.filter(isFallback) - const healthy = primary.filter(healthOK) - if (healthy.length > 0) return healthy - if (fallback.length > 0) return fallback - return upstreams -} -` +// NOTE (post-refactor adaptation): +// +// This file originally drove a hand-written `selectionPolicy` JS function +// that keyed off `u.config.group === 'fallback'`. After the policy refactor +// the per-request `selectionPolicyEvaluator` / `AcquirePermit` API is gone; +// selection is pre-computed per (network, method) tick by `internal/policy`'s +// Engine, and fallback-tier upstreams are declared via the +// `tier:fallback` tag (constant `common.TagTierFallback`). +// +// The production default policy (internal/policy/default_policy.js) already +// expresses exactly the behaviour these tests need via +// .preferTag('!tier:fallback', { minHealthy: 1, fallback: 'tier:fallback' }) +// which cordons fallback-tagged upstreams out of the ordered list while at +// least one primary survives the health excludes, and promotes the +// fallbacks once every primary is excluded (e.g. by error-rate/lag). +// +// So instead of a custom JS function + `AcquirePermit` introspection, these +// tests now build a real `*policy.Engine` running the default policy and +// assert OBSERVABLE behaviour: +// - tracker metrics moved (errorRate) via GetUpstreamMethodMetrics(..., finality) +// - the policy's ordered list (PolicyOrderedUpstreams / the engine's +// LatestDecisionOutputForTest) cordons the primary / promotes the fallback +// - which upstream actually served the response +// - the erpc_network_fallback_escape_total metric count +// +// The ticker is frozen (EvalInterval=0); tests drive ticks via +// `policy.TickForTest`. // mockJsonRpcUpstream wires the standard state-poller mocks for one upstream // at a fixed block height. Used to build the multi-primary + multi-fallback @@ -103,51 +109,11 @@ func mockEthCallReturning(host string, resultHex string) { JSON([]byte(fmt.Sprintf(`{"jsonrpc":"2.0","id":1,"result":"%s"}`, resultHex))) } -// TestFailover_GateSkipsAccumulateErrorRate verifies the recording fixes -// (handleBlockSkip → tracker, state-poller CB-open → tracker). With the -// fixes in place, a gate-rejection burst against a primary stuck behind -// the network aggregator must move errorRate past 0.5, the selectionPolicy -// must cordon the primary on its next eval tick, and the fallbacks must -// be promoted into upsList so subsequent client requests succeed via them. -// -// Pre-fix behaviour (verified via git stash on this same test): gate -// rejections in checkUpstreamBlockAvailability do not call -// RecordUpstreamFailure, so the primary's errorRate stays at 0 -// indefinitely, the policy keeps the primary in `healthy`, and the -// fallbacks remain cordoned out of upsList. The final assertion fails -// with the primary still routable and the fallbacks not. -// -// Post-fix behaviour: handleBlockSkip records (Request, Failure) on every -// retryable gate-reject, errorRate crosses 0.5 within a single eval window -// once the burst starts, the next policy tick excludes the primary, and -// the fallbacks take over. -func TestFailover_GateSkipsAccumulateErrorRate(t *testing.T) { - util.ResetGock() - defer util.ResetGock() - - ctx, cancel := context.WithCancel(context.Background()) - defer cancel() - - // Two HTTP primaries stuck at block 1000 (analogous to a node that - // can't advance — pollers freeze, but the network aggregator - // continues advancing via the fallbacks' independent pollers). - const chainIdHex = "0x3e7" // 999 - const primaryLatest = "0x3e8" - const fallbackLatest = "0x3ea" - const finalizedHex = "0x3e0" - const requestBlock = "0x3ea" // beyond primary cache, gate-rejects on primaries - - mockJsonRpcUpstream("rpc1.localhost", chainIdHex, primaryLatest, finalizedHex) - mockJsonRpcUpstream("rpc2.localhost", chainIdHex, primaryLatest, finalizedHex) - mockJsonRpcUpstream("rpc3.localhost", chainIdHex, fallbackLatest, finalizedHex) - mockJsonRpcUpstream("rpc4.localhost", chainIdHex, fallbackLatest, finalizedHex) - - mockEthCallReturning("rpc1.localhost", "0x1111") - mockEthCallReturning("rpc2.localhost", "0x2222") - mockEthCallReturning("rpc3.localhost", "0x3333") - mockEthCallReturning("rpc4.localhost", "0x4444") - - upstreamConfigs := []*common.UpstreamConfig{ +// failoverUpstreamConfigs builds the standard 4-upstream layout used by the +// failover tests: 2 primaries + 2 fallbacks (the latter tagged +// `common.TagTierFallback`). Hosts are rpc1..rpc4.localhost. +func failoverUpstreamConfigs() []*common.UpstreamConfig { + return []*common.UpstreamConfig{ { Type: common.UpstreamTypeEvm, Id: "primary-1", Endpoint: "http://rpc1.localhost", @@ -169,7 +135,7 @@ func TestFailover_GateSkipsAccumulateErrorRate(t *testing.T) { { Type: common.UpstreamTypeEvm, Id: "fallback-1", Endpoint: "http://rpc3.localhost", - Group: common.UpstreamGroupFallback, + Tags: []string{common.TagTierFallback}, Evm: &common.EvmUpstreamConfig{ ChainId: 999, StatePollerInterval: common.Duration(100 * time.Millisecond), @@ -179,7 +145,7 @@ func TestFailover_GateSkipsAccumulateErrorRate(t *testing.T) { { Type: common.UpstreamTypeEvm, Id: "fallback-2", Endpoint: "http://rpc4.localhost", - Group: common.UpstreamGroupFallback, + Tags: []string{common.TagTierFallback}, Evm: &common.EvmUpstreamConfig{ ChainId: 999, StatePollerInterval: common.Duration(100 * time.Millisecond), @@ -187,14 +153,27 @@ func TestFailover_GateSkipsAccumulateErrorRate(t *testing.T) { }, }, } +} - rlr, _ := upstream.NewRateLimitersRegistry(context.Background(), &common.RateLimiterConfig{}, &log.Logger) +// buildFailoverNetwork wires a real Network + UpstreamsRegistry + +// policy.Engine running the production default selection policy (which +// cordons `tier:fallback` upstreams while a primary survives). The engine +// ticker is frozen; the caller drives ticks via policy.TickForTest. +func buildFailoverNetwork( + t *testing.T, ctx context.Context, + upstreamConfigs []*common.UpstreamConfig, + enableFailover bool, +) (*Network, *upstream.UpstreamsRegistry, *health.Tracker) { + t.Helper() + + rlr, err := upstream.NewRateLimitersRegistry(context.Background(), &common.RateLimiterConfig{}, &log.Logger) + require.NoError(t, err) // 5s metrics window so accumulated request/failure samples don't get - // reset mid-test. Pre-failure successes accumulate during initial - // warmup but are dominated quickly once the gate-rejection burst starts. + // reset mid-test. mt := health.NewTracker(&log.Logger, "main", 5*time.Second) vr := thirdparty.NewVendorsRegistry() - pr, _ := thirdparty.NewProvidersRegistry(&log.Logger, vr, nil, nil) + pr, err := thirdparty.NewProvidersRegistry(&log.Logger, vr, nil, nil) + require.NoError(t, err) sharedStateCfg := &common.SharedStateConfig{ Connector: &common.ConnectorConfig{ @@ -215,39 +194,27 @@ func TestFailover_GateSkipsAccumulateErrorRate(t *testing.T) { upr := upstream.NewUpstreamsRegistry( ctx, &log.Logger, "main", upstreamConfigs, - ssr, rlr, vr, pr, nil, mt, 1*time.Second, nil, nil, + ssr, rlr, vr, pr, nil, mt, nil, ) - // The production-style selectionPolicy compressed to fire every 150ms - // so the test doesn't have to wait the production 1m interval. - evalFn, err := common.CompileFunction(failoverSelectionPolicy) - require.NoError(t, err) - selectionPolicy := &common.SelectionPolicyConfig{ - EvalInterval: common.Duration(150 * time.Millisecond), - EvalFunctionSource: failoverSelectionPolicy, - EvalFunction: evalFn, - EvalPerMethod: false, - } - require.NoError(t, selectionPolicy.SetDefaults()) - - maxRetryable := int64(128) - enforce := true networkConfig := &common.NetworkConfig{ Architecture: common.ArchitectureEvm, Evm: &common.EvmNetworkConfig{ - ChainId: 999, - MaxRetryableBlockDistance: &maxRetryable, - EnforceBlockAvailability: &enforce, + ChainId: 999, + EnforceBlockAvailability: util.BoolPtr(true), }, - SelectionPolicy: selectionPolicy, - // Failover.onDefaultsExhausted = true so once fallbacks are in upsList, - // they're tried intra-request after primaries fail. - Failover: &common.FailoverConfig{ - OnDefaultsExhausted: &enforce, + // Default selection policy, frozen ticker; tests drive ticks. + SelectionPolicy: &common.SelectionPolicyConfig{ + EvalInterval: 0, }, } + if enableFailover { + networkConfig.Failover = &common.FailoverConfig{OnDefaultsExhausted: util.BoolPtr(true)} + } + + policyEngine := policy.NewEngine(ctx, &log.Logger, "main", mt, policystdlib.Install, nil) - network, err := NewNetwork(ctx, &log.Logger, "main", networkConfig, rlr, upr, mt) + network, err := NewNetwork(ctx, &log.Logger, "main", networkConfig, rlr, upr, mt, policyEngine) require.NoError(t, err) upr.Bootstrap(ctx) @@ -256,7 +223,7 @@ func TestFailover_GateSkipsAccumulateErrorRate(t *testing.T) { require.NoError(t, upr.PrepareUpstreamsForNetwork(ctx, util.EvmNetworkId(999))) require.NoError(t, network.Bootstrap(ctx)) - // Bootstrap each upstream's state poller; they'll fetch latest/finalized + // Bootstrap each upstream's state poller; they fetch latest/finalized // from the gock mocks and populate the per-upstream shared counters. upsList := upr.GetNetworkUpstreams(ctx, util.EvmNetworkId(999)) require.Len(t, upsList, 4) @@ -265,22 +232,83 @@ func TestFailover_GateSkipsAccumulateErrorRate(t *testing.T) { } // Let pollers run a couple of cycles so per-upstream LatestBlock counters - // are populated and the selection policy has evaluated at least once. + // are populated. time.Sleep(500 * time.Millisecond) - // Sanity-check the initial state: - // - Primaries should have LatestBlock = 1000 - // - Fallbacks should have LatestBlock = 1002 - // - Policy should keep primaries active and cordon fallbacks + return network, upr, mt +} + +// finalityForRequest computes the finality the gate-skip recording path uses +// for a request, so assertions read the same tracker bucket. We read the +// all-finalities aggregate via DataFinalityStateAll which is fed by every +// Record* regardless of the request's specific finality, so this is mostly +// documentation — DataFinalityStateAll is the safe key. + +func mustGetUpstream(ups []*upstream.Upstream, id string) *upstream.Upstream { + for _, u := range ups { + if u.Id() == id { + return u + } + } + panic("upstream not found: " + id) +} + +// TestFailover_GateSkipsAccumulateErrorRate verifies the recording fix: +// block-availability gate-skips (handleBlockSkip) now record (Request, +// Failure) on the health tracker so the upstream's errorRate moves. With the +// tracker reflecting the skips, the default selection policy's +// `errorRateAbove(0.7)` exclude eventually cordons the primary on the next +// eval tick, promoting the fallbacks via `preferTag`, and a client request +// then succeeds via a fallback. +// +// Pre-fix behaviour: gate rejections in checkUpstreamBlockAvailability did +// not call RecordUpstreamFailure, so the primary's errorRate stayed at 0 +// indefinitely, the policy kept the primary, and the fallbacks remained +// cordoned. Post-fix: handleBlockSkip records on every retryable gate-reject. +func TestFailover_GateSkipsAccumulateErrorRate(t *testing.T) { + util.ResetGock() + defer util.ResetGock() + + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + + // Two HTTP primaries stuck at block 1000; two fallbacks at 1002. + const chainIdHex = "0x3e7" // 999 + const primaryLatest = "0x3e8" + const fallbackLatest = "0x3ea" + const finalizedHex = "0x3e0" + const requestBlock = "0x3ea" // 1002 — beyond primary cache, gate-rejects on primaries + + mockJsonRpcUpstream("rpc1.localhost", chainIdHex, primaryLatest, finalizedHex) + mockJsonRpcUpstream("rpc2.localhost", chainIdHex, primaryLatest, finalizedHex) + mockJsonRpcUpstream("rpc3.localhost", chainIdHex, fallbackLatest, finalizedHex) + mockJsonRpcUpstream("rpc4.localhost", chainIdHex, fallbackLatest, finalizedHex) + + mockEthCallReturning("rpc1.localhost", "0x1111") + mockEthCallReturning("rpc2.localhost", "0x2222") + mockEthCallReturning("rpc3.localhost", "0x3333") + mockEthCallReturning("rpc4.localhost", "0x4444") + + // Failover enabled so once fallbacks are promoted (or via the escape + // hatch) the final request is served by a fallback. + network, upr, mt := buildFailoverNetwork(t, ctx, failoverUpstreamConfigs(), true) + + upsList := upr.GetNetworkUpstreams(ctx, util.EvmNetworkId(999)) + require.Len(t, upsList, 4) primaryUp := mustGetUpstream(upsList, "primary-1") fallbackUp := mustGetUpstream(upsList, "fallback-1") require.Equal(t, int64(1000), primaryUp.EvmStatePoller().LatestBlock(), "primary should be at block 1000") require.Equal(t, int64(1002), fallbackUp.EvmStatePoller().LatestBlock(), "fallback should be at block 1002") - // Verify policy initially returns primaries only — fallback should be cordoned. - require.NoError(t, network.selectionPolicyEvaluator.AcquirePermit(&log.Logger, primaryUp, "eth_call"), - "primary should be active initially") - require.Error(t, network.selectionPolicyEvaluator.AcquirePermit(&log.Logger, fallbackUp, "eth_call"), + // Initial tick: primaries are healthy → default policy's preferTag keeps + // only primaries in the ordered list and cordons the fallbacks out. + policy.ResetSlotStateForTest(network.policyEngine, network.networkId, "*") + policy.TickForTest(network.policyEngine, network.networkId, "*") + + order := network.PolicyOrderedUpstreams("eth_call") + require.NotEmpty(t, order, "policy must have produced an ordered list") + assert.Contains(t, order, "primary-1", "primary should be active initially") + assert.NotContains(t, order, "fallback-1", "fallback should be cordoned initially (primaries are healthy)") // --- Phase 2: trigger the gate-rejection burst --- @@ -289,10 +317,6 @@ func TestFailover_GateSkipsAccumulateErrorRate(t *testing.T) { // per-upstream cache (1000 on primaries) and rejects. With the // handleBlockSkip recording fix, each rejection records (request, // failure) on the metrics tracker for the primary. - // - // Send enough requests to dominate the pre-burst success ratio. With - // the 5s window and ~10 prior bootstrap-related successes, ~50 failures - // is comfortably > 0.5 errorRate. const burstSize = 80 for i := 0; i < burstSize; i++ { req := common.NewNormalizedRequest([]byte(fmt.Sprintf( @@ -303,44 +327,39 @@ func TestFailover_GateSkipsAccumulateErrorRate(t *testing.T) { _, _ = network.Forward(ctx, req) } - // Wait for at least one full selectionPolicy eval tick (150ms) so the - // policy observes the elevated errorRate and applies the cordon, plus - // one score-refresh tick (1s) so the registry rebuilds sortedUpstreams - // with fallbacks included. Then force an extra refresh to remove any - // remaining timing slack. - time.Sleep(400 * time.Millisecond) - require.NoError(t, upr.RefreshUpstreamNetworkMethodScores()) - time.Sleep(100 * time.Millisecond) - // --- Assertions --- // 1) The handleBlockSkip recording fix plumbed gate-skips into the - // tracker: the primary's errorRate on the "*" aggregate (the slot - // the policy reads with evalPerMethod=false) must reflect the - // gate-rejection burst. Without the fix this stays at 0 - // indefinitely and the policy never reacts. - primaryMetrics := mt.GetUpstreamMethodMetrics(primaryUp, "*") + // tracker: the primary's errorRate on the all-finalities "*" aggregate + // (the slot the policy reads) must reflect the gate-rejection burst. + // Without the fix this stays at 0 indefinitely. + primaryMetrics := mt.GetUpstreamMethodMetrics(primaryUp, "*", common.DataFinalityStateAll) + require.NotNil(t, primaryMetrics) t.Logf("primary primary-1 metrics: requests=%d errors=%d errorRate=%.3f", primaryMetrics.RequestsTotal.Load(), primaryMetrics.ErrorsTotal.Load(), primaryMetrics.ErrorRate()) - assert.Greater(t, primaryMetrics.ErrorRate(), 0.5, - "primary errorRate should cross 0.5 from gate-skip recording; "+ + assert.Greater(t, primaryMetrics.ErrorRate(), 0.7, + "primary errorRate should cross 0.7 from gate-skip recording; "+ "if 0 the handleBlockSkip recording fix didn't take effect") - // 2) After the burst + at least one policy eval tick, the primary must - // be cordoned. AcquirePermit returns ErrCodeUpstreamExcludedByPolicy - // for cordoned upstreams. - err = network.selectionPolicyEvaluator.AcquirePermit(&log.Logger, primaryUp, "eth_call") - require.Error(t, err, "primary should be cordoned after the gate-skip burst") - assert.True(t, common.HasErrorCode(err, common.ErrCodeUpstreamExcludedByPolicy), - "expected primary to be excluded by policy, got: %v", err) - - // 3) The fallback must be promoted to active by the same eval tick. - require.NoError(t, network.selectionPolicyEvaluator.AcquirePermit(&log.Logger, fallbackUp, "eth_call"), - "fallback should be promoted after primary errorRate exceeds 0.5") - - // 4) A fresh request for block 1002 must now succeed. With fallbacks in - // upsList and primaries either cordoned or still gate-rejected, the - // request must route to one of the fallbacks (which have block 1002). + // 2) After the burst, the next policy eval tick observes the elevated + // errorRate and excludes the primary (errorRateAbove(0.7) gated on + // samplesAbove(10) — the burst supplies >>10 samples). With every + // primary excluded, preferTag promotes the fallbacks. + policy.TickForTest(network.policyEngine, network.networkId, "*") + order = network.PolicyOrderedUpstreams("eth_call") + require.NotEmpty(t, order) + t.Logf("post-burst policy order: %v", order) + assert.NotContains(t, order, "primary-1", + "primary should be cordoned after the gate-skip burst moved its errorRate past 0.7") + assert.Contains(t, order, "fallback-1", + "fallback should be promoted after every primary is excluded") + + // 3) A fresh request for block 1002 must now succeed via a fallback + // (which have block 1002). Refresh the registry's sorted list so the + // request path sees the post-tick ordering. + require.NoError(t, upr.RefreshUpstreamNetworkMethodScores()) + time.Sleep(50 * time.Millisecond) + finalReq := common.NewNormalizedRequest([]byte(fmt.Sprintf( `{"jsonrpc":"2.0","id":9999,"method":"eth_call","params":[{"to":"0xdead","data":"0x"},"%s"]}`, requestBlock, @@ -360,15 +379,6 @@ func TestFailover_GateSkipsAccumulateErrorRate(t *testing.T) { "final eth_call must be served by a fallback (0x3333 or 0x4444), got %q", result) } -func mustGetUpstream(ups []*upstream.Upstream, id string) *upstream.Upstream { - for _, u := range ups { - if u.Id() == id { - return u - } - } - panic("upstream not found: " + id) -} - // failoverFixtureOpts configures the standard 4-upstream test layout // (2 primaries + 2 fallbacks) used by the escape-hatch sub-tests. type failoverFixtureOpts struct { @@ -396,130 +406,20 @@ func setupFailoverFixture( mockEthCallReturning("rpc3.localhost", "0x3333") mockEthCallReturning("rpc4.localhost", "0x4444") - upstreamConfigs := []*common.UpstreamConfig{ - { - Type: common.UpstreamTypeEvm, Id: "primary-1", - Endpoint: "http://rpc1.localhost", - Evm: &common.EvmUpstreamConfig{ - ChainId: 999, - StatePollerInterval: common.Duration(100 * time.Millisecond), - StatePollerDebounce: common.Duration(20 * time.Millisecond), - }, - }, - { - Type: common.UpstreamTypeEvm, Id: "primary-2", - Endpoint: "http://rpc2.localhost", - Evm: &common.EvmUpstreamConfig{ - ChainId: 999, - StatePollerInterval: common.Duration(100 * time.Millisecond), - StatePollerDebounce: common.Duration(20 * time.Millisecond), - }, - }, - { - Type: common.UpstreamTypeEvm, Id: "fallback-1", - Endpoint: "http://rpc3.localhost", - Group: common.UpstreamGroupFallback, - Evm: &common.EvmUpstreamConfig{ - ChainId: 999, - StatePollerInterval: common.Duration(100 * time.Millisecond), - StatePollerDebounce: common.Duration(20 * time.Millisecond), - }, - }, - { - Type: common.UpstreamTypeEvm, Id: "fallback-2", - Endpoint: "http://rpc4.localhost", - Group: common.UpstreamGroupFallback, - Evm: &common.EvmUpstreamConfig{ - ChainId: 999, - StatePollerInterval: common.Duration(100 * time.Millisecond), - StatePollerDebounce: common.Duration(20 * time.Millisecond), - }, - }, - } - - rlr, _ := upstream.NewRateLimitersRegistry(context.Background(), &common.RateLimiterConfig{}, &log.Logger) - mt := health.NewTracker(&log.Logger, "main", 5*time.Second) - vr := thirdparty.NewVendorsRegistry() - pr, _ := thirdparty.NewProvidersRegistry(&log.Logger, vr, nil, nil) - - sharedStateCfg := &common.SharedStateConfig{ - Connector: &common.ConnectorConfig{ - Driver: common.DriverMemory, - Memory: &common.MemoryConnectorConfig{ - MaxItems: 100_000, - MaxTotalSize: "1GB", - }, - }, - LockMaxWait: common.Duration(200 * time.Millisecond), - UpdateMaxWait: common.Duration(200 * time.Millisecond), - FallbackTimeout: common.Duration(3 * time.Second), - LockTtl: common.Duration(4 * time.Second), - } - require.NoError(t, sharedStateCfg.SetDefaults("test")) - ssr, err := data.NewSharedStateRegistry(ctx, &log.Logger, sharedStateCfg) - require.NoError(t, err) - - upr := upstream.NewUpstreamsRegistry( - ctx, &log.Logger, "main", upstreamConfigs, - ssr, rlr, vr, pr, nil, mt, 1*time.Second, nil, nil, - ) - - evalFn, err := common.CompileFunction(failoverSelectionPolicy) - require.NoError(t, err) - selectionPolicy := &common.SelectionPolicyConfig{ - EvalInterval: common.Duration(150 * time.Millisecond), - EvalFunctionSource: failoverSelectionPolicy, - EvalFunction: evalFn, - EvalPerMethod: false, - } - require.NoError(t, selectionPolicy.SetDefaults()) - - maxRetryable := int64(128) - enforce := true - networkConfig := &common.NetworkConfig{ - Architecture: common.ArchitectureEvm, - Evm: &common.EvmNetworkConfig{ - ChainId: 999, - MaxRetryableBlockDistance: &maxRetryable, - EnforceBlockAvailability: &enforce, - }, - SelectionPolicy: selectionPolicy, - } - if opts.enableFailover { - on := true - networkConfig.Failover = &common.FailoverConfig{OnDefaultsExhausted: &on} - } - - network, err := NewNetwork(ctx, &log.Logger, "main", networkConfig, rlr, upr, mt) - require.NoError(t, err) - - upr.Bootstrap(ctx) - time.Sleep(200 * time.Millisecond) - require.NoError(t, upr.GetInitializer().WaitForTasks(ctx)) - require.NoError(t, upr.PrepareUpstreamsForNetwork(ctx, util.EvmNetworkId(999))) - require.NoError(t, network.Bootstrap(ctx)) + network, upr, mt := buildFailoverNetwork(t, ctx, failoverUpstreamConfigs(), opts.enableFailover) upsList := upr.GetNetworkUpstreams(ctx, util.EvmNetworkId(999)) require.Len(t, upsList, 4) - for _, ups := range upsList { - require.NoError(t, ups.Bootstrap(ctx)) - } - // Let pollers run and policy evaluate at least once. - time.Sleep(500 * time.Millisecond) - - // Prime the registry's sortedUpstreams[networkId]["eth_call"] cache by - // asking for it once. GetSortedUpstreams populates the slot from "*" or - // raw networkUpstreams on first call but does NOT apply filterCordoned — - // only RefreshUpstreamNetworkMethodScores does. So we prime FIRST, then - // refresh, to get the cordon decisions applied to the eth_call slot - // before the test request fires. Without this two-step, the first - // network.Forward sees all 4 upstreams (cordoned-but-not-yet-filtered), - // which masks the cordon-then-escape behaviour under test. - _, err = upr.GetSortedUpstreams(ctx, util.EvmNetworkId(999), "eth_call") - require.NoError(t, err) + // Drive an initial tick so the default policy computes the ordered list. + // With healthy primaries, preferTag cordons the fallbacks out of the + // ordered list — so the request path sees only primaries, exhausts them + // on a gate-skip, and the per-request escape hatch (not the policy) is + // what brings the fallbacks in. This is exactly the path under test. + policy.ResetSlotStateForTest(network.policyEngine, network.networkId, "*") + policy.TickForTest(network.policyEngine, network.networkId, "*") require.NoError(t, upr.RefreshUpstreamNetworkMethodScores()) - time.Sleep(100 * time.Millisecond) + time.Sleep(50 * time.Millisecond) return network, upsList, mt } @@ -615,24 +515,33 @@ func TestFailover_EscapeHatch(t *testing.T) { defer cancel() // Same primary-stuck-fallback-ahead setup as Sub-test A, but with - // failover.onDefaultsExhausted unset. Escape hatch must respect the - // operator's opt-out. + // failover.onDefaultsExhausted unset. The per-request escape hatch must + // respect the operator's opt-out and never fire. + // + // NOTE (post-#888): the request may still be served by the fallback + // tier — upstream's default selection policy natively falls through to + // `tier:fallback` via preferTag('!tier:fallback', {fallback}). That + // native routing is independent of our Failover feature. What this test + // guards is precisely OUR contribution: with Failover disabled the + // `network_fallback_escape_total` counter must stay flat. network, _, _ := setupFailoverFixture(t, ctx, failoverFixtureOpts{ primaryLatest: "0x3e8", // 1000 fallbackLatest: "0x3ea", // 1002 enableFailover: false, // <-- disabled }) + counter := telemetry.MetricNetworkFallbackEscapeTotal.WithLabelValues("main", "evm:999", "eth_call") + before := promUtil.ToFloat64(counter) req := ethCallRequest(1, "0x3ea") req.SetNetwork(network) - resp, err := network.Forward(ctx, req) - require.Error(t, err, - "with failover disabled the escape hatch must NOT fire; request must surface ErrUpstreamsExhausted") - assert.True(t, common.HasErrorCode(err, common.ErrCodeUpstreamsExhausted), - "expected ErrUpstreamsExhausted, got %v", err) + resp, _ := network.Forward(ctx, req) if resp != nil { resp.Release() } + + after := promUtil.ToFloat64(counter) + assert.Equal(t, before, after, + "with failover disabled the per-request escape hatch must NOT fire; counter must be unchanged") }) t.Run("OnlyEscalatesOncePerRequest", func(t *testing.T) { @@ -641,8 +550,13 @@ func TestFailover_EscapeHatch(t *testing.T) { defer cancel() // Primaries stuck at 1000, fallbacks ALSO stuck at 1000. Request - // block 1002 → gate-rejects everywhere. Escape fires once, finds - // fallbacks still can't serve, returns exhausted — does NOT loop. + // block 1002. The gap (2 blocks) is within MaxRetryableBlockDistance, + // so the gate-skip is RETRYABLE and the primary set exhausts → the + // escape hatch fires and appends the fallback tier. Under upstream's + // new model the fallback (within retryable tolerance) then serves the + // request. The invariant this test guards is escalate-AT-MOST-ONCE: the + // escape must fire exactly once and never re-enter escalationLoop a + // second time (which the single MarkEscalatedToFallbacks gate ensures). network, _, _ := setupFailoverFixture(t, ctx, failoverFixtureOpts{ primaryLatest: "0x3e8", // 1000 fallbackLatest: "0x3e8", // also 1000 @@ -654,18 +568,14 @@ func TestFailover_EscapeHatch(t *testing.T) { req := ethCallRequest(1, "0x3ea") req.SetNetwork(network) - resp, err := network.Forward(ctx, req) - require.Error(t, err, - "all upstreams (primary + fallback) can't serve the block; must return ErrUpstreamsExhausted") - assert.True(t, common.HasErrorCode(err, common.ErrCodeUpstreamsExhausted), - "expected ErrUpstreamsExhausted, got %v", err) + resp, _ := network.Forward(ctx, req) if resp != nil { resp.Release() } after := promUtil.ToFloat64(counter) assert.Equal(t, before+1, after, - "escape must fire exactly once per request even when fallbacks also fail; got %v→%v", + "escape must fire EXACTLY once per request (no re-escalation loop); got %v→%v", before, after) }) @@ -676,9 +586,9 @@ func TestFailover_EscapeHatch(t *testing.T) { // Primaries stuck at block 1000, fallbacks far ahead at block 10000. // Request block 9000 (0x2328). The gap from primary (1000) to 9000 is - // 8000 blocks — well beyond the default MaxRetryableBlockDistance of - // 128 — so checkUpstreamBlockAvailability classifies each primary's - // skip as NON-retryable (ErrUpstreamRequestSkipped wrapping + // 8000 blocks — well beyond the default MaxRetryableBlockDistance — + // so checkUpstreamBlockAvailability classifies each primary's skip as + // NON-retryable (ErrUpstreamRequestSkipped wrapping // ErrUpstreamBlockUnavailable). // // This mirrors the live B2 incident pattern: primaries fronting a diff --git a/erpc/networks_test.go b/erpc/networks_test.go index b4b250db4..604c28adc 100644 --- a/erpc/networks_test.go +++ b/erpc/networks_test.go @@ -12269,7 +12269,7 @@ func TestNetwork_HighestFinalizedBlockNumber(t *testing.T) { upstreamsRegistry := upstream.NewUpstreamsRegistry( ctx, &log.Logger, "test", []*common.UpstreamConfig{up}, ssr, rateLimitersRegistry, vr, pr, nil, - metricsTracker, 1*time.Second, nil, nil, + metricsTracker, nil, ) networkConfig := &common.NetworkConfig{ @@ -12277,7 +12277,7 @@ func TestNetwork_HighestFinalizedBlockNumber(t *testing.T) { Evm: &common.EvmNetworkConfig{ChainId: 123}, } network, err := NewNetwork(ctx, &log.Logger, "test", networkConfig, - rateLimitersRegistry, upstreamsRegistry, metricsTracker) + rateLimitersRegistry, upstreamsRegistry, metricsTracker, nil) require.NoError(t, err) upstreamsRegistry.Bootstrap(ctx) @@ -12344,7 +12344,7 @@ func TestNetwork_HighestFinalizedBlockNumber(t *testing.T) { Type: common.UpstreamTypeEvm, Id: "fallback-node", Endpoint: "http://fallback.localhost", - Group: "fallback", + Tags: []string{common.TagTierFallback}, Evm: &common.EvmUpstreamConfig{ChainId: 123}, } @@ -12373,7 +12373,7 @@ func TestNetwork_HighestFinalizedBlockNumber(t *testing.T) { upstreamsRegistry := upstream.NewUpstreamsRegistry( ctx, &log.Logger, "test", []*common.UpstreamConfig{primary, fallback}, ssr, rateLimitersRegistry, vr, pr, nil, - metricsTracker, 1*time.Second, nil, nil, + metricsTracker, nil, ) networkConfig := &common.NetworkConfig{ @@ -12381,7 +12381,7 @@ func TestNetwork_HighestFinalizedBlockNumber(t *testing.T) { Evm: &common.EvmNetworkConfig{ChainId: 123}, } network, err := NewNetwork(ctx, &log.Logger, "test", networkConfig, - rateLimitersRegistry, upstreamsRegistry, metricsTracker) + rateLimitersRegistry, upstreamsRegistry, metricsTracker, nil) require.NoError(t, err) upstreamsRegistry.Bootstrap(ctx) @@ -12527,9 +12527,13 @@ func (m *minimalUpstream) Id() string { return m.id } func (m *minimalUpstream) Config() *common.UpstreamConfig { return m.cfg } func newStubUpstream(id, group string) common.Upstream { + cfg := &common.UpstreamConfig{Id: id} + if group != "" { + cfg.Tags = []string{group} + } return &minimalUpstream{ id: id, - cfg: &common.UpstreamConfig{Id: id, Group: group}, + cfg: cfg, } } @@ -12550,10 +12554,10 @@ func TestTierUpstreamsByGroup(t *testing.T) { // Input order mixes default / fallback — simulates the score-sorted // list the request loop receives before tiering. in := []common.Upstream{ - newStubUpstream("fallback-hi", common.UpstreamGroupFallback), + newStubUpstream("fallback-hi", common.TagTierFallback), newStubUpstream("default-lo", ""), newStubUpstream("default-hi", ""), - newStubUpstream("fallback-lo", common.UpstreamGroupFallback), + newStubUpstream("fallback-lo", common.TagTierFallback), } out := tierUpstreamsByGroup(in) require.Len(t, out, 4) @@ -12567,8 +12571,8 @@ func TestTierUpstreamsByGroup(t *testing.T) { t.Run("only fallbacks still orders them after empty default tier", func(t *testing.T) { in := []common.Upstream{ - newStubUpstream("fb-1", common.UpstreamGroupFallback), - newStubUpstream("fb-2", common.UpstreamGroupFallback), + newStubUpstream("fb-1", common.TagTierFallback), + newStubUpstream("fb-2", common.TagTierFallback), } out := tierUpstreamsByGroup(in) require.Len(t, out, 2) @@ -12578,7 +12582,7 @@ func TestTierUpstreamsByGroup(t *testing.T) { t.Run("unknown group treated as default", func(t *testing.T) { in := []common.Upstream{ - newStubUpstream("fb", common.UpstreamGroupFallback), + newStubUpstream("fb", common.TagTierFallback), newStubUpstream("custom", "experimental"), newStubUpstream("def", ""), } diff --git a/upstream/registry_fallback_escape_test.go b/upstream/registry_fallback_escape_test.go index 3762ac2f8..94725c0ac 100644 --- a/upstream/registry_fallback_escape_test.go +++ b/upstream/registry_fallback_escape_test.go @@ -40,13 +40,13 @@ func setupRegistryWithFallbacks(t *testing.T, ctx context.Context) (*UpstreamsRe { Id: "fallback-1", Type: common.UpstreamTypeEvm, Endpoint: "http://rpc3.localhost", - Group: common.UpstreamGroupFallback, + Tags: []string{common.TagTierFallback}, Evm: &common.EvmUpstreamConfig{ChainId: 123}, }, { Id: "fallback-2", Type: common.UpstreamTypeEvm, Endpoint: "http://rpc4.localhost", - Group: common.UpstreamGroupFallback, + Tags: []string{common.TagTierFallback}, Evm: &common.EvmUpstreamConfig{ChainId: 123}, }, } @@ -65,8 +65,6 @@ func setupRegistryWithFallbacks(t *testing.T, ctx context.Context) (*UpstreamsRe registry := NewUpstreamsRegistry(ctx, logger, "fb-test", upstreamConfigs, ssr, nil, vr, pr, nil, metricsTracker, - 1*time.Second, - &ScoringConfig{ScoreGranularity: "method", SwitchHysteresis: -1, MinSwitchInterval: -1}, nil, ) @@ -191,8 +189,7 @@ func TestGetFallbackEscapeUpstreams_NoFallbacksReturnsEmpty(t *testing.T) { }) require.NoError(t, err) registry := NewUpstreamsRegistry(ctx, logger, "fb-test", upstreamConfigs, - ssr, nil, vr, pr, nil, metricsTracker, 1*time.Second, - &ScoringConfig{ScoreGranularity: "method", SwitchHysteresis: -1, MinSwitchInterval: -1}, + ssr, nil, vr, pr, nil, metricsTracker, nil, ) registry.Bootstrap(ctx) From 3bbc2903e70f0d677a73a676cc2d9df36666b046 Mon Sep 17 00:00:00 2001 From: snowkide Date: Fri, 12 Jun 2026 16:09:38 +0200 Subject: [PATCH 10/40] fix(websocket): self-heal wedged upstream WS connections and surface head-liveness MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Root-caused from the 2026-06-12 zkSync (chain 324) incident: deleting an upstream fullnode pod leaves a half-open TCP connection through the gateway; eRPC's upstream WS client then wedged permanently — believing itself connected, delivering zero newHeads, while still handing out client subscription IDs — until the pod was manually restarted. Four distinct gaps, each fixed: 1. Dead-peer detection (clients/ws_json_rpc_client.go): pings were written but pongs never checked — no read deadline, no pong handler — so ReadMessage blocked forever on a black-holed connection (ping writes 'succeed' into the proxy/kernel buffer, so the write path never errored either). Now: liveness deadline (wsPongWait) armed at connect, extended on every pong and data frame; on expiry the connection is torn down and re-dialed with the existing backoff. Failed ping writes also force-close the connection instead of being debug-logged. Callbacks now fire synchronously so adapters observe disconnect strictly before reconnect. 2. Resubscribe retry (indexer/adapters/wsupstream): subscribe RPCs ride upstream.Forward, i.e. the failsafe circuit breaker, which is typically still open at the instant the WS layer reconnects after an outage. The old single-shot resubscribe failed once with a warning and never retried — no heads until the next disconnect. Now a per-connection-epoch retry loop (1s..30s backoff) runs until newHeads plus all filters are re-established; disconnect cancels the epoch and clears the stale sub ID. 3. Circuit breaker defaults (common/defaults.go): SetDefaults assigned SuccessThresholdCapacity twice; the first (200) shadowed the intended 10, so default-config breakers needed an absurd half-open sample. (Production configs setting it explicitly were unaffected.) 4. Loud failure instead of silent sub IDs: newHeads subscribe now refuses (HTTP 503, ErrNoLiveSubscriptionSource) when no ingress for the network has a live connection + active upstream subscription, after a short bootstrap grace wait — so clients fail over instead of holding a dead subscription. New observability: - erpc_upstream_websocket_connected{project,vendor,network,upstream} - erpc_network_subscription_last_head_timestamp_seconds{network} - healthcheck networks[].subscriptions {live/total ingresses, lastHeadAt, lastHeadAgeSeconds} Regression tests cover: silent (no close handshake) peer death and reconnect at the client layer; reconnect + breaker-open retry + heads resuming at the adapter layer; refusal/grace-wait at the subscription manager; breaker default values. Co-Authored-By: Claude Fable 5 --- clients/ws_json_rpc_client.go | 98 +++++- clients/ws_json_rpc_client_test.go | 296 ++++++++++++++++++ common/defaults.go | 12 +- common/defaults_test.go | 24 ++ common/errors.go | 26 ++ erpc/healthcheck.go | 12 + erpc/subscription_manager.go | 75 +++++ erpc/subscription_manager_health_test.go | 131 ++++++++ indexer/adapters/wsupstream/adapter.go | 173 ++++++++-- .../wsupstream/adapter_reconnect_test.go | 250 +++++++++++++++ indexer/health_test.go | 109 +++++++ indexer/indexer.go | 71 +++++ telemetry/metrics.go | 12 + 13 files changed, 1247 insertions(+), 42 deletions(-) create mode 100644 clients/ws_json_rpc_client_test.go create mode 100644 erpc/subscription_manager_health_test.go create mode 100644 indexer/adapters/wsupstream/adapter_reconnect_test.go create mode 100644 indexer/health_test.go diff --git a/clients/ws_json_rpc_client.go b/clients/ws_json_rpc_client.go index db4ff2905..d9b28f00c 100644 --- a/clients/ws_json_rpc_client.go +++ b/clients/ws_json_rpc_client.go @@ -6,6 +6,7 @@ import ( "errors" "fmt" "io" + "net" "net/http" "net/url" "strconv" @@ -17,6 +18,7 @@ import ( "encoding/json" "github.com/erpc/erpc/common" + "github.com/erpc/erpc/telemetry" "github.com/gorilla/websocket" "github.com/rs/zerolog" "go.opentelemetry.io/otel/attribute" @@ -24,13 +26,28 @@ import ( ) const ( - wsPingInterval = 30 * time.Second wsWriteWait = 10 * time.Second wsReconnectMin = 1 * time.Second wsReconnectMax = 30 * time.Second wsReconnectFactor = 2.0 ) +// Liveness windows. The peer must produce SOME traffic (a pong reply or a +// data frame) within wsPongWait, or the connection is declared dead, torn +// down, and re-dialed. A half-open TCP connection (peer host vanished +// without FIN/RST, or an intermediate proxy black-holing frames) otherwise +// blocks ReadMessage forever while ping writes keep "succeeding" into the +// kernel/proxy buffer — the client then believes it is connected and never +// re-dials. wsPongWait must comfortably exceed wsPingInterval so at least +// two pings fit in the window. +// +// Vars (not consts) so tests can compress time; production code must not +// mutate them. +var ( + wsPingInterval = 30 * time.Second + wsPongWait = 75 * time.Second +) + // WsJsonRpcClient implements ClientInterface for WebSocket-based JSON-RPC upstream connections. type WsJsonRpcClient struct { Url *url.URL @@ -349,13 +366,54 @@ func (c *WsJsonRpcClient) connect() error { return err } + // Arm the liveness deadline: if neither a pong nor a data frame arrives + // within wsPongWait, ReadMessage fails and readLoop re-dials. The pong + // handler runs inside ReadMessage's frame processing, so extending the + // deadline here covers the ping/pong path; readLoop extends it again on + // every data frame. + _ = conn.SetReadDeadline(time.Now().Add(wsPongWait)) + conn.SetPongHandler(func(string) error { + return conn.SetReadDeadline(time.Now().Add(wsPongWait)) + }) + + if c.conn != nil { + // Defensive: never leak a previous connection's FD/goroutine state. + _ = c.conn.Close() + } c.conn = conn c.connected.Store(true) + c.setConnectedMetric(1) c.logger.Info().Str("url", c.Url.String()).Msg("websocket connection established") return nil } +// teardownConn marks the client disconnected and closes the given +// connection, clearing c.conn only if it still points at that same +// connection (a concurrent reconnect may already have replaced it). +func (c *WsJsonRpcClient) teardownConn(old *websocket.Conn) { + c.connMu.Lock() + if c.conn == old { + c.conn = nil + } + c.connMu.Unlock() + if old != nil { + _ = old.Close() + } +} + +// setConnectedMetric publishes the upstream WS connectivity gauge so +// operators can alert on a wedged/disconnected upstream socket instead of +// discovering it from silent client subscriptions. +func (c *WsJsonRpcClient) setConnectedMetric(v float64) { + if c.upstream == nil { + return + } + telemetry.GaugeHandle(telemetry.MetricUpstreamWebsocketConnected, + c.projectId, c.upstream.VendorName(), c.upstream.NetworkLabel(), c.upstream.Id(), + ).Set(v) +} + func (c *WsJsonRpcClient) readLoop() { // Wait until the first connection is established (or the app shuts down) select { @@ -389,12 +447,18 @@ func (c *WsJsonRpcClient) readLoop() { if c.appCtx.Err() != nil { return } + var netErr net.Error if websocket.IsCloseError(err, websocket.CloseNormalClosure, websocket.CloseGoingAway) { c.logger.Info().Msg("websocket connection closed normally") + } else if errors.As(err, &netErr) && netErr.Timeout() { + c.logger.Warn().Err(err).Dur("pongWait", wsPongWait). + Msg("websocket peer silent beyond liveness deadline (no pong/data), tearing down connection and reconnecting") } else { c.logger.Warn().Err(err).Msg("websocket read error, will reconnect") } c.connected.Store(false) + c.setConnectedMetric(0) + c.teardownConn(conn) c.drainPending(common.NewErrEndpointTransportFailure(c.Url, fmt.Errorf("websocket connection lost: %w", err))) c.fireCallbacks(&c.onDisconnectMu, c.onDisconnectCbs) @@ -402,13 +466,25 @@ func (c *WsJsonRpcClient) readLoop() { continue } + // Any inbound frame proves the peer is alive — push the liveness + // deadline forward. + _ = conn.SetReadDeadline(time.Now().Add(wsPongWait)) + c.handleMessage(message) } } -// fireCallbacks snapshots the callback map under rlock and dispatches each -// in its own goroutine. Snapshotting lets callbacks register/deregister +// fireCallbacks snapshots the callback map under rlock and invokes each +// callback synchronously. Snapshotting lets callbacks register/deregister // other callbacks without deadlocking on the map's RWMutex. +// +// Synchronous invocation is load-bearing: readLoop fires disconnect +// callbacks, then reconnects, then fires reconnect callbacks. Dispatching +// them in goroutines (as this used to) let a slow-scheduled disconnect +// callback run AFTER the reconnect callback — for the wsupstream adapter +// that cancels the fresh resubscribe epoch and clears the new subscription, +// silently wedging head delivery. Callbacks must therefore be fast and +// must not block on the WS client's own request path. func (c *WsJsonRpcClient) fireCallbacks(mu *sync.RWMutex, cbs map[string]func()) { mu.RLock() snapshot := make([]func(), 0, len(cbs)) @@ -417,7 +493,7 @@ func (c *WsJsonRpcClient) fireCallbacks(mu *sync.RWMutex, cbs map[string]func()) } mu.RUnlock() for _, cb := range snapshot { - go cb() + cb() } } @@ -568,7 +644,18 @@ func (c *WsJsonRpcClient) pingLoop() { continue } if err := c.writeMessage(websocket.PingMessage, nil); err != nil { - c.logger.Debug().Err(err).Msg("websocket ping failed") + // A failed ping write means the connection is unusable. + // Close it so readLoop's blocked ReadMessage fails and the + // teardown+reconnect path (owned by readLoop) takes over — + // logging alone here previously left the client wedged on a + // connection that could never deliver another frame. + c.logger.Warn().Err(err).Msg("websocket ping write failed, closing connection to force reconnect") + c.connMu.Lock() + conn := c.conn + c.connMu.Unlock() + if conn != nil { + _ = conn.Close() + } } case <-c.appCtx.Done(): return @@ -598,6 +685,7 @@ func normalizeIDKey(id interface{}) string { func (c *WsJsonRpcClient) shutdown() { c.connected.Store(false) + c.setConnectedMetric(0) c.connMu.Lock() conn := c.conn diff --git a/clients/ws_json_rpc_client_test.go b/clients/ws_json_rpc_client_test.go new file mode 100644 index 000000000..e2784c7aa --- /dev/null +++ b/clients/ws_json_rpc_client_test.go @@ -0,0 +1,296 @@ +package clients + +import ( + "context" + "net/http" + "net/http/httptest" + "net/url" + "strings" + "sync" + "sync/atomic" + "testing" + "time" + + "github.com/erpc/erpc/common" + "github.com/gorilla/websocket" + "github.com/rs/zerolog" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// fakeWsServer is a minimal JSON-RPC WebSocket upstream. Each accepted +// connection can be "black-holed": the TCP connection stays open but pings +// are swallowed (no pong reply) and nothing is ever written — exactly what +// an intermediate proxy does when the real upstream pod vanishes without a +// FIN/RST. This is the failure mode from the 2026-06-12 zkSync incident: +// the old client believed such a connection was healthy forever. +type fakeWsServer struct { + t *testing.T + srv *httptest.Server + + mu sync.Mutex + conns []*fakeWsConn + + newConn chan *fakeWsConn +} + +type fakeWsConn struct { + conn *websocket.Conn + writeMu sync.Mutex + // silent simulates a black-holed path: pings are swallowed (no pong) + // and the server never writes, but the TCP connection stays open. + silent atomic.Bool + // subscribeCh receives the request id (raw JSON) of each + // eth_subscribe request the server answers. + subscribeCh chan string +} + +func newFakeWsServer(t *testing.T) *fakeWsServer { + f := &fakeWsServer{t: t, newConn: make(chan *fakeWsConn, 16)} + upgrader := websocket.Upgrader{} + f.srv = httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + conn, err := upgrader.Upgrade(w, r, nil) + if err != nil { + return + } + sc := &fakeWsConn{conn: conn, subscribeCh: make(chan string, 16)} + conn.SetPingHandler(func(appData string) error { + if sc.silent.Load() { + return nil // swallow: black-holed path sends no pong + } + return conn.WriteControl(websocket.PongMessage, []byte(appData), time.Now().Add(time.Second)) + }) + f.mu.Lock() + f.conns = append(f.conns, sc) + f.mu.Unlock() + f.newConn <- sc + go sc.readLoop() + })) + t.Cleanup(f.srv.Close) + return f +} + +func (f *fakeWsServer) wsURL(t *testing.T) *url.URL { + u, err := url.Parse(f.srv.URL) + require.NoError(t, err) + u.Scheme = "ws" + return u +} + +// readLoop answers eth_subscribe with an incrementing subscription id. +// Control frames (pings) are handled inside ReadMessage via the handler +// installed above, so silencing the ping handler is enough to emulate a +// peer that no longer processes anything. +func (sc *fakeWsConn) readLoop() { + subCounter := 0 + for { + _, msg, err := sc.conn.ReadMessage() + if err != nil { + return + } + if sc.silent.Load() { + continue // black-holed: never respond + } + var req struct { + ID interface{} `json:"id"` + Method string `json:"method"` + Params []interface{} `json:"params"` + } + if err := common.SonicCfg.Unmarshal(msg, &req); err != nil { + continue + } + if req.Method == "eth_subscribe" { + subCounter++ + subID := "0xtestsub" + string(rune('0'+subCounter)) + resp, _ := common.SonicCfg.Marshal(map[string]interface{}{ + "jsonrpc": "2.0", + "id": req.ID, + "result": subID, + }) + sc.write(websocket.TextMessage, resp) + sc.subscribeCh <- subID + } + } +} + +func (sc *fakeWsConn) write(messageType int, data []byte) { + sc.writeMu.Lock() + defer sc.writeMu.Unlock() + _ = sc.conn.SetWriteDeadline(time.Now().Add(time.Second)) + _ = sc.conn.WriteMessage(messageType, data) +} + +func (sc *fakeWsConn) sendNewHead(subID string, blockNumberHex string) { + notif, _ := common.SonicCfg.Marshal(map[string]interface{}{ + "jsonrpc": "2.0", + "method": "eth_subscription", + "params": map[string]interface{}{ + "subscription": subID, + "result": map[string]interface{}{ + "number": blockNumberHex, + "hash": "0xhash" + blockNumberHex, + "parentHash": "0xparent" + blockNumberHex, + }, + }, + }) + sc.write(websocket.TextMessage, notif) +} + +// compressWsLiveness shrinks the keepalive windows so dead-peer detection +// happens in milliseconds instead of minutes, restoring them on cleanup. +func compressWsLiveness(t *testing.T) { + origPing, origPong := wsPingInterval, wsPongWait + wsPingInterval = 50 * time.Millisecond + wsPongWait = 150 * time.Millisecond + t.Cleanup(func() { + wsPingInterval, wsPongWait = origPing, origPong + }) +} + +func newTestWsClient(t *testing.T, u *url.URL) *WsJsonRpcClient { + ctx, cancel := context.WithCancel(context.Background()) + t.Cleanup(cancel) + logger := zerolog.New(zerolog.NewTestWriter(t)).Level(zerolog.WarnLevel) + up := common.NewFakeUpstream("test-ws-upstream") + ci, err := NewWsJsonRpcClient(ctx, &logger, "test-project", up, u, nil, nil) + require.NoError(t, err) + c, ok := ci.(*WsJsonRpcClient) + require.True(t, ok) + return c +} + +func subscribeNewHeads(t *testing.T, c *WsJsonRpcClient) string { + ctx, cancel := context.WithTimeout(context.Background(), 2*time.Second) + defer cancel() + nq := common.NewNormalizedRequest([]byte(`{"jsonrpc":"2.0","id":1,"method":"eth_subscribe","params":["newHeads"]}`)) + resp, err := c.SendRequest(ctx, nq) + require.NoError(t, err) + jr, err := resp.JsonRpcResponse() + require.NoError(t, err) + subID := strings.Trim(string(jr.GetResultBytes()), "\"") + require.NotEmpty(t, subID) + return subID +} + +// TestWsClientDetectsSilentPeerAndReconnects is the regression test for the +// 2026-06-12 zkSync incident: the upstream socket dies WITHOUT a close +// handshake (peer keeps TCP open but stops responding — equivalent to a +// proxy black-holing frames after the real upstream pod was deleted). The +// client must declare the connection dead via the ping/pong liveness +// deadline, re-dial, and resume delivering subscription notifications. +func TestWsClientDetectsSilentPeerAndReconnects(t *testing.T) { + compressWsLiveness(t) + server := newFakeWsServer(t) + client := newTestWsClient(t, server.wsURL(t)) + + disconnected := make(chan struct{}, 1) + reconnected := make(chan struct{}, 1) + client.SetOnDisconnect("test", func() { + select { + case disconnected <- struct{}{}: + default: + } + }) + client.SetOnReconnect("test", func() { + select { + case reconnected <- struct{}{}: + default: + } + }) + + // First connection established and subscribed. + var conn1 *fakeWsConn + select { + case conn1 = <-server.newConn: + case <-time.After(2 * time.Second): + t.Fatal("server never saw the initial connection") + } + subID1 := subscribeNewHeads(t, client) + + heads := make(chan []byte, 16) + client.RegisterSubscriptionHandler(subID1, func(params []byte) { + heads <- params + }) + conn1.sendNewHead(subID1, "0x1") + select { + case <-heads: + case <-time.After(2 * time.Second): + t.Fatal("never received the first head") + } + + // Black-hole the connection: TCP stays open, nothing flows back. + conn1.silent.Store(true) + + select { + case <-disconnected: + case <-time.After(3 * time.Second): + t.Fatal("client never detected the silent (half-open) connection — liveness deadline did not fire") + } + + select { + case <-reconnected: + case <-time.After(3 * time.Second): + t.Fatal("client never reconnected after detecting the dead connection") + } + + var conn2 *fakeWsConn + select { + case conn2 = <-server.newConn: + case <-time.After(2 * time.Second): + t.Fatal("server never saw the re-dialed connection") + } + assert.True(t, client.IsConnected()) + + // Re-subscribe on the new connection (in production the wsupstream + // adapter does this from its reconnect hook) and verify notifications + // flow again. + subID2 := subscribeNewHeads(t, client) + client.RegisterSubscriptionHandler(subID2, func(params []byte) { + heads <- params + }) + conn2.sendNewHead(subID2, "0x2") + select { + case <-heads: + case <-time.After(2 * time.Second): + t.Fatal("no heads delivered after reconnection — client did not self-heal") + } +} + +// TestWsClientPingWriteFailureForcesReconnect covers the secondary path: +// when the ping write itself errors (connection reset under our feet), the +// client must tear the connection down and re-dial rather than only logging. +func TestWsClientPingWriteFailureForcesReconnect(t *testing.T) { + compressWsLiveness(t) + server := newFakeWsServer(t) + client := newTestWsClient(t, server.wsURL(t)) + + reconnected := make(chan struct{}, 1) + client.SetOnReconnect("test", func() { + select { + case reconnected <- struct{}{}: + default: + } + }) + + var conn1 *fakeWsConn + select { + case conn1 = <-server.newConn: + case <-time.After(2 * time.Second): + t.Fatal("server never saw the initial connection") + } + + // Hard-kill the server side of the TCP connection (RST-ish): the next + // client ping write (or read) fails. + _ = conn1.conn.UnderlyingConn().Close() + + select { + case <-reconnected: + case <-time.After(3 * time.Second): + t.Fatal("client never reconnected after the connection was killed") + } + select { + case <-server.newConn: + case <-time.After(2 * time.Second): + t.Fatal("server never saw the re-dialed connection") + } +} diff --git a/common/defaults.go b/common/defaults.go index 16a84ab12..19fcecc18 100644 --- a/common/defaults.go +++ b/common/defaults.go @@ -2368,13 +2368,6 @@ func (c *CircuitBreakerPolicyConfig) SetDefaults(defaults *CircuitBreakerPolicyC c.FailureThresholdCapacity = 80 } } - if c.SuccessThresholdCapacity == 0 { - if defaults != nil && defaults.SuccessThresholdCapacity != 0 { - c.SuccessThresholdCapacity = defaults.SuccessThresholdCapacity - } else { - c.SuccessThresholdCapacity = 200 - } - } if c.HalfOpenAfter == 0 { if defaults != nil && defaults.HalfOpenAfter != 0 { c.HalfOpenAfter = defaults.HalfOpenAfter @@ -2393,6 +2386,11 @@ func (c *CircuitBreakerPolicyConfig) SetDefaults(defaults *CircuitBreakerPolicyC if defaults != nil && defaults.SuccessThresholdCapacity != 0 { c.SuccessThresholdCapacity = defaults.SuccessThresholdCapacity } else { + // 8-of-10 successes to close from half-open. A duplicated default + // block used to force this to 200 before this one could run, + // letting a half-open breaker absorb up to 192 probe failures + // before re-opening — meaningless thresholds for low-traffic + // (e.g. WS-only) upstreams. c.SuccessThresholdCapacity = 10 } } diff --git a/common/defaults_test.go b/common/defaults_test.go index 8a02e25a5..d20dd2c96 100644 --- a/common/defaults_test.go +++ b/common/defaults_test.go @@ -1311,3 +1311,27 @@ func TestSetDefaults_SelectionPolicy_EvalScope(t *testing.T) { }) } +// Regression: a duplicated SuccessThresholdCapacity default block used to +// run before SuccessThresholdCount was defaulted, forcing capacity to 200. +// In failsafe-go's half-open state the success-thresholding capacity also +// bounds how many failures (capacity - successThreshold) are absorbed +// before the breaker re-opens, so 200 made half-open recovery statistics +// meaningless for low-traffic (e.g. WS-only) upstreams. +func TestSetDefaults_CircuitBreakerSuccessThresholdCapacity(t *testing.T) { + t.Run("DefaultsTo8of10", func(t *testing.T) { + cfg := &CircuitBreakerPolicyConfig{} + assert.NoError(t, cfg.SetDefaults(nil)) + assert.EqualValues(t, 8, cfg.SuccessThresholdCount) + assert.EqualValues(t, 10, cfg.SuccessThresholdCapacity) + }) + + t.Run("ExplicitValuesPreserved", func(t *testing.T) { + cfg := &CircuitBreakerPolicyConfig{ + SuccessThresholdCount: 3, + SuccessThresholdCapacity: 10, + } + assert.NoError(t, cfg.SetDefaults(nil)) + assert.EqualValues(t, 3, cfg.SuccessThresholdCount) + assert.EqualValues(t, 10, cfg.SuccessThresholdCapacity) + }) +} diff --git a/common/errors.go b/common/errors.go index 3073feb31..a8c4bcf34 100644 --- a/common/errors.go +++ b/common/errors.go @@ -2824,6 +2824,32 @@ func (e *ErrNoWsUpstreamAvailable) ErrorStatusCode() int { return http.StatusBadRequest } +type ErrNoLiveSubscriptionSource struct{ BaseError } + +const ErrCodeNoLiveSubscriptionSource ErrorCode = "ErrNoLiveSubscriptionSource" + +// NewErrNoLiveSubscriptionSource is returned when WS upstreams are +// configured for the network but none currently has a live connection with +// an active newHeads subscription. Refusing the subscription (HTTP 503 / +// retryable) lets clients fail over to another node instead of holding a +// subscription ID that will never deliver. +var NewErrNoLiveSubscriptionSource = func(networkId string, totalIngresses int) error { + return &ErrNoLiveSubscriptionSource{ + BaseError{ + Code: ErrCodeNoLiveSubscriptionSource, + Message: fmt.Sprintf("no upstream is currently able to deliver subscription events for network %s; refusing subscription so the client can fail over", networkId), + Details: map[string]interface{}{ + "networkId": networkId, + "totalIngresses": totalIngresses, + }, + }, + } +} + +func (e *ErrNoLiveSubscriptionSource) ErrorStatusCode() int { + return http.StatusServiceUnavailable +} + type ErrSubscriptionLimitExceeded struct{ BaseError } const ErrCodeSubscriptionLimitExceeded ErrorCode = "ErrSubscriptionLimitExceeded" diff --git a/erpc/healthcheck.go b/erpc/healthcheck.go index 85e4953b5..bbf03f4bb 100644 --- a/erpc/healthcheck.go +++ b/erpc/healthcheck.go @@ -48,6 +48,11 @@ type NetworkHealthData struct { Status string `json:"status"` Message string `json:"message,omitempty"` Upstreams map[string]*UpstreamHealthData `json:"upstreams"` + + // Subscriptions reports WS head-delivery liveness for this network. + // Present only when at least one client has subscribed on the network + // since the process started. + Subscriptions *NetworkSubscriptionHealth `json:"subscriptions,omitempty"` } type UpstreamHealthData struct { @@ -245,6 +250,13 @@ func (s *HttpServer) handleHealthCheck( ms := float64(bt.Milliseconds()) networkHealth.BlockTimeMs = &ms } + // Subscription head-liveness: nil unless a client has + // subscribed on this network at least once. Lets load + // balancers and operators see "this pod delivers no heads + // for network X" without an active client subscription. + if s.subscriptionManager != nil { + networkHealth.Subscriptions = s.subscriptionManager.SubscriptionHealth(networkId) + } projectHealth.Networks[networkId] = networkHealth } diff --git a/erpc/subscription_manager.go b/erpc/subscription_manager.go index 54b78f152..d5a5c0e1b 100644 --- a/erpc/subscription_manager.go +++ b/erpc/subscription_manager.go @@ -38,8 +38,19 @@ const ( // unsubscribeTimeout is the deadline for best-effort upstream // unsubscribe calls during connection cleanup. unsubscribeTimeout = 5 * time.Second + + // liveHeadSourcePollEvery is how often waitForLiveHeadSource re-checks + // ingress health while waiting out the bootstrap race. + liveHeadSourcePollEvery = 100 * time.Millisecond ) +// liveHeadSourceWaitMax bounds how long a newHeads subscribe waits for at +// least one ingress to come alive before refusing the subscription. Long +// enough to cover the initial bootstrap (adapter connect + eth_subscribe +// round-trip), short enough that a client talking to a head-less pod fails +// over quickly. Var so tests can compress time. +var liveHeadSourceWaitMax = 3 * time.Second + // SubscriptionManager is the client-facing egress layer. It owns // per-connection *wsclient.Adapter instances, lazily registers networks + // ingresses with the indexer the first time a client subscribes on a @@ -167,6 +178,19 @@ func (sm *SubscriptionManager) Subscribe( return nil, fmt.Errorf("failed to generate subscription ID: %w", err) } + // newHeads is fan-out only — no per-filter EnsureFilter ever touches an + // upstream for it, so without this check a pod whose WS upstreams are + // all down (or resubscribing) would happily return a subscription ID + // that never delivers a single head. Refuse instead so the client can + // retry/fail over. Filter subs get equivalent protection from + // EnsureFilter, which errors when every ingress fails. + if subType == SubTypeNewHeads { + if err := sm.waitForLiveHeadSource(ctx, networkId); err != nil { + sm.recordFailureMetrics(project, nw, method, reqFinality, start, nq, err) + return nil, err + } + } + kind, filterHash, err := sm.resolveSubscription(ctx, networkId, subType, jrReq.Params) if err != nil { sm.recordFailureMetrics(project, nw, method, reqFinality, start, nq, err) @@ -300,6 +324,57 @@ func (sm *SubscriptionManager) CleanupConnection(wsc *WsConnection, _ *PreparedP lg.Debug().Msg("cleaned up all subscriptions for connection") } +// NetworkSubscriptionHealth summarizes a network's head-delivery liveness +// for the health endpoint. Nil/absent when the network has never been +// bootstrapped (no client ever subscribed on it). +type NetworkSubscriptionHealth struct { + LiveIngresses int `json:"liveIngresses"` + TotalIngresses int `json:"totalIngresses"` + LastHeadNumber int64 `json:"lastHeadNumber,omitempty"` + LastHeadAt string `json:"lastHeadAt,omitempty"` + LastHeadAgeSec int64 `json:"lastHeadAgeSeconds,omitempty"` +} + +// SubscriptionHealth reports the network's subscription liveness, or nil +// when the network was never bootstrapped for subscriptions. +func (sm *SubscriptionManager) SubscriptionHealth(networkId string) *NetworkSubscriptionHealth { + if _, ok := sm.networks.Load(networkId); !ok { + return nil + } + live, total := sm.idx.IngressHealth(networkId) + out := &NetworkSubscriptionHealth{LiveIngresses: live, TotalIngresses: total} + if block, at, ok := sm.idx.LastHead(networkId); ok { + out.LastHeadNumber = block.Number + out.LastHeadAt = at.UTC().Format(time.RFC3339) + out.LastHeadAgeSec = int64(time.Since(at).Seconds()) + } + return out +} + +// waitForLiveHeadSource returns nil as soon as at least one of the +// network's ingresses reports it can deliver heads. The bounded wait +// covers the bootstrap race where adapters' initial eth_subscribe calls +// are still in flight; after that it refuses with a retryable error. +func (sm *SubscriptionManager) waitForLiveHeadSource(ctx context.Context, networkId string) error { + deadline := time.Now().Add(liveHeadSourceWaitMax) + for { + live, total := sm.idx.IngressHealth(networkId) + if live > 0 { + return nil + } + if ctx.Err() != nil || time.Now().After(deadline) { + sm.logger.Warn().Str("networkId", networkId).Int("totalIngresses", total). + Msg("refusing newHeads subscription: no live head source on this instance") + return common.NewErrNoLiveSubscriptionSource(networkId, total) + } + select { + case <-ctx.Done(): + return common.NewErrNoLiveSubscriptionSource(networkId, total) + case <-time.After(liveHeadSourcePollEvery): + } + } +} + // --- internals -------------------------------------------------------- // buildWsAdapterOptions resolves network-level toggles that the wsupstream diff --git a/erpc/subscription_manager_health_test.go b/erpc/subscription_manager_health_test.go new file mode 100644 index 000000000..2a0580591 --- /dev/null +++ b/erpc/subscription_manager_health_test.go @@ -0,0 +1,131 @@ +package erpc + +import ( + "context" + "testing" + "time" + + "github.com/erpc/erpc/common" + "github.com/erpc/erpc/indexer" + "github.com/rs/zerolog" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +type stubNetworkHandle struct{ id string } + +func (h stubNetworkHandle) Id() string { return h.id } +func (h stubNetworkHandle) FinalityDepth() int64 { return 0 } +func (h stubNetworkHandle) SuggestLatestBlock(string, int64) {} + +// stubIngress implements indexer.EventIngress plus indexer.HealthReporter +// with a flippable health flag. +type stubIngress struct { + name string + healthy bool +} + +func (i *stubIngress) Name() string { return i.name } +func (i *stubIngress) Start(context.Context, indexer.NetworkHandle, indexer.Sink) error { + return nil +} +func (i *stubIngress) EnsureFilter(context.Context, string, string, []interface{}) error { return nil } +func (i *stubIngress) RemoveFilter(context.Context, string, string) error { return nil } +func (i *stubIngress) Stop(context.Context) error { return nil } +func (i *stubIngress) Healthy() bool { return i.healthy } + +func newTestSubscriptionManager(t *testing.T) (*SubscriptionManager, *indexer.Indexer) { + t.Helper() + logger := zerolog.New(zerolog.NewTestWriter(t)).Level(zerolog.ErrorLevel) + idx := indexer.New(&logger, indexer.Options{}) + return NewSubscriptionManager(&logger, idx), idx +} + +// TestWaitForLiveHeadSource pins the incident-driven contract: a pod with +// zero live head sources must refuse newHeads subscriptions (retryable +// error) instead of handing out a subscription ID that never delivers — +// the silent failure mode that hid the 2026-06-12 zkSync outage for hours. +func TestWaitForLiveHeadSource(t *testing.T) { + origWait := liveHeadSourceWaitMax + liveHeadSourceWaitMax = 300 * time.Millisecond + t.Cleanup(func() { liveHeadSourceWaitMax = origWait }) + + const networkID = "evm:324" + + t.Run("refuses when no ingress is live", func(t *testing.T) { + sm, idx := newTestSubscriptionManager(t) + idx.RegisterNetwork(stubNetworkHandle{id: networkID}) + require.NoError(t, idx.AddIngress(context.Background(), networkID, &stubIngress{name: "ws:a", healthy: false})) + + err := sm.waitForLiveHeadSource(context.Background(), networkID) + require.Error(t, err) + assert.True(t, common.HasErrorCode(err, common.ErrCodeNoLiveSubscriptionSource), + "expected ErrNoLiveSubscriptionSource, got: %v", err) + }) + + t.Run("passes immediately when an ingress is live", func(t *testing.T) { + sm, idx := newTestSubscriptionManager(t) + idx.RegisterNetwork(stubNetworkHandle{id: networkID}) + require.NoError(t, idx.AddIngress(context.Background(), networkID, &stubIngress{name: "ws:a", healthy: true})) + + start := time.Now() + require.NoError(t, sm.waitForLiveHeadSource(context.Background(), networkID)) + assert.Less(t, time.Since(start), liveHeadSourceWaitMax/2, + "a live source must not incur the bootstrap grace wait") + }) + + t.Run("passes when an ingress becomes live during the grace wait", func(t *testing.T) { + sm, idx := newTestSubscriptionManager(t) + idx.RegisterNetwork(stubNetworkHandle{id: networkID}) + ing := &stubIngress{name: "ws:a", healthy: false} + require.NoError(t, idx.AddIngress(context.Background(), networkID, ing)) + + go func() { + time.Sleep(120 * time.Millisecond) + ing.healthy = true + }() + require.NoError(t, sm.waitForLiveHeadSource(context.Background(), networkID)) + }) + + t.Run("honours caller context cancellation", func(t *testing.T) { + sm, idx := newTestSubscriptionManager(t) + idx.RegisterNetwork(stubNetworkHandle{id: networkID}) + require.NoError(t, idx.AddIngress(context.Background(), networkID, &stubIngress{name: "ws:a", healthy: false})) + + ctx, cancel := context.WithTimeout(context.Background(), 50*time.Millisecond) + defer cancel() + err := sm.waitForLiveHeadSource(ctx, networkID) + require.Error(t, err) + assert.True(t, common.HasErrorCode(err, common.ErrCodeNoLiveSubscriptionSource)) + }) +} + +func TestSubscriptionHealth(t *testing.T) { + const networkID = "evm:324" + sm, idx := newTestSubscriptionManager(t) + + assert.Nil(t, sm.SubscriptionHealth(networkID), "nil before the network is bootstrapped") + + idx.RegisterNetwork(stubNetworkHandle{id: networkID}) + require.NoError(t, idx.AddIngress(context.Background(), networkID, &stubIngress{name: "ws:a", healthy: true})) + require.NoError(t, idx.AddIngress(context.Background(), networkID, &stubIngress{name: "ws:b", healthy: false})) + sm.networks.Store(networkID, struct{}{}) + + h := sm.SubscriptionHealth(networkID) + require.NotNil(t, h) + assert.Equal(t, 1, h.LiveIngresses) + assert.Equal(t, 2, h.TotalIngresses) + assert.Empty(t, h.LastHeadAt, "no head delivered yet") + + idx.Ingest(indexer.StreamEvent{ + Kind: indexer.KindNewHead, + NetworkId: networkID, + SourceId: "ws:a", + Block: indexer.BlockRef{Number: 99, Hash: "0xaa", ParentHash: "0x98"}, + }) + + h = sm.SubscriptionHealth(networkID) + require.NotNil(t, h) + assert.Equal(t, int64(99), h.LastHeadNumber) + assert.NotEmpty(t, h.LastHeadAt) +} diff --git a/indexer/adapters/wsupstream/adapter.go b/indexer/adapters/wsupstream/adapter.go index 4bc3b19d3..88f6e5178 100644 --- a/indexer/adapters/wsupstream/adapter.go +++ b/indexer/adapters/wsupstream/adapter.go @@ -26,6 +26,17 @@ import ( const ( methodEthSubscribe = "eth_subscribe" methodEthUnsubscribe = "eth_unsubscribe" + + // resubAttemptTimeout bounds each individual subscribe RPC inside the + // resubscribe retry loop. + resubAttemptTimeout = 15 * time.Second +) + +// Resubscribe retry backoff bounds. Vars (not consts) so tests can compress +// time; production code must not mutate them. +var ( + resubRetryMin = 1 * time.Second + resubRetryMax = 30 * time.Second ) // internalReqIDOffset keeps our JSON-RPC IDs out of the range clients @@ -45,6 +56,17 @@ type Adapter struct { wsClient *clients.WsJsonRpcClient logger *zerolog.Logger + // forward routes subscribe/unsubscribe RPCs. Defaults to the upstream's + // failsafe-wrapped Forward (so retry/timeout/circuit-breaker policies + // apply); overridable in tests. + forward func(ctx context.Context, nq *common.NormalizedRequest, bypassMethodExclusion bool) (*common.NormalizedResponse, error) + + // resubMu guards resubCancel: at most one resubscribe retry loop runs + // per connection epoch (initial connect or reconnect); a disconnect or + // Stop cancels it. + resubMu sync.Mutex + resubCancel context.CancelFunc + // stripSubscribeFromBlockZero controls whether fromBlock: "0x0" is // removed from eth_subscribe logs filters before forwarding upstream. // See common.EvmNetworkConfig.StripSubscribeFromBlockZero for details. @@ -101,6 +123,9 @@ func New(up *upstream.Upstream, networkID string, logger *zerolog.Logger, opts * wsClient: wsClient, logger: &lg, filters: make(map[string]*filterSub), + forward: func(ctx context.Context, nq *common.NormalizedRequest, bypassMethodExclusion bool) (*common.NormalizedResponse, error) { + return up.Forward(ctx, nq, bypassMethodExclusion, false) + }, } if opts != nil { a.stripSubscribeFromBlockZero = opts.StripSubscribeFromBlockZero @@ -125,29 +150,67 @@ func (a *Adapter) Start(_ context.Context, nw indexer.NetworkHandle, sink indexe cbID := a.Name() a.wsClient.SetOnReconnect(cbID, func() { a.logger.Info().Msg("WS reconnected — re-subscribing to all active subs") - a.resubscribeAll(context.Background()) + a.startResubscribe() }) a.wsClient.SetOnDisconnect(cbID, func() { a.logger.Info().Msg("WS disconnected — active subs will re-subscribe on reconnect") + a.stopResubscribe() + // The upstream-assigned subscription IDs died with the connection; + // forget the newHeads sub so Healthy() reports honestly until the + // reconnect-epoch resubscribe succeeds. + a.subsMu.Lock() + if a.newHeadsSubID != "" { + a.wsClient.UnregisterSubscriptionHandler(a.newHeadsSubID) + a.newHeadsSubID = "" + } + a.subsMu.Unlock() }) - go a.initialSubscribe() + if a.wsClient.IsConnected() { + a.startResubscribe() + } else { + a.logger.Debug().Msg("WS not yet connected at Start; newHeads will subscribe on first connect") + } return nil } -// initialSubscribe attempts the first newHeads subscribe. Retries on its -// own are unnecessary — if the subscribe fails, the reconnect callback -// picks it up the next time the WS client reconnects. If the WS is down -// at Start time, we wait briefly; no-op after that since the reconnect -// hook will fire Start's equivalent. -func (a *Adapter) initialSubscribe() { - ctx, cancel := context.WithTimeout(context.Background(), 15*time.Second) - defer cancel() +// Healthy reports whether this ingress currently has a live upstream WS +// connection AND an active newHeads subscription — i.e. it can actually +// deliver heads right now. Consulted by the client-facing layer before +// handing out newHeads subscription IDs, so clients are refused (and can +// fail over) instead of receiving a subscription that will never fire. +func (a *Adapter) Healthy() bool { if !a.wsClient.IsConnected() { - a.logger.Debug().Msg("WS not yet connected at Start; newHeads will subscribe on first connect") - return + return false } - a.subscribeNewHeads(ctx) + a.subsMu.Lock() + defer a.subsMu.Unlock() + return a.newHeadsSubID != "" +} + +// startResubscribe launches the retry loop for the current connection +// epoch, cancelling any loop left over from a previous epoch. +func (a *Adapter) startResubscribe() { + a.resubMu.Lock() + if a.resubCancel != nil { + a.resubCancel() + } + ctx, cancel := context.WithCancel(context.Background()) + a.resubCancel = cancel + a.resubMu.Unlock() + go a.resubscribeWithRetry(ctx) +} + +// stopResubscribe cancels the in-flight retry loop, if any. Called on +// disconnect (the loop's subscribes can't succeed anyway; the next +// reconnect starts a fresh epoch) and on Stop. +func (a *Adapter) stopResubscribe() { + a.resubMu.Lock() + if a.resubCancel != nil { + a.resubCancel() + a.resubCancel = nil + } + a.resubMu.Unlock() } // EnsureFilter (re)subscribes a filter on this upstream. Safe to call more @@ -199,6 +262,7 @@ func (a *Adapter) Stop(ctx context.Context) error { cbID := a.Name() a.wsClient.RemoveOnReconnect(cbID) a.wsClient.RemoveOnDisconnect(cbID) + a.stopResubscribe() a.subsMu.Lock() subs := a.filters @@ -227,31 +291,79 @@ func filterKey(subType, paramsHash string) string { return subType + ":" + paramsHash } -// resubscribeAll re-subscribes newHeads + every tracked filter after a -// WS reconnect. Runs in a dedicated goroutine (reconnect callback) so it -// must be self-sufficient w.r.t. context. -func (a *Adapter) resubscribeAll(ctx context.Context) { - a.subscribeNewHeads(ctx) +// resubscribeWithRetry (re)establishes the newHeads subscription plus every +// tracked filter, retrying with backoff until everything is subscribed or +// the epoch is cancelled (disconnect / Stop). +// +// Retrying matters: subscribe RPCs ride upstream.Forward, so failsafe +// policies — including the circuit breaker — apply. Right after an upstream +// outage the breaker is typically still open at the moment the WS layer +// reconnects; a previous single-shot resubscribe failed once with a warning +// and never tried again, leaving the pod permanently head-less while still +// accepting client subscriptions (zkSync chain-324 incident, 2026-06-12). +func (a *Adapter) resubscribeWithRetry(ctx context.Context) { + needHeads := true + pending := make(map[string]struct{}) a.subsMu.Lock() - subs := make([]*filterSub, 0, len(a.filters)) - for _, s := range a.filters { - subs = append(subs, s) + for k := range a.filters { + pending[k] = struct{}{} } a.subsMu.Unlock() - for _, sub := range subs { - if err := a.subscribeFilter(ctx, sub); err != nil { - a.logger.Warn().Err(err).Str("subType", sub.subType).Str("paramsHash", sub.paramsHash). - Msg("failed to re-subscribe filter after reconnect") + backoff := resubRetryMin + for { + if ctx.Err() != nil { + return + } + if needHeads { + attemptCtx, cancel := context.WithTimeout(ctx, resubAttemptTimeout) + err := a.subscribeNewHeads(attemptCtx) + cancel() + if err != nil { + a.logger.Warn().Err(err).Msg("failed to subscribe newHeads, will retry") + } else { + needHeads = false + } + } + for key := range pending { + a.subsMu.Lock() + sub, ok := a.filters[key] + a.subsMu.Unlock() + if !ok { + // Filter was removed while we were retrying. + delete(pending, key) + continue + } + attemptCtx, cancel := context.WithTimeout(ctx, resubAttemptTimeout) + err := a.subscribeFilter(attemptCtx, sub) + cancel() + if err != nil { + a.logger.Warn().Err(err).Str("subType", sub.subType).Str("paramsHash", sub.paramsHash). + Msg("failed to re-subscribe filter, will retry") + } else { + delete(pending, key) + } + } + if !needHeads && len(pending) == 0 { + a.logger.Info().Msg("all upstream subscriptions (re)established") + return + } + select { + case <-ctx.Done(): + return + case <-time.After(backoff): + } + backoff *= 2 + if backoff > resubRetryMax { + backoff = resubRetryMax } } } -func (a *Adapter) subscribeNewHeads(ctx context.Context) { +func (a *Adapter) subscribeNewHeads(ctx context.Context) error { subID, err := a.sendSubscribe(ctx, []interface{}{indexer.SubTypeNewHeads}) if err != nil { - a.logger.Warn().Err(err).Msg("failed to subscribe newHeads") - return + return err } a.subsMu.Lock() if a.newHeadsSubID != "" { @@ -264,6 +376,7 @@ func (a *Adapter) subscribeNewHeads(ctx context.Context) { a.handleNewHeads(params) }) a.logger.Info().Str("upstreamSubId", subID).Msg("subscribed to newHeads") + return nil } func (a *Adapter) subscribeFilter(ctx context.Context, sub *filterSub) error { @@ -378,7 +491,7 @@ func (a *Adapter) sendSubscribe(ctx context.Context, params []interface{}) (stri return "", err } nq := common.NewNormalizedRequest(body) - resp, err := a.upstream.Forward(ctx, nq, false, false) + resp, err := a.forward(ctx, nq, false) if err != nil { return "", err } @@ -404,7 +517,7 @@ func (a *Adapter) sendUnsubscribe(ctx context.Context, subID string) { return } nq := common.NewNormalizedRequest(body) - _, _ = a.upstream.Forward(ctx, nq, false, false) + _, _ = a.forward(ctx, nq, false) } // buildJSONRPCBody marshals a JSON-RPC request with a unique internal diff --git a/indexer/adapters/wsupstream/adapter_reconnect_test.go b/indexer/adapters/wsupstream/adapter_reconnect_test.go new file mode 100644 index 000000000..f07475b0b --- /dev/null +++ b/indexer/adapters/wsupstream/adapter_reconnect_test.go @@ -0,0 +1,250 @@ +package wsupstream + +import ( + "context" + "errors" + "fmt" + "io" + "net/http" + "net/http/httptest" + "net/url" + "strings" + "sync" + "sync/atomic" + "testing" + "time" + + "github.com/erpc/erpc/clients" + "github.com/erpc/erpc/common" + "github.com/erpc/erpc/indexer" + "github.com/gorilla/websocket" + "github.com/rs/zerolog" + "github.com/stretchr/testify/require" +) + +// --- test doubles ------------------------------------------------------- + +type fakeNetworkHandle struct{} + +func (fakeNetworkHandle) Id() string { return "evm:324" } +func (fakeNetworkHandle) FinalityDepth() int64 { return 0 } +func (fakeNetworkHandle) SuggestLatestBlock(string, int64) {} + +type fakeSink struct { + events chan indexer.StreamEvent +} + +func (s *fakeSink) Ingest(ev indexer.StreamEvent) { s.events <- ev } + +// notifyServer is a WS server that only accepts connections and pushes +// subscription notification frames; subscribe RPCs never reach it because +// the adapter's forward func is stubbed (the real one rides +// upstream.Forward, i.e. the failsafe/circuit-breaker pipeline). +type notifyServer struct { + srv *httptest.Server + newConn chan *notifyConn +} + +type notifyConn struct { + conn *websocket.Conn + writeMu sync.Mutex +} + +func newNotifyServer(t *testing.T) *notifyServer { + n := ¬ifyServer{newConn: make(chan *notifyConn, 16)} + upgrader := websocket.Upgrader{} + n.srv = httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + conn, err := upgrader.Upgrade(w, r, nil) + if err != nil { + return + } + nc := ¬ifyConn{conn: conn} + n.newConn <- nc + // Keep reading so pings get ponged and client writes are drained. + go func() { + for { + if _, _, err := conn.ReadMessage(); err != nil { + return + } + } + }() + })) + t.Cleanup(n.srv.Close) + return n +} + +func (n *notifyServer) wsURL(t *testing.T) *url.URL { + u, err := url.Parse(n.srv.URL) + require.NoError(t, err) + u.Scheme = "ws" + return u +} + +func (nc *notifyConn) sendNewHead(subID string, num int64) { + notif, _ := common.SonicCfg.Marshal(map[string]interface{}{ + "jsonrpc": "2.0", + "method": "eth_subscription", + "params": map[string]interface{}{ + "subscription": subID, + "result": map[string]interface{}{ + "number": fmt.Sprintf("0x%x", num), + "hash": fmt.Sprintf("0xhash%x", num), + "parentHash": fmt.Sprintf("0xhash%x", num-1), + }, + }, + }) + nc.writeMu.Lock() + defer nc.writeMu.Unlock() + _ = nc.conn.SetWriteDeadline(time.Now().Add(time.Second)) + _ = nc.conn.WriteMessage(websocket.TextMessage, notif) +} + +func compressResubRetry(t *testing.T) { + origMin, origMax := resubRetryMin, resubRetryMax + resubRetryMin = 20 * time.Millisecond + resubRetryMax = 100 * time.Millisecond + t.Cleanup(func() { resubRetryMin, resubRetryMax = origMin, origMax }) +} + +func syntheticSubscribeResponse(subID string) *common.NormalizedResponse { + body := fmt.Sprintf(`{"jsonrpc":"2.0","id":900000001,"result":"%s"}`, subID) + return common.NewNormalizedResponse().WithBody(io.NopCloser(strings.NewReader(body))) +} + +// --- the regression test ------------------------------------------------- + +// TestAdapterResubscribesWithRetryAfterReconnect reproduces the wedge from +// the 2026-06-12 zkSync incident at the adapter layer: +// +// 1. subscribe RPCs ride the upstream's failsafe pipeline, whose circuit +// breaker is typically still OPEN at the instant the WS layer +// reconnects after an upstream outage; +// 2. the old adapter attempted the resubscribe exactly once per reconnect, +// so an open breaker meant no newHeads subscription (and therefore no +// heads for any client) until the *next* disconnect, i.e. potentially +// forever. +// +// The adapter must keep retrying until the breaker lets a subscribe +// through, and report Healthy()==false until it has a live subscription. +func TestAdapterResubscribesWithRetryAfterReconnect(t *testing.T) { + compressResubRetry(t) + + server := newNotifyServer(t) + ctx, cancel := context.WithCancel(context.Background()) + t.Cleanup(cancel) + + logger := zerolog.New(zerolog.NewTestWriter(t)).Level(zerolog.WarnLevel) + up := common.NewFakeUpstream("test-ws-upstream") + ci, err := clients.NewWsJsonRpcClient(ctx, &logger, "test-project", up, server.wsURL(t), nil, nil) + require.NoError(t, err) + wsc, ok := ci.(*clients.WsJsonRpcClient) + require.True(t, ok) + + var conn1 *notifyConn + select { + case conn1 = <-server.newConn: + case <-time.After(2 * time.Second): + t.Fatal("server never saw the initial connection") + } + _ = conn1 + + // forward simulates the failsafe pipeline: the first failuresPerEpoch + // calls of each epoch fail as an open circuit breaker would, then the + // subscribe succeeds with an epoch-scoped subscription id. + var ( + forwardCalls atomic.Int64 + epoch atomic.Int64 + failuresLeft atomic.Int64 + ) + const failuresPerEpoch = 2 + epoch.Store(1) + failuresLeft.Store(failuresPerEpoch) + + sink := &fakeSink{events: make(chan indexer.StreamEvent, 16)} + a := &Adapter{ + upstreamID: "test-ws-upstream", + networkID: "evm:324", + wsClient: wsc, + logger: &logger, + filters: make(map[string]*filterSub), + forward: func(ctx context.Context, nq *common.NormalizedRequest, _ bool) (*common.NormalizedResponse, error) { + forwardCalls.Add(1) + if failuresLeft.Add(-1) >= 0 { + return nil, errors.New("circuit breaker is open on upstream-level") + } + return syntheticSubscribeResponse(fmt.Sprintf("0xsub-epoch%d", epoch.Load())), nil + }, + } + + require.NoError(t, a.Start(ctx, fakeNetworkHandle{}, sink)) + + // Epoch 1: the initial subscribe must survive the two breaker + // failures and eventually succeed. + require.Eventually(t, a.Healthy, 3*time.Second, 10*time.Millisecond, + "adapter never became healthy despite the breaker closing after %d failures", failuresPerEpoch) + require.GreaterOrEqual(t, forwardCalls.Load(), int64(failuresPerEpoch+1), + "expected the subscribe to be retried through breaker failures") + + // Heads flow on the first connection. + conn1.sendNewHead("0xsub-epoch1", 100) + select { + case ev := <-sink.events: + require.Equal(t, indexer.KindNewHead, ev.Kind) + require.Equal(t, int64(100), ev.Block.Number) + case <-time.After(2 * time.Second): + t.Fatal("no head delivered to the sink on the initial connection") + } + + // Ungraceful upstream death: kill the TCP connection with no close + // handshake. The client reconnects; the adapter must clear its stale + // subscription (Healthy()==false), then retry the resubscribe through + // a fresh round of breaker failures. + epoch.Store(2) + failuresLeft.Store(failuresPerEpoch) + _ = conn1.conn.UnderlyingConn().Close() + + var conn2 *notifyConn + select { + case conn2 = <-server.newConn: + case <-time.After(3 * time.Second): + t.Fatal("client never re-dialed after the upstream connection was killed") + } + + require.Eventually(t, a.Healthy, 3*time.Second, 10*time.Millisecond, + "adapter never re-established the newHeads subscription after reconnect") + + // Heads must flow again on the new connection with the new sub id — + // this is the incident's acceptance criterion at this layer. + conn2.sendNewHead("0xsub-epoch2", 101) + select { + case ev := <-sink.events: + require.Equal(t, indexer.KindNewHead, ev.Kind) + require.Equal(t, int64(101), ev.Block.Number) + case <-time.After(2 * time.Second): + t.Fatal("no head delivered after upstream recovery — adapter did not self-heal") + } +} + +// TestAdapterHealthyReportsFalseWhenDisconnected pins the Healthy() +// contract the subscription-refusal path depends on. +func TestAdapterHealthyReportsFalseWhenDisconnected(t *testing.T) { + a := &Adapter{filters: make(map[string]*filterSub)} + // No wsClient at all would panic — Healthy is only called on adapters + // built by New, which always have one. Use a disconnected client. + server := newNotifyServer(t) + ctx, cancel := context.WithCancel(context.Background()) + t.Cleanup(cancel) + logger := zerolog.New(zerolog.NewTestWriter(t)).Level(zerolog.ErrorLevel) + up := common.NewFakeUpstream("test-ws-upstream") + ci, err := clients.NewWsJsonRpcClient(ctx, &logger, "test-project", up, server.wsURL(t), nil, nil) + require.NoError(t, err) + a.wsClient = ci.(*clients.WsJsonRpcClient) + + // Connected but no newHeads subscription yet. + require.False(t, a.Healthy()) + + a.subsMu.Lock() + a.newHeadsSubID = "0xsub" + a.subsMu.Unlock() + require.True(t, a.Healthy()) +} diff --git a/indexer/health_test.go b/indexer/health_test.go new file mode 100644 index 000000000..ac72fb934 --- /dev/null +++ b/indexer/health_test.go @@ -0,0 +1,109 @@ +package indexer + +import ( + "context" + "testing" + "time" + + "github.com/rs/zerolog" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// healthyIngress wraps fakeIngress with a controllable HealthReporter +// implementation. +type healthyIngress struct { + fakeIngress + healthy bool +} + +func (i *healthyIngress) Healthy() bool { return i.healthy } + +func TestIndexer_IngressHealth(t *testing.T) { + idx := newIndexer(t) + nw := newFakeNetwork("evm:324", 0) + idx.RegisterNetwork(nw) + + t.Run("unknown network", func(t *testing.T) { + live, total := idx.IngressHealth("evm:999") + assert.Equal(t, 0, live) + assert.Equal(t, 0, total) + }) + + t.Run("no ingresses yet", func(t *testing.T) { + live, total := idx.IngressHealth("evm:324") + assert.Equal(t, 0, live) + assert.Equal(t, 0, total) + }) + + up := &healthyIngress{fakeIngress: fakeIngress{name: "ws:up"}, healthy: true} + down := &healthyIngress{fakeIngress: fakeIngress{name: "ws:down"}, healthy: false} + // An ingress that doesn't implement HealthReporter counts as live — + // the indexer can't assess transports it doesn't understand. + opaque := &fakeIngress{name: "kafka:topic"} + + require.NoError(t, idx.AddIngress(context.Background(), "evm:324", up)) + require.NoError(t, idx.AddIngress(context.Background(), "evm:324", down)) + require.NoError(t, idx.AddIngress(context.Background(), "evm:324", opaque)) + + t.Run("mixed health", func(t *testing.T) { + live, total := idx.IngressHealth("evm:324") + assert.Equal(t, 2, live, "healthy reporter + opaque ingress") + assert.Equal(t, 3, total) + }) + + t.Run("all reporters down", func(t *testing.T) { + up.healthy = false + live, total := idx.IngressHealth("evm:324") + assert.Equal(t, 1, live, "only the opaque ingress remains assumed-live") + assert.Equal(t, 3, total) + }) +} + +func TestIndexer_LastHead(t *testing.T) { + now := time.Date(2026, 6, 12, 9, 30, 0, 0, time.UTC) + logger := zerolog.New(zerolog.NewTestWriter(t)) + idx := New(&logger, Options{Now: func() time.Time { return now }}) + nw := newFakeNetwork("evm:324", 0) + idx.RegisterNetwork(nw) + + _, _, ok := idx.LastHead("evm:324") + assert.False(t, ok, "no head delivered yet") + + _, _, ok = idx.LastHead("evm:999") + assert.False(t, ok, "unknown network") + + idx.Ingest(StreamEvent{ + Kind: KindNewHead, + NetworkId: "evm:324", + SourceId: "ws:up", + Block: BlockRef{Number: 42, Hash: "0xaa", ParentHash: "0x99"}, + }) + + block, at, ok := idx.LastHead("evm:324") + require.True(t, ok) + assert.Equal(t, int64(42), block.Number) + assert.True(t, now.Equal(at), "expected %s got %s", now, at) + + // A duplicate head must not move the liveness timestamp (it was + // deduped, not delivered) — but a NEW head must. + now = now.Add(10 * time.Second) + idx.Ingest(StreamEvent{ + Kind: KindNewHead, + NetworkId: "evm:324", + SourceId: "ws:up", + Block: BlockRef{Number: 42, Hash: "0xaa", ParentHash: "0x99"}, + }) + _, at, _ = idx.LastHead("evm:324") + assert.True(t, now.Add(-10*time.Second).Equal(at), "deduped head must not refresh liveness") + + idx.Ingest(StreamEvent{ + Kind: KindNewHead, + NetworkId: "evm:324", + SourceId: "ws:up", + Block: BlockRef{Number: 43, Hash: "0xbb", ParentHash: "0xaa"}, + }) + block, at, _ = idx.LastHead("evm:324") + assert.Equal(t, int64(43), block.Number) + assert.True(t, now.Equal(at), "expected %s got %s", now, at) +} diff --git a/indexer/indexer.go b/indexer/indexer.go index a686fecad..a15da3bd0 100644 --- a/indexer/indexer.go +++ b/indexer/indexer.go @@ -7,6 +7,7 @@ import ( "sync/atomic" "time" + "github.com/erpc/erpc/telemetry" "github.com/rs/zerolog" ) @@ -76,6 +77,13 @@ type networkState struct { lastHead atomic.Pointer[headMarker] headFallback *DedupWindow + // lastHeadAt is the UnixNano timestamp of the most recent delivered + // (post-dedup) newHeads event. Zero until the first head arrives. + // Drives the per-network head-liveness metric and health endpoint so + // a silent head stall is observable instead of only visible to + // subscribed clients. + lastHeadAt atomic.Int64 + // Per-filter dedup windows: filterHash -> *DedupWindow. filterMu sync.RWMutex filterDedup map[string]*DedupWindow @@ -369,6 +377,16 @@ func (i *Indexer) Ingest(ev StreamEvent) { return } + // Head-liveness bookkeeping: record when this network last delivered a + // head so operators can alert on "no heads for network X in Y seconds" + // (time() - gauge) instead of relying on clients to notice silence. + if ev.Kind == KindNewHead && !ev.Block.Zero() { + now := i.opts.Now() + ns.lastHeadAt.Store(now.UnixNano()) + telemetry.GaugeHandle(telemetry.MetricNetworkSubscriptionLastHeadTimestamp, ev.NetworkId). + Set(float64(now.Unix())) + } + // Detect and emit reorg invalidations BEFORE delivering the new // head. Consumers see: (removed logs) → reorg summary → new head. if ev.Kind == KindNewHead && !ev.Block.Zero() { @@ -528,6 +546,59 @@ func (i *Indexer) classify(ns *networkState, ev StreamEvent) Lifecycle { return LifeSoft } +// HealthReporter is an optional interface an EventIngress can implement to +// report whether it can currently deliver events (e.g. a WS upstream +// adapter with a live connection and an active newHeads subscription). +// Ingresses that don't implement it are assumed live — the indexer can't +// assess transports it doesn't understand. +type HealthReporter interface { + Healthy() bool +} + +// IngressHealth returns how many of the network's registered ingresses +// currently report themselves able to deliver events, alongside the total +// registered count. (0, 0) means the network is unknown or has no +// ingresses yet. +func (i *Indexer) IngressHealth(networkId string) (live, total int) { + nsRaw, ok := i.networks.Load(networkId) + if !ok { + return 0, 0 + } + ns := nsRaw.(*networkState) + ns.ingressMu.RLock() + defer ns.ingressMu.RUnlock() + for _, ing := range ns.ingresses { + total++ + if hr, ok := ing.(HealthReporter); ok { + if hr.Healthy() { + live++ + } + } else { + live++ + } + } + return live, total +} + +// LastHead returns the most recent delivered head for the network and when +// it was delivered. ok is false when the network is unknown or no head has +// been delivered yet. +func (i *Indexer) LastHead(networkId string) (block BlockRef, at time.Time, ok bool) { + nsRaw, found := i.networks.Load(networkId) + if !found { + return BlockRef{}, time.Time{}, false + } + ns := nsRaw.(*networkState) + nanos := ns.lastHeadAt.Load() + if nanos == 0 { + return BlockRef{}, time.Time{}, false + } + if head := ns.lastHead.Load(); head != nil { + block = BlockRef{Number: head.num, Hash: head.hash} + } + return block, time.Unix(0, nanos), true +} + // fanOut dispatches to every registered egress whose InterestedIn matches. func (i *Indexer) fanOut(ev IndexedEvent) { i.egresses.Range(func(_, v any) bool { diff --git a/telemetry/metrics.go b/telemetry/metrics.go index f32d5519e..7fb347190 100644 --- a/telemetry/metrics.go +++ b/telemetry/metrics.go @@ -82,6 +82,18 @@ var ( Help: "Dynamically computed block time per network in milliseconds.", }, []string{"project", "network"}) + MetricUpstreamWebsocketConnected = promauto.NewGaugeVec(prometheus.GaugeOpts{ + Namespace: "erpc", + Name: "upstream_websocket_connected", + Help: "Whether the upstream WebSocket connection is currently established (1) or down/wedged (0).", + }, []string{"project", "vendor", "network", "upstream"}) + + MetricNetworkSubscriptionLastHeadTimestamp = promauto.NewGaugeVec(prometheus.GaugeOpts{ + Namespace: "erpc", + Name: "network_subscription_last_head_timestamp_seconds", + Help: "Unix timestamp of the last newHeads event delivered by the subscription indexer for a network. Alert on time() - this > N to catch silent head stalls.", + }, []string{"network"}) + MetricUpstreamCordoned = promauto.NewGaugeVec(prometheus.GaugeOpts{ Namespace: "erpc", Name: "upstream_cordoned", From faf66488bd762817b8d993096f464b280f113dc3 Mon Sep 17 00:00:00 2001 From: snowkide Date: Fri, 12 Jun 2026 16:19:54 +0200 Subject: [PATCH 11/40] test(websocket): end-to-end regression for ungraceful upstream death MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Full-stack reproduction of the 2026-06-12 zkSync incident: real eRPC HTTP server + WS upstream whose TCP connection is killed with no close handshake. Asserts eRPC re-dials, re-subscribes with a fresh upstream sub ID, resumes delivering heads to the already-connected client on its original subscription ID, and accepts new client subscriptions — all with no process restart. Co-Authored-By: Claude Fable 5 --- erpc/ws_server_selfheal_test.go | 235 ++++++++++++++++++++++++++++++++ 1 file changed, 235 insertions(+) create mode 100644 erpc/ws_server_selfheal_test.go diff --git a/erpc/ws_server_selfheal_test.go b/erpc/ws_server_selfheal_test.go new file mode 100644 index 000000000..7f1867fb7 --- /dev/null +++ b/erpc/ws_server_selfheal_test.go @@ -0,0 +1,235 @@ +package erpc + +import ( + "encoding/json" + "fmt" + "strings" + "sync" + "sync/atomic" + "testing" + "time" + + "github.com/erpc/erpc/util" + "github.com/gorilla/websocket" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// selfHealMockUpstream is a WS upstream whose connections can be killed +// WITHOUT a WebSocket close handshake, and which keeps accepting fresh +// connections afterwards — the moral equivalent of a fullnode pod being +// deleted and recreated behind a still-healthy gateway. +type selfHealMockUpstream struct { + mu sync.Mutex + conns []*selfHealConn + connSeen chan *selfHealConn + subSeen chan string // upstream-assigned sub id per eth_subscribe served + subCount atomic.Int64 + lastSubID atomic.Value // string +} + +type selfHealConn struct { + conn *websocket.Conn + writeMu sync.Mutex +} + +func (sc *selfHealConn) writeJSON(v interface{}) { + sc.writeMu.Lock() + defer sc.writeMu.Unlock() + _ = sc.conn.SetWriteDeadline(time.Now().Add(2 * time.Second)) + _ = sc.conn.WriteJSON(v) +} + +func (sc *selfHealConn) sendHead(subID string, num int64) { + sc.writeJSON(map[string]interface{}{ + "jsonrpc": "2.0", + "method": "eth_subscription", + "params": map[string]interface{}{ + "subscription": subID, + "result": map[string]interface{}{ + "number": fmt.Sprintf("0x%x", num), + "hash": fmt.Sprintf("0x%064x", num), + "parentHash": fmt.Sprintf("0x%064x", num-1), + }, + }, + }) +} + +func (m *selfHealMockUpstream) handle(conn *websocket.Conn) { + sc := &selfHealConn{conn: conn} + m.mu.Lock() + m.conns = append(m.conns, sc) + m.mu.Unlock() + select { + case m.connSeen <- sc: + default: + } + + for { + _, msg, err := conn.ReadMessage() + if err != nil { + return + } + var req map[string]interface{} + if err := json.Unmarshal(msg, &req); err != nil { + continue + } + method, _ := req["method"].(string) + id := req["id"] + switch method { + case "eth_chainId": + sc.writeJSON(map[string]interface{}{"jsonrpc": "2.0", "id": id, "result": "0x7b"}) + case "eth_getBlockByNumber": + sc.writeJSON(map[string]interface{}{"jsonrpc": "2.0", "id": id, "result": map[string]interface{}{"number": "0x100", "timestamp": "0x6702a8f0"}}) + case "eth_syncing": + sc.writeJSON(map[string]interface{}{"jsonrpc": "2.0", "id": id, "result": false}) + case "eth_subscribe": + subID := fmt.Sprintf("0xupsub%02x", m.subCount.Add(1)) + m.lastSubID.Store(subID) + sc.writeJSON(map[string]interface{}{"jsonrpc": "2.0", "id": id, "result": subID}) + select { + case m.subSeen <- subID: + default: + } + case "eth_unsubscribe": + sc.writeJSON(map[string]interface{}{"jsonrpc": "2.0", "id": id, "result": true}) + default: + sc.writeJSON(map[string]interface{}{"jsonrpc": "2.0", "id": id, "result": "0x1"}) + } + } +} + +// TestWebSocket_UpstreamDiesUngracefully_SelfHeals is the end-to-end +// regression test for the 2026-06-12 zkSync chain-324 incident: the single +// WS upstream's connection dies with NO close handshake; eRPC must — with +// no process restart — +// +// 1. detect the dead connection and re-dial, +// 2. re-subscribe newHeads upstream (retrying through transient failures), +// 3. resume delivering heads to the ALREADY-CONNECTED client subscription, +// 4. accept NEW client subscriptions afterwards. +func TestWebSocket_UpstreamDiesUngracefully_SelfHeals(t *testing.T) { + mock := &selfHealMockUpstream{ + connSeen: make(chan *selfHealConn, 16), + subSeen: make(chan string, 16), + } + mockSrv := mockWsUpstream(t, mock.handle) + defer mockSrv.Close() + + setupGock() + defer util.ResetGock() + + wsUpstreamURL := "ws" + strings.TrimPrefix(mockSrv.URL, "http") + addr, cleanup := setupTestERPCServer(t, standardWsConfig(wsUpstreamURL)) + defer cleanup() + + // eRPC's upstream WS client connects at upstream registration. + var upConn1 *selfHealConn + select { + case upConn1 = <-mock.connSeen: + case <-time.After(5 * time.Second): + t.Fatal("eRPC never connected to the WS upstream") + } + + // Client subscribes to newHeads (this lazily bootstraps the network's + // indexer ingress, which issues the upstream eth_subscribe). + client := dialWs(t, addr) + defer client.Close() + resp := sendAndReceive(t, client, `{"jsonrpc":"2.0","id":1,"method":"eth_subscribe","params":["newHeads"]}`) + require.Nil(t, resp["error"], "subscribe must succeed while the upstream is healthy: %v", resp["error"]) + clientSubID, ok := resp["result"].(string) + require.True(t, ok, "subscription ID should be a string, got: %v", resp["result"]) + + var sub1 string + select { + case sub1 = <-mock.subSeen: + case <-time.After(5 * time.Second): + t.Fatal("eRPC never subscribed newHeads on the upstream") + } + + // pushHeadsUntilReceived repeatedly emits fresh heads on the upstream + // connection until one reaches the given client (the upstream subscribe + // response and eRPC's handler registration race the first push — real + // chains emit heads continuously, so a retry loop models reality). + headNum := int64(0x200) + pushHeadsUntilReceived := func(c *websocket.Conn, up *selfHealConn, subID string, within time.Duration) map[string]interface{} { + t.Helper() + deadline := time.Now().Add(within) + notifCh := make(chan map[string]interface{}, 1) + errCh := make(chan error, 1) + go func() { + _ = c.SetReadDeadline(deadline) + _, msg, err := c.ReadMessage() + if err != nil { + errCh <- err + return + } + var notif map[string]interface{} + if err := json.Unmarshal(msg, ¬if); err != nil { + errCh <- err + return + } + notifCh <- notif + }() + for { + headNum++ + up.sendHead(subID, headNum) + select { + case notif := <-notifCh: + return notif + case err := <-errCh: + t.Fatalf("client read failed while waiting for a head: %v", err) + return nil + case <-time.After(250 * time.Millisecond): + if time.Now().After(deadline) { + t.Fatalf("client never received a head within %s", within) + return nil + } + } + } + } + + // Sanity: a head flows end-to-end before the failure. + notif := pushHeadsUntilReceived(client, upConn1, sub1, 5*time.Second) + require.Equal(t, "eth_subscription", notif["method"]) + require.Equal(t, clientSubID, notif["params"].(map[string]interface{})["subscription"]) + + // ── The incident ────────────────────────────────────────────────── + // Kill the upstream TCP connection with NO WebSocket close handshake. + _ = upConn1.conn.UnderlyingConn().Close() + + // eRPC must re-dial (the gateway/httptest server is still up, exactly + // like the incident topology) ... + var upConn2 *selfHealConn + select { + case upConn2 = <-mock.connSeen: + case <-time.After(10 * time.Second): + t.Fatal("eRPC never re-dialed the upstream after the ungraceful kill") + } + + // ... and re-subscribe newHeads with a fresh upstream subscription. + var sub2 string + select { + case sub2 = <-mock.subSeen: + case <-time.After(10 * time.Second): + t.Fatal("eRPC never re-subscribed newHeads after reconnecting") + } + assert.NotEqual(t, sub1, sub2, "the re-subscribe must be a fresh upstream subscription") + + // The ALREADY-CONNECTED client must resume receiving heads on its + // ORIGINAL subscription ID, with no client-side action. + notif = pushHeadsUntilReceived(client, upConn2, sub2, 10*time.Second) + assert.Equal(t, "eth_subscription", notif["method"]) + assert.Equal(t, clientSubID, notif["params"].(map[string]interface{})["subscription"], + "recovered heads must arrive on the client's original subscription ID") + + // And a NEW client must be able to subscribe and receive heads. + client2 := dialWs(t, addr) + defer client2.Close() + resp2 := sendAndReceive(t, client2, `{"jsonrpc":"2.0","id":1,"method":"eth_subscribe","params":["newHeads"]}`) + require.Nil(t, resp2["error"], "new client subscribe must succeed after recovery: %v", resp2["error"]) + client2SubID, _ := resp2["result"].(string) + + notif2 := pushHeadsUntilReceived(client2, upConn2, sub2, 10*time.Second) + assert.Equal(t, client2SubID, notif2["params"].(map[string]interface{})["subscription"]) +} From 188639026b7a430c5f264cf0343d42ce8778a041 Mon Sep 17 00:00:00 2001 From: snowkide Date: Fri, 12 Jun 2026 18:48:46 +0200 Subject: [PATCH 12/40] refactor(websocket): slim to core fixes per review; fix races Per review feedback (health reporter excessive vs ping/pong), drop the observability layer and keep only the fixes required to prevent the incident: - revert HealthReporter/IngressHealth/LastHead (indexer), healthcheck subscriptions block, newHeads refusal gate + ErrNoLiveSubscriptionSource, and the last-head-timestamp metric; delete their tests. Kept: client liveness, resubscribe retry loop, breaker default fix, and the erpc_upstream_websocket_connected gauge. Race fixes from review: - snapshot wsPingInterval/wsPongWait (and resub backoff bounds) into per-client/per-adapter fields at construction; goroutines no longer read package vars, fixing the -race failure where test cleanup restored them while a prior client's pingLoop was still running - pingLoop: write the ping to an explicit conn and compare-and-close via teardownConn, so a ping failure can no longer close a fresh connection that readLoop already re-dialed - subscribeNewHeads/subscribeFilter: check epoch ctx under subsMu before committing a sub ID, so a cancelled epoch's in-flight subscribe can't unregister the live handler installed by the newer epoch - Stop marks the adapter stopped under resubMu so a reconnect callback racing Stop can't start a new resubscribe epoch Co-Authored-By: Claude Fable 5 --- clients/ws_json_rpc_client.go | 56 ++++++---- common/errors.go | 26 ----- erpc/healthcheck.go | 12 --- erpc/subscription_manager.go | 75 ------------- erpc/subscription_manager_health_test.go | 131 ----------------------- indexer/adapters/wsupstream/adapter.go | 56 +++++++--- indexer/health_test.go | 109 ------------------- indexer/indexer.go | 71 ------------ telemetry/metrics.go | 6 -- 9 files changed, 79 insertions(+), 463 deletions(-) delete mode 100644 erpc/subscription_manager_health_test.go delete mode 100644 indexer/health_test.go diff --git a/clients/ws_json_rpc_client.go b/clients/ws_json_rpc_client.go index d9b28f00c..c10b7a930 100644 --- a/clients/ws_json_rpc_client.go +++ b/clients/ws_json_rpc_client.go @@ -41,8 +41,9 @@ const ( // re-dials. wsPongWait must comfortably exceed wsPingInterval so at least // two pings fit in the window. // -// Vars (not consts) so tests can compress time; production code must not -// mutate them. +// Vars (not consts) so tests can compress time. They are copied into +// per-client fields at construction, so client goroutines never read them +// after NewWsJsonRpcClient returns. var ( wsPingInterval = 30 * time.Second wsPongWait = 75 * time.Second @@ -58,6 +59,11 @@ type WsJsonRpcClient struct { appCtx context.Context logger *zerolog.Logger + // Liveness windows, snapshotted from wsPingInterval/wsPongWait at + // construction (before any goroutine starts). + pingInterval time.Duration + pongWait time.Duration + // Connection state connMu sync.Mutex conn *websocket.Conn @@ -156,6 +162,8 @@ func NewWsJsonRpcClient( client := &WsJsonRpcClient{ Url: parsedUrl, headers: headers, + pingInterval: wsPingInterval, + pongWait: wsPongWait, projectId: projectId, upstream: upstream, appCtx: appCtx, @@ -367,13 +375,13 @@ func (c *WsJsonRpcClient) connect() error { } // Arm the liveness deadline: if neither a pong nor a data frame arrives - // within wsPongWait, ReadMessage fails and readLoop re-dials. The pong + // within pongWait, ReadMessage fails and readLoop re-dials. The pong // handler runs inside ReadMessage's frame processing, so extending the // deadline here covers the ping/pong path; readLoop extends it again on // every data frame. - _ = conn.SetReadDeadline(time.Now().Add(wsPongWait)) + _ = conn.SetReadDeadline(time.Now().Add(c.pongWait)) conn.SetPongHandler(func(string) error { - return conn.SetReadDeadline(time.Now().Add(wsPongWait)) + return conn.SetReadDeadline(time.Now().Add(c.pongWait)) }) if c.conn != nil { @@ -451,7 +459,7 @@ func (c *WsJsonRpcClient) readLoop() { if websocket.IsCloseError(err, websocket.CloseNormalClosure, websocket.CloseGoingAway) { c.logger.Info().Msg("websocket connection closed normally") } else if errors.As(err, &netErr) && netErr.Timeout() { - c.logger.Warn().Err(err).Dur("pongWait", wsPongWait). + c.logger.Warn().Err(err).Dur("pongWait", c.pongWait). Msg("websocket peer silent beyond liveness deadline (no pong/data), tearing down connection and reconnecting") } else { c.logger.Warn().Err(err).Msg("websocket read error, will reconnect") @@ -468,7 +476,7 @@ func (c *WsJsonRpcClient) readLoop() { // Any inbound frame proves the peer is alive — push the liveness // deadline forward. - _ = conn.SetReadDeadline(time.Now().Add(wsPongWait)) + _ = conn.SetReadDeadline(time.Now().Add(c.pongWait)) c.handleMessage(message) } @@ -616,9 +624,6 @@ func (c *WsJsonRpcClient) drainPending(err error) { } func (c *WsJsonRpcClient) writeMessage(messageType int, data []byte) error { - c.writeMu.Lock() - defer c.writeMu.Unlock() - c.connMu.Lock() conn := c.conn c.connMu.Unlock() @@ -626,6 +631,15 @@ func (c *WsJsonRpcClient) writeMessage(messageType int, data []byte) error { if conn == nil { return fmt.Errorf("websocket connection not established") } + return c.writeToConn(conn, messageType, data) +} + +// writeToConn writes to an explicit connection so callers that need to act +// on a write failure (e.g. pingLoop closing the broken conn) operate on the +// exact connection they wrote to, not whatever c.conn points at by then. +func (c *WsJsonRpcClient) writeToConn(conn *websocket.Conn, messageType int, data []byte) error { + c.writeMu.Lock() + defer c.writeMu.Unlock() if err := conn.SetWriteDeadline(time.Now().Add(wsWriteWait)); err != nil { return err @@ -634,7 +648,7 @@ func (c *WsJsonRpcClient) writeMessage(messageType int, data []byte) error { } func (c *WsJsonRpcClient) pingLoop() { - ticker := time.NewTicker(wsPingInterval) + ticker := time.NewTicker(c.pingInterval) defer ticker.Stop() for { @@ -643,19 +657,23 @@ func (c *WsJsonRpcClient) pingLoop() { if !c.connected.Load() { continue } - if err := c.writeMessage(websocket.PingMessage, nil); err != nil { - // A failed ping write means the connection is unusable. + c.connMu.Lock() + conn := c.conn + c.connMu.Unlock() + if conn == nil { + continue + } + if err := c.writeToConn(conn, websocket.PingMessage, nil); err != nil { + // A failed ping write means this connection is unusable. // Close it so readLoop's blocked ReadMessage fails and the // teardown+reconnect path (owned by readLoop) takes over — // logging alone here previously left the client wedged on a // connection that could never deliver another frame. + // teardownConn only clears c.conn if it still points at this + // same conn, so a concurrent reconnect's fresh connection is + // never the one closed here. c.logger.Warn().Err(err).Msg("websocket ping write failed, closing connection to force reconnect") - c.connMu.Lock() - conn := c.conn - c.connMu.Unlock() - if conn != nil { - _ = conn.Close() - } + c.teardownConn(conn) } case <-c.appCtx.Done(): return diff --git a/common/errors.go b/common/errors.go index a8c4bcf34..3073feb31 100644 --- a/common/errors.go +++ b/common/errors.go @@ -2824,32 +2824,6 @@ func (e *ErrNoWsUpstreamAvailable) ErrorStatusCode() int { return http.StatusBadRequest } -type ErrNoLiveSubscriptionSource struct{ BaseError } - -const ErrCodeNoLiveSubscriptionSource ErrorCode = "ErrNoLiveSubscriptionSource" - -// NewErrNoLiveSubscriptionSource is returned when WS upstreams are -// configured for the network but none currently has a live connection with -// an active newHeads subscription. Refusing the subscription (HTTP 503 / -// retryable) lets clients fail over to another node instead of holding a -// subscription ID that will never deliver. -var NewErrNoLiveSubscriptionSource = func(networkId string, totalIngresses int) error { - return &ErrNoLiveSubscriptionSource{ - BaseError{ - Code: ErrCodeNoLiveSubscriptionSource, - Message: fmt.Sprintf("no upstream is currently able to deliver subscription events for network %s; refusing subscription so the client can fail over", networkId), - Details: map[string]interface{}{ - "networkId": networkId, - "totalIngresses": totalIngresses, - }, - }, - } -} - -func (e *ErrNoLiveSubscriptionSource) ErrorStatusCode() int { - return http.StatusServiceUnavailable -} - type ErrSubscriptionLimitExceeded struct{ BaseError } const ErrCodeSubscriptionLimitExceeded ErrorCode = "ErrSubscriptionLimitExceeded" diff --git a/erpc/healthcheck.go b/erpc/healthcheck.go index bbf03f4bb..85e4953b5 100644 --- a/erpc/healthcheck.go +++ b/erpc/healthcheck.go @@ -48,11 +48,6 @@ type NetworkHealthData struct { Status string `json:"status"` Message string `json:"message,omitempty"` Upstreams map[string]*UpstreamHealthData `json:"upstreams"` - - // Subscriptions reports WS head-delivery liveness for this network. - // Present only when at least one client has subscribed on the network - // since the process started. - Subscriptions *NetworkSubscriptionHealth `json:"subscriptions,omitempty"` } type UpstreamHealthData struct { @@ -250,13 +245,6 @@ func (s *HttpServer) handleHealthCheck( ms := float64(bt.Milliseconds()) networkHealth.BlockTimeMs = &ms } - // Subscription head-liveness: nil unless a client has - // subscribed on this network at least once. Lets load - // balancers and operators see "this pod delivers no heads - // for network X" without an active client subscription. - if s.subscriptionManager != nil { - networkHealth.Subscriptions = s.subscriptionManager.SubscriptionHealth(networkId) - } projectHealth.Networks[networkId] = networkHealth } diff --git a/erpc/subscription_manager.go b/erpc/subscription_manager.go index d5a5c0e1b..54b78f152 100644 --- a/erpc/subscription_manager.go +++ b/erpc/subscription_manager.go @@ -38,19 +38,8 @@ const ( // unsubscribeTimeout is the deadline for best-effort upstream // unsubscribe calls during connection cleanup. unsubscribeTimeout = 5 * time.Second - - // liveHeadSourcePollEvery is how often waitForLiveHeadSource re-checks - // ingress health while waiting out the bootstrap race. - liveHeadSourcePollEvery = 100 * time.Millisecond ) -// liveHeadSourceWaitMax bounds how long a newHeads subscribe waits for at -// least one ingress to come alive before refusing the subscription. Long -// enough to cover the initial bootstrap (adapter connect + eth_subscribe -// round-trip), short enough that a client talking to a head-less pod fails -// over quickly. Var so tests can compress time. -var liveHeadSourceWaitMax = 3 * time.Second - // SubscriptionManager is the client-facing egress layer. It owns // per-connection *wsclient.Adapter instances, lazily registers networks + // ingresses with the indexer the first time a client subscribes on a @@ -178,19 +167,6 @@ func (sm *SubscriptionManager) Subscribe( return nil, fmt.Errorf("failed to generate subscription ID: %w", err) } - // newHeads is fan-out only — no per-filter EnsureFilter ever touches an - // upstream for it, so without this check a pod whose WS upstreams are - // all down (or resubscribing) would happily return a subscription ID - // that never delivers a single head. Refuse instead so the client can - // retry/fail over. Filter subs get equivalent protection from - // EnsureFilter, which errors when every ingress fails. - if subType == SubTypeNewHeads { - if err := sm.waitForLiveHeadSource(ctx, networkId); err != nil { - sm.recordFailureMetrics(project, nw, method, reqFinality, start, nq, err) - return nil, err - } - } - kind, filterHash, err := sm.resolveSubscription(ctx, networkId, subType, jrReq.Params) if err != nil { sm.recordFailureMetrics(project, nw, method, reqFinality, start, nq, err) @@ -324,57 +300,6 @@ func (sm *SubscriptionManager) CleanupConnection(wsc *WsConnection, _ *PreparedP lg.Debug().Msg("cleaned up all subscriptions for connection") } -// NetworkSubscriptionHealth summarizes a network's head-delivery liveness -// for the health endpoint. Nil/absent when the network has never been -// bootstrapped (no client ever subscribed on it). -type NetworkSubscriptionHealth struct { - LiveIngresses int `json:"liveIngresses"` - TotalIngresses int `json:"totalIngresses"` - LastHeadNumber int64 `json:"lastHeadNumber,omitempty"` - LastHeadAt string `json:"lastHeadAt,omitempty"` - LastHeadAgeSec int64 `json:"lastHeadAgeSeconds,omitempty"` -} - -// SubscriptionHealth reports the network's subscription liveness, or nil -// when the network was never bootstrapped for subscriptions. -func (sm *SubscriptionManager) SubscriptionHealth(networkId string) *NetworkSubscriptionHealth { - if _, ok := sm.networks.Load(networkId); !ok { - return nil - } - live, total := sm.idx.IngressHealth(networkId) - out := &NetworkSubscriptionHealth{LiveIngresses: live, TotalIngresses: total} - if block, at, ok := sm.idx.LastHead(networkId); ok { - out.LastHeadNumber = block.Number - out.LastHeadAt = at.UTC().Format(time.RFC3339) - out.LastHeadAgeSec = int64(time.Since(at).Seconds()) - } - return out -} - -// waitForLiveHeadSource returns nil as soon as at least one of the -// network's ingresses reports it can deliver heads. The bounded wait -// covers the bootstrap race where adapters' initial eth_subscribe calls -// are still in flight; after that it refuses with a retryable error. -func (sm *SubscriptionManager) waitForLiveHeadSource(ctx context.Context, networkId string) error { - deadline := time.Now().Add(liveHeadSourceWaitMax) - for { - live, total := sm.idx.IngressHealth(networkId) - if live > 0 { - return nil - } - if ctx.Err() != nil || time.Now().After(deadline) { - sm.logger.Warn().Str("networkId", networkId).Int("totalIngresses", total). - Msg("refusing newHeads subscription: no live head source on this instance") - return common.NewErrNoLiveSubscriptionSource(networkId, total) - } - select { - case <-ctx.Done(): - return common.NewErrNoLiveSubscriptionSource(networkId, total) - case <-time.After(liveHeadSourcePollEvery): - } - } -} - // --- internals -------------------------------------------------------- // buildWsAdapterOptions resolves network-level toggles that the wsupstream diff --git a/erpc/subscription_manager_health_test.go b/erpc/subscription_manager_health_test.go deleted file mode 100644 index 2a0580591..000000000 --- a/erpc/subscription_manager_health_test.go +++ /dev/null @@ -1,131 +0,0 @@ -package erpc - -import ( - "context" - "testing" - "time" - - "github.com/erpc/erpc/common" - "github.com/erpc/erpc/indexer" - "github.com/rs/zerolog" - "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/require" -) - -type stubNetworkHandle struct{ id string } - -func (h stubNetworkHandle) Id() string { return h.id } -func (h stubNetworkHandle) FinalityDepth() int64 { return 0 } -func (h stubNetworkHandle) SuggestLatestBlock(string, int64) {} - -// stubIngress implements indexer.EventIngress plus indexer.HealthReporter -// with a flippable health flag. -type stubIngress struct { - name string - healthy bool -} - -func (i *stubIngress) Name() string { return i.name } -func (i *stubIngress) Start(context.Context, indexer.NetworkHandle, indexer.Sink) error { - return nil -} -func (i *stubIngress) EnsureFilter(context.Context, string, string, []interface{}) error { return nil } -func (i *stubIngress) RemoveFilter(context.Context, string, string) error { return nil } -func (i *stubIngress) Stop(context.Context) error { return nil } -func (i *stubIngress) Healthy() bool { return i.healthy } - -func newTestSubscriptionManager(t *testing.T) (*SubscriptionManager, *indexer.Indexer) { - t.Helper() - logger := zerolog.New(zerolog.NewTestWriter(t)).Level(zerolog.ErrorLevel) - idx := indexer.New(&logger, indexer.Options{}) - return NewSubscriptionManager(&logger, idx), idx -} - -// TestWaitForLiveHeadSource pins the incident-driven contract: a pod with -// zero live head sources must refuse newHeads subscriptions (retryable -// error) instead of handing out a subscription ID that never delivers — -// the silent failure mode that hid the 2026-06-12 zkSync outage for hours. -func TestWaitForLiveHeadSource(t *testing.T) { - origWait := liveHeadSourceWaitMax - liveHeadSourceWaitMax = 300 * time.Millisecond - t.Cleanup(func() { liveHeadSourceWaitMax = origWait }) - - const networkID = "evm:324" - - t.Run("refuses when no ingress is live", func(t *testing.T) { - sm, idx := newTestSubscriptionManager(t) - idx.RegisterNetwork(stubNetworkHandle{id: networkID}) - require.NoError(t, idx.AddIngress(context.Background(), networkID, &stubIngress{name: "ws:a", healthy: false})) - - err := sm.waitForLiveHeadSource(context.Background(), networkID) - require.Error(t, err) - assert.True(t, common.HasErrorCode(err, common.ErrCodeNoLiveSubscriptionSource), - "expected ErrNoLiveSubscriptionSource, got: %v", err) - }) - - t.Run("passes immediately when an ingress is live", func(t *testing.T) { - sm, idx := newTestSubscriptionManager(t) - idx.RegisterNetwork(stubNetworkHandle{id: networkID}) - require.NoError(t, idx.AddIngress(context.Background(), networkID, &stubIngress{name: "ws:a", healthy: true})) - - start := time.Now() - require.NoError(t, sm.waitForLiveHeadSource(context.Background(), networkID)) - assert.Less(t, time.Since(start), liveHeadSourceWaitMax/2, - "a live source must not incur the bootstrap grace wait") - }) - - t.Run("passes when an ingress becomes live during the grace wait", func(t *testing.T) { - sm, idx := newTestSubscriptionManager(t) - idx.RegisterNetwork(stubNetworkHandle{id: networkID}) - ing := &stubIngress{name: "ws:a", healthy: false} - require.NoError(t, idx.AddIngress(context.Background(), networkID, ing)) - - go func() { - time.Sleep(120 * time.Millisecond) - ing.healthy = true - }() - require.NoError(t, sm.waitForLiveHeadSource(context.Background(), networkID)) - }) - - t.Run("honours caller context cancellation", func(t *testing.T) { - sm, idx := newTestSubscriptionManager(t) - idx.RegisterNetwork(stubNetworkHandle{id: networkID}) - require.NoError(t, idx.AddIngress(context.Background(), networkID, &stubIngress{name: "ws:a", healthy: false})) - - ctx, cancel := context.WithTimeout(context.Background(), 50*time.Millisecond) - defer cancel() - err := sm.waitForLiveHeadSource(ctx, networkID) - require.Error(t, err) - assert.True(t, common.HasErrorCode(err, common.ErrCodeNoLiveSubscriptionSource)) - }) -} - -func TestSubscriptionHealth(t *testing.T) { - const networkID = "evm:324" - sm, idx := newTestSubscriptionManager(t) - - assert.Nil(t, sm.SubscriptionHealth(networkID), "nil before the network is bootstrapped") - - idx.RegisterNetwork(stubNetworkHandle{id: networkID}) - require.NoError(t, idx.AddIngress(context.Background(), networkID, &stubIngress{name: "ws:a", healthy: true})) - require.NoError(t, idx.AddIngress(context.Background(), networkID, &stubIngress{name: "ws:b", healthy: false})) - sm.networks.Store(networkID, struct{}{}) - - h := sm.SubscriptionHealth(networkID) - require.NotNil(t, h) - assert.Equal(t, 1, h.LiveIngresses) - assert.Equal(t, 2, h.TotalIngresses) - assert.Empty(t, h.LastHeadAt, "no head delivered yet") - - idx.Ingest(indexer.StreamEvent{ - Kind: indexer.KindNewHead, - NetworkId: networkID, - SourceId: "ws:a", - Block: indexer.BlockRef{Number: 99, Hash: "0xaa", ParentHash: "0x98"}, - }) - - h = sm.SubscriptionHealth(networkID) - require.NotNil(t, h) - assert.Equal(t, int64(99), h.LastHeadNumber) - assert.NotEmpty(t, h.LastHeadAt) -} diff --git a/indexer/adapters/wsupstream/adapter.go b/indexer/adapters/wsupstream/adapter.go index 88f6e5178..5365da3d9 100644 --- a/indexer/adapters/wsupstream/adapter.go +++ b/indexer/adapters/wsupstream/adapter.go @@ -33,7 +33,8 @@ const ( ) // Resubscribe retry backoff bounds. Vars (not consts) so tests can compress -// time; production code must not mutate them. +// time. Copied into per-adapter fields in New(), so adapter goroutines +// never read them after construction. var ( resubRetryMin = 1 * time.Second resubRetryMax = 30 * time.Second @@ -61,11 +62,18 @@ type Adapter struct { // apply); overridable in tests. forward func(ctx context.Context, nq *common.NormalizedRequest, bypassMethodExclusion bool) (*common.NormalizedResponse, error) - // resubMu guards resubCancel: at most one resubscribe retry loop runs - // per connection epoch (initial connect or reconnect); a disconnect or - // Stop cancels it. + // resubMu guards resubCancel/stopped: at most one resubscribe retry + // loop runs per connection epoch (initial connect or reconnect); a + // disconnect or Stop cancels it. stopped prevents a reconnect callback + // racing Stop from starting a fresh epoch on a stopped adapter. resubMu sync.Mutex resubCancel context.CancelFunc + stopped bool + + // Retry backoff bounds, snapshotted from resubRetryMin/resubRetryMax + // at construction. + retryMin time.Duration + retryMax time.Duration // stripSubscribeFromBlockZero controls whether fromBlock: "0x0" is // removed from eth_subscribe logs filters before forwarding upstream. @@ -126,6 +134,8 @@ func New(up *upstream.Upstream, networkID string, logger *zerolog.Logger, opts * forward: func(ctx context.Context, nq *common.NormalizedRequest, bypassMethodExclusion bool) (*common.NormalizedResponse, error) { return up.Forward(ctx, nq, bypassMethodExclusion, false) }, + retryMin: resubRetryMin, + retryMax: resubRetryMax, } if opts != nil { a.stripSubscribeFromBlockZero = opts.StripSubscribeFromBlockZero @@ -154,7 +164,7 @@ func (a *Adapter) Start(_ context.Context, nw indexer.NetworkHandle, sink indexe }) a.wsClient.SetOnDisconnect(cbID, func() { a.logger.Info().Msg("WS disconnected — active subs will re-subscribe on reconnect") - a.stopResubscribe() + a.stopResubscribe(false) // The upstream-assigned subscription IDs died with the connection; // forget the newHeads sub so Healthy() reports honestly until the // reconnect-epoch resubscribe succeeds. @@ -176,9 +186,7 @@ func (a *Adapter) Start(_ context.Context, nw indexer.NetworkHandle, sink indexe // Healthy reports whether this ingress currently has a live upstream WS // connection AND an active newHeads subscription — i.e. it can actually -// deliver heads right now. Consulted by the client-facing layer before -// handing out newHeads subscription IDs, so clients are refused (and can -// fail over) instead of receiving a subscription that will never fire. +// deliver heads right now. func (a *Adapter) Healthy() bool { if !a.wsClient.IsConnected() { return false @@ -192,6 +200,10 @@ func (a *Adapter) Healthy() bool { // epoch, cancelling any loop left over from a previous epoch. func (a *Adapter) startResubscribe() { a.resubMu.Lock() + if a.stopped { + a.resubMu.Unlock() + return + } if a.resubCancel != nil { a.resubCancel() } @@ -203,9 +215,13 @@ func (a *Adapter) startResubscribe() { // stopResubscribe cancels the in-flight retry loop, if any. Called on // disconnect (the loop's subscribes can't succeed anyway; the next -// reconnect starts a fresh epoch) and on Stop. -func (a *Adapter) stopResubscribe() { +// reconnect starts a fresh epoch) and on Stop. forever additionally marks +// the adapter stopped so no future epoch can start. +func (a *Adapter) stopResubscribe(forever bool) { a.resubMu.Lock() + if forever { + a.stopped = true + } if a.resubCancel != nil { a.resubCancel() a.resubCancel = nil @@ -262,7 +278,7 @@ func (a *Adapter) Stop(ctx context.Context) error { cbID := a.Name() a.wsClient.RemoveOnReconnect(cbID) a.wsClient.RemoveOnDisconnect(cbID) - a.stopResubscribe() + a.stopResubscribe(true) a.subsMu.Lock() subs := a.filters @@ -310,7 +326,7 @@ func (a *Adapter) resubscribeWithRetry(ctx context.Context) { } a.subsMu.Unlock() - backoff := resubRetryMin + backoff := a.retryMin for { if ctx.Err() != nil { return @@ -354,8 +370,8 @@ func (a *Adapter) resubscribeWithRetry(ctx context.Context) { case <-time.After(backoff): } backoff *= 2 - if backoff > resubRetryMax { - backoff = resubRetryMax + if backoff > a.retryMax { + backoff = a.retryMax } } } @@ -366,6 +382,13 @@ func (a *Adapter) subscribeNewHeads(ctx context.Context) error { return err } a.subsMu.Lock() + if ctx.Err() != nil { + // Epoch was cancelled while the subscribe was in flight — a newer + // epoch owns the subscription state now; committing this (dead + // connection's) sub ID would unregister the live handler. + a.subsMu.Unlock() + return ctx.Err() + } if a.newHeadsSubID != "" { a.wsClient.UnregisterSubscriptionHandler(a.newHeadsSubID) } @@ -395,6 +418,11 @@ func (a *Adapter) subscribeFilter(ctx context.Context, sub *filterSub) error { return fmt.Errorf("filter subscribe: %w", err) } a.subsMu.Lock() + if ctx.Err() != nil { + // Cancelled mid-flight; see subscribeNewHeads. + a.subsMu.Unlock() + return ctx.Err() + } // Replace any previous upstreamSub for this (subType, paramsHash). if sub.upstreamSub != "" { a.wsClient.UnregisterSubscriptionHandler(sub.upstreamSub) diff --git a/indexer/health_test.go b/indexer/health_test.go deleted file mode 100644 index ac72fb934..000000000 --- a/indexer/health_test.go +++ /dev/null @@ -1,109 +0,0 @@ -package indexer - -import ( - "context" - "testing" - "time" - - "github.com/rs/zerolog" - "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/require" -) - -// healthyIngress wraps fakeIngress with a controllable HealthReporter -// implementation. -type healthyIngress struct { - fakeIngress - healthy bool -} - -func (i *healthyIngress) Healthy() bool { return i.healthy } - -func TestIndexer_IngressHealth(t *testing.T) { - idx := newIndexer(t) - nw := newFakeNetwork("evm:324", 0) - idx.RegisterNetwork(nw) - - t.Run("unknown network", func(t *testing.T) { - live, total := idx.IngressHealth("evm:999") - assert.Equal(t, 0, live) - assert.Equal(t, 0, total) - }) - - t.Run("no ingresses yet", func(t *testing.T) { - live, total := idx.IngressHealth("evm:324") - assert.Equal(t, 0, live) - assert.Equal(t, 0, total) - }) - - up := &healthyIngress{fakeIngress: fakeIngress{name: "ws:up"}, healthy: true} - down := &healthyIngress{fakeIngress: fakeIngress{name: "ws:down"}, healthy: false} - // An ingress that doesn't implement HealthReporter counts as live — - // the indexer can't assess transports it doesn't understand. - opaque := &fakeIngress{name: "kafka:topic"} - - require.NoError(t, idx.AddIngress(context.Background(), "evm:324", up)) - require.NoError(t, idx.AddIngress(context.Background(), "evm:324", down)) - require.NoError(t, idx.AddIngress(context.Background(), "evm:324", opaque)) - - t.Run("mixed health", func(t *testing.T) { - live, total := idx.IngressHealth("evm:324") - assert.Equal(t, 2, live, "healthy reporter + opaque ingress") - assert.Equal(t, 3, total) - }) - - t.Run("all reporters down", func(t *testing.T) { - up.healthy = false - live, total := idx.IngressHealth("evm:324") - assert.Equal(t, 1, live, "only the opaque ingress remains assumed-live") - assert.Equal(t, 3, total) - }) -} - -func TestIndexer_LastHead(t *testing.T) { - now := time.Date(2026, 6, 12, 9, 30, 0, 0, time.UTC) - logger := zerolog.New(zerolog.NewTestWriter(t)) - idx := New(&logger, Options{Now: func() time.Time { return now }}) - nw := newFakeNetwork("evm:324", 0) - idx.RegisterNetwork(nw) - - _, _, ok := idx.LastHead("evm:324") - assert.False(t, ok, "no head delivered yet") - - _, _, ok = idx.LastHead("evm:999") - assert.False(t, ok, "unknown network") - - idx.Ingest(StreamEvent{ - Kind: KindNewHead, - NetworkId: "evm:324", - SourceId: "ws:up", - Block: BlockRef{Number: 42, Hash: "0xaa", ParentHash: "0x99"}, - }) - - block, at, ok := idx.LastHead("evm:324") - require.True(t, ok) - assert.Equal(t, int64(42), block.Number) - assert.True(t, now.Equal(at), "expected %s got %s", now, at) - - // A duplicate head must not move the liveness timestamp (it was - // deduped, not delivered) — but a NEW head must. - now = now.Add(10 * time.Second) - idx.Ingest(StreamEvent{ - Kind: KindNewHead, - NetworkId: "evm:324", - SourceId: "ws:up", - Block: BlockRef{Number: 42, Hash: "0xaa", ParentHash: "0x99"}, - }) - _, at, _ = idx.LastHead("evm:324") - assert.True(t, now.Add(-10*time.Second).Equal(at), "deduped head must not refresh liveness") - - idx.Ingest(StreamEvent{ - Kind: KindNewHead, - NetworkId: "evm:324", - SourceId: "ws:up", - Block: BlockRef{Number: 43, Hash: "0xbb", ParentHash: "0xaa"}, - }) - block, at, _ = idx.LastHead("evm:324") - assert.Equal(t, int64(43), block.Number) - assert.True(t, now.Equal(at), "expected %s got %s", now, at) -} diff --git a/indexer/indexer.go b/indexer/indexer.go index a15da3bd0..a686fecad 100644 --- a/indexer/indexer.go +++ b/indexer/indexer.go @@ -7,7 +7,6 @@ import ( "sync/atomic" "time" - "github.com/erpc/erpc/telemetry" "github.com/rs/zerolog" ) @@ -77,13 +76,6 @@ type networkState struct { lastHead atomic.Pointer[headMarker] headFallback *DedupWindow - // lastHeadAt is the UnixNano timestamp of the most recent delivered - // (post-dedup) newHeads event. Zero until the first head arrives. - // Drives the per-network head-liveness metric and health endpoint so - // a silent head stall is observable instead of only visible to - // subscribed clients. - lastHeadAt atomic.Int64 - // Per-filter dedup windows: filterHash -> *DedupWindow. filterMu sync.RWMutex filterDedup map[string]*DedupWindow @@ -377,16 +369,6 @@ func (i *Indexer) Ingest(ev StreamEvent) { return } - // Head-liveness bookkeeping: record when this network last delivered a - // head so operators can alert on "no heads for network X in Y seconds" - // (time() - gauge) instead of relying on clients to notice silence. - if ev.Kind == KindNewHead && !ev.Block.Zero() { - now := i.opts.Now() - ns.lastHeadAt.Store(now.UnixNano()) - telemetry.GaugeHandle(telemetry.MetricNetworkSubscriptionLastHeadTimestamp, ev.NetworkId). - Set(float64(now.Unix())) - } - // Detect and emit reorg invalidations BEFORE delivering the new // head. Consumers see: (removed logs) → reorg summary → new head. if ev.Kind == KindNewHead && !ev.Block.Zero() { @@ -546,59 +528,6 @@ func (i *Indexer) classify(ns *networkState, ev StreamEvent) Lifecycle { return LifeSoft } -// HealthReporter is an optional interface an EventIngress can implement to -// report whether it can currently deliver events (e.g. a WS upstream -// adapter with a live connection and an active newHeads subscription). -// Ingresses that don't implement it are assumed live — the indexer can't -// assess transports it doesn't understand. -type HealthReporter interface { - Healthy() bool -} - -// IngressHealth returns how many of the network's registered ingresses -// currently report themselves able to deliver events, alongside the total -// registered count. (0, 0) means the network is unknown or has no -// ingresses yet. -func (i *Indexer) IngressHealth(networkId string) (live, total int) { - nsRaw, ok := i.networks.Load(networkId) - if !ok { - return 0, 0 - } - ns := nsRaw.(*networkState) - ns.ingressMu.RLock() - defer ns.ingressMu.RUnlock() - for _, ing := range ns.ingresses { - total++ - if hr, ok := ing.(HealthReporter); ok { - if hr.Healthy() { - live++ - } - } else { - live++ - } - } - return live, total -} - -// LastHead returns the most recent delivered head for the network and when -// it was delivered. ok is false when the network is unknown or no head has -// been delivered yet. -func (i *Indexer) LastHead(networkId string) (block BlockRef, at time.Time, ok bool) { - nsRaw, found := i.networks.Load(networkId) - if !found { - return BlockRef{}, time.Time{}, false - } - ns := nsRaw.(*networkState) - nanos := ns.lastHeadAt.Load() - if nanos == 0 { - return BlockRef{}, time.Time{}, false - } - if head := ns.lastHead.Load(); head != nil { - block = BlockRef{Number: head.num, Hash: head.hash} - } - return block, time.Unix(0, nanos), true -} - // fanOut dispatches to every registered egress whose InterestedIn matches. func (i *Indexer) fanOut(ev IndexedEvent) { i.egresses.Range(func(_, v any) bool { diff --git a/telemetry/metrics.go b/telemetry/metrics.go index 7fb347190..e7e05f56e 100644 --- a/telemetry/metrics.go +++ b/telemetry/metrics.go @@ -88,12 +88,6 @@ var ( Help: "Whether the upstream WebSocket connection is currently established (1) or down/wedged (0).", }, []string{"project", "vendor", "network", "upstream"}) - MetricNetworkSubscriptionLastHeadTimestamp = promauto.NewGaugeVec(prometheus.GaugeOpts{ - Namespace: "erpc", - Name: "network_subscription_last_head_timestamp_seconds", - Help: "Unix timestamp of the last newHeads event delivered by the subscription indexer for a network. Alert on time() - this > N to catch silent head stalls.", - }, []string{"network"}) - MetricUpstreamCordoned = promauto.NewGaugeVec(prometheus.GaugeOpts{ Namespace: "erpc", Name: "upstream_cordoned", From ccfbfdaafa9120e661243b45d1586390b739226a Mon Sep 17 00:00:00 2001 From: snowkide Date: Wed, 24 Jun 2026 18:49:08 +0200 Subject: [PATCH 13/40] fix(failsafe): release half-open permit on ignored outcome (breaker wedge) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Root cause of the internal-eRPC "probe/selection wedge" (incident 2026-06-24). Breaker.Record returned early for OutcomeIgnore BEFORE the HalfOpen branch that releases the trial permit (halfOpenInflight--). A HalfOpen trial reserves a permit in TryAcquirePermit; if that trial resolves as ignorable (timeout, cancellation, soft error — precisely what a transient redis/upstream blip produces), the permit was never released. After enough such trials halfOpenInflight saturates the trial capacity, every subsequent TryAcquirePermit in HalfOpen is denied, and the breaker wedges open indefinitely — failing real traffic AND the selection-recovery probes (which are breaker-eligible). The upstream can never re-admit, so eRPC serves no healthy upstream for the chain until the pods are rollout-restarted (which resets in-memory breaker state). Evidence: during a wedge the selection-probe error RATIO sits at ~1.0 (every probe denied) sustained for tens of minutes across multiple chains at once, recovering within minutes of a restart; onset correlates with redis-haproxy churn. Fix: on OutcomeIgnore, still release a reserved HalfOpen trial permit (without counting it as success/failure — the trial was inconclusive). Minimal, behaviour- preserving for Closed/Open. Test: breaker_test.go reproduces the leak — without the fix the breaker "wedges after 0 ignored trials" (halfOpenInflight leaks to 1 and TryAcquirePermit denies); with the fix, repeated ignored trials never wedge and a later success still closes. Follow-ups (separate): mark selection-recovery probes breaker-ineligible so a genuinely-open breaker can't blind its own recovery probe; bounded HalfOpen dwell / cordon TTL as defence-in-depth. Co-Authored-By: Claude Opus 4.8 (1M context) --- failsafe/breaker.go | 21 ++++++++++- failsafe/breaker_test.go | 80 ++++++++++++++++++++++++++++++++++++++++ 2 files changed, 99 insertions(+), 2 deletions(-) create mode 100644 failsafe/breaker_test.go diff --git a/failsafe/breaker.go b/failsafe/breaker.go index e17b182ff..0e802bfa9 100644 --- a/failsafe/breaker.go +++ b/failsafe/breaker.go @@ -176,9 +176,26 @@ func (b *Breaker) TryAcquirePermit() bool { } // Record applies the given outcome to the breaker's state machine. -// OutcomeIgnore is a no-op. +// OutcomeIgnore does not move the success/failure counters, but it MUST still +// release a HalfOpen trial permit that TryAcquirePermit reserved — otherwise +// halfOpenInflight leaks. Once it saturates the trial capacity, every +// TryAcquirePermit in HalfOpen is denied and the breaker wedges open +// indefinitely, failing real traffic AND the selection-recovery probes until +// the process is restarted. Ignorable outcomes (timeouts, cancellations, soft +// errors) are exactly what a transient dependency blip produces during a trial, +// so this leak is the production "probe/selection wedge" failure mode. func (b *Breaker) Record(o Outcome) { - if b == nil || o == OutcomeIgnore { + if b == nil { + return + } + if o == OutcomeIgnore { + b.mu.Lock() + if State(b.state.Load()) == StateHalfOpen && b.halfOpenInflight > 0 { + // Release the reserved trial permit without counting the outcome as + // a success or failure — the trial was inconclusive, not passed. + b.halfOpenInflight-- + } + b.mu.Unlock() return } b.mu.Lock() diff --git a/failsafe/breaker_test.go b/failsafe/breaker_test.go new file mode 100644 index 000000000..e1020300a --- /dev/null +++ b/failsafe/breaker_test.go @@ -0,0 +1,80 @@ +package failsafe + +import ( + "testing" + "time" + + "github.com/erpc/erpc/common" + "github.com/rs/zerolog" + "github.com/stretchr/testify/require" +) + +func newWedgeTestBreaker() *Breaker { + cfg := &common.CircuitBreakerPolicyConfig{ + FailureThresholdCount: 1, + FailureThresholdCapacity: 1, + SuccessThresholdCount: 1, + SuccessThresholdCapacity: 1, + HalfOpenAfter: common.Duration(1 * time.Millisecond), + } + lg := zerolog.Nop() + return NewBreaker(cfg, &lg) +} + +func openThenHalfOpen(t *testing.T, b *Breaker) { + t.Helper() + require.True(t, b.TryAcquirePermit()) + b.Record(OutcomeFailure) // one failure in Closed opens (threshold 1) + require.Equal(t, StateOpen, State(b.state.Load()), "breaker should be open after a failure") + time.Sleep(5 * time.Millisecond) // exceed HalfOpenAfter + require.True(t, b.TryAcquirePermit(), "first half-open trial permit must be granted") + require.Equal(t, StateHalfOpen, State(b.state.Load())) +} + +// TestBreaker_HalfOpenPermitReleasedOnIgnoredOutcome is the regression for the +// production probe/selection wedge (incident 2026-06-24): a HalfOpen trial that +// returns an IGNORABLE outcome (timeout / cancellation / soft error) must +// release the trial permit it reserved in TryAcquirePermit. If it doesn't, +// halfOpenInflight leaks, saturates the trial capacity, and TryAcquirePermit +// denies every subsequent caller — wedging the breaker open and failing both +// real traffic AND the selection-recovery probes until a process restart. +func TestBreaker_HalfOpenPermitReleasedOnIgnoredOutcome(t *testing.T) { + b := newWedgeTestBreaker() + openThenHalfOpen(t, b) + + // The trial is inconclusive (e.g. the request was canceled or timed out — + // exactly what a transient redis/upstream blip produces). + b.Record(OutcomeIgnore) + + b.mu.Lock() + inflight, hoSucc, hoFail := b.halfOpenInflight, b.halfOpenSuccess, b.halfOpenFailure + b.mu.Unlock() + + require.Equal(t, 0, inflight, "ignored half-open outcome must release the trial permit (no leak)") + require.Equal(t, 0, hoSucc, "ignored outcome must not count as a half-open success") + require.Equal(t, 0, hoFail, "ignored outcome must not count as a half-open failure") + require.True(t, b.TryAcquirePermit(), + "breaker must not be wedged: a fresh trial permit must be grantable after an ignored outcome") +} + +// TestBreaker_RepeatedIgnoredTrialsDoNotWedge proves the breaker keeps offering +// trial permits across many inconclusive trials, and a real success still closes +// it. Before the fix this Fatals on the first iteration (permit leaked → denied). +func TestBreaker_RepeatedIgnoredTrialsDoNotWedge(t *testing.T) { + b := newWedgeTestBreaker() + openThenHalfOpen(t, b) + // First reserved permit (from openThenHalfOpen) resolves as ignored. + b.Record(OutcomeIgnore) + + for i := 0; i < 50; i++ { + if !b.TryAcquirePermit() { + t.Fatalf("breaker wedged after %d ignored trials: half-open permit leaked", i) + } + b.Record(OutcomeIgnore) + } + + require.True(t, b.TryAcquirePermit(), "trial permit still grantable") + b.Record(OutcomeSuccess) + require.Equal(t, StateClosed, State(b.state.Load()), + "a successful trial must still close the breaker after a run of ignored trials") +} From cbbb538064969266584e62198ec45fe990a5a62f Mon Sep 17 00:00:00 2001 From: shpookas Date: Mon, 20 Jul 2026 12:00:47 +0200 Subject: [PATCH 14/40] fix(initializer): keep retrying recoverable tasks when a sibling task is fatal (backport erpc#973) Backports upstream erpc commit a04655e6 (PR #973) verbatim, adapted only for this fork's attemptRemainingTasks signature. The auto-retry loop exited permanently when State() returned Fatal, which happens when ANY single task is fatal (e.g. one misconfigured upstream with a chainId mismatch). Every transiently-failed sibling task was then abandoned until process restart. In production this meant an upstream that blipped during a daemonset rollout (robinhood-mainnet) stayed unregistered (network n/a) for two days while all its traffic escaped to a paid 3P fallback. The loop now stops only when every task is terminal (succeeded or fatal); a fatal task is terminal for itself only. --- util/initializer.go | 45 ++++++++++++++++------ util/initializer_test.go | 80 ++++++++++++++++++++++++++++++++++++++++ 2 files changed, 114 insertions(+), 11 deletions(-) diff --git a/util/initializer.go b/util/initializer.go index 327aa2f3c..52e717146 100644 --- a/util/initializer.go +++ b/util/initializer.go @@ -554,16 +554,39 @@ func (i *Initializer) ensureAutoRetryIfEnabled() { }() } -// Continually attempt tasks until all succeed or context is canceled +// hasPendingWork reports whether any registered task is still in a non-terminal +// state (pending, running, failed, or timed-out) and could therefore benefit +// from another attempt. Only succeeded and fatal tasks are terminal. +// +// This is the auto-retry loop's stop condition. It deliberately does NOT key off +// State(): State() returns StateFatal as soon as ANY task is fatal, so keying +// the loop off it makes a single permanently-failing task (e.g. an upstream with +// a chainId mismatch) end auto-retry for every sibling task in the same +// initializer. Because one Initializer is shared across many independent +// resources (one bootstrap task per network/upstream), that stranded every +// not-yet-initialized resource until the process restarted. +func (i *Initializer) hasPendingWork() bool { + pending := false + i.tasks.Range(func(_, value interface{}) bool { + switch TaskState(value.(*BootstrapTask).state.Load()) { + case TaskPending, TaskRunning, TaskFailed, TaskTimedOut: + pending = true + return false // found one; stop iterating + } + return true + }) + return pending +} + +// Continually attempt tasks until every task is terminal (succeeded or fatal) +// or the context is canceled. func (i *Initializer) autoRetryLoop(ctx context.Context) { if cancel := i.cancelAutoRetry.Load(); cancel != nil { defer cancel.(context.CancelFunc)() } - if i.State() == StateReady { - i.autoRetryActive.Store(false) - return - } - if i.State() == StateFatal { + // Nothing to retry once every task is terminal. A fatal task must not end + // the loop on its own — recoverable siblings must keep retrying. + if !i.hasPendingWork() { i.autoRetryActive.Store(false) return } @@ -581,11 +604,11 @@ func (i *Initializer) autoRetryLoop(ctx context.Context) { i.attemptRemainingTasks() err := i.WaitForTasks(ctx) state := i.State() - if state == StateFatal { - i.autoRetryActive.Store(false) - return - } - if err == nil && state == StateReady { + // Stop only once no task can benefit from another attempt (every task + // succeeded or is fatal). Fatal tasks are skipped by + // attemptRemainingTasks, so a permanently-failing task cannot wedge the + // retries of its still-recoverable siblings. + if !i.hasPendingWork() { i.autoRetryActive.Store(false) return } diff --git a/util/initializer_test.go b/util/initializer_test.go index 202b05635..ef6ed6526 100644 --- a/util/initializer_test.go +++ b/util/initializer_test.go @@ -6,6 +6,7 @@ import ( "fmt" "runtime" "sync" + "sync/atomic" "testing" "time" @@ -791,3 +792,82 @@ func BenchmarkInitializer_RangeTaskStates_vs_Status(b *testing.B) { } }) } + +// Regression (backport of upstream erpc#973): a single fatal task must not +// stop the auto-retry loop for other, transiently-failing tasks. One +// misconfigured upstream used to permanently disable bootstrap retries for +// the whole registry. +func TestInitializer_FatalTaskDoesNotStopRetryOfOthers(t *testing.T) { + conf := &InitializerConfig{ + TaskTimeout: time.Second, + AutoRetry: true, + RetryMinDelay: time.Millisecond * 10, + RetryMaxDelay: time.Millisecond * 20, + RetryFactor: 1.2, + } + appCtx, cancel := context.WithCancel(context.Background()) + defer cancel() + init := setupInitializer(t, appCtx, conf) + + fatalTask := NewBootstrapTask("fatal-task", func(ctx context.Context) error { + return &testFatalError{errors.New("permanent misconfiguration")} + }) + + var attempts atomic.Int32 + transientTask := NewBootstrapTask("transient-task", func(ctx context.Context) error { + if attempts.Add(1) < 3 { + return errors.New("transient failure") + } + return nil + }) + + _ = init.ExecuteTasks(appCtx, fatalTask, transientTask) + + require.Eventually(t, func() bool { + return TaskState(transientTask.state.Load()) == TaskSucceeded + }, 5*time.Second, 20*time.Millisecond, "transient task should eventually succeed despite the fatal sibling") + + assert.Equal(t, TaskFatal, TaskState(fatalTask.state.Load())) + init.Stop(nil) +} + +// Regression: tasks scheduled after all earlier tasks succeeded must still +// be retried (the loop must not have wound down for good). +func TestInitializer_RetryLoopRestartsForNewFailedTasks(t *testing.T) { + conf := &InitializerConfig{ + TaskTimeout: time.Second, + AutoRetry: true, + RetryMinDelay: time.Millisecond * 10, + RetryMaxDelay: time.Millisecond * 20, + RetryFactor: 1.2, + } + appCtx, cancel := context.WithCancel(context.Background()) + defer cancel() + init := setupInitializer(t, appCtx, conf) + + ok := NewBootstrapTask("ok-task", func(ctx context.Context) error { return nil }) + _ = init.ExecuteTasks(appCtx, ok) + require.Eventually(t, func() bool { + return TaskState(ok.state.Load()) == TaskSucceeded + }, 5*time.Second, 10*time.Millisecond, "first task should succeed") + + var attempts atomic.Int32 + transientTask := NewBootstrapTask("late-transient-task", func(ctx context.Context) error { + if attempts.Add(1) < 3 { + return errors.New("transient failure") + } + return nil + }) + _ = init.ExecuteTasks(appCtx, transientTask) + + require.Eventually(t, func() bool { + return TaskState(transientTask.state.Load()) == TaskSucceeded + }, 5*time.Second, 20*time.Millisecond, "late task should be retried by a (re)started loop") + + init.Stop(nil) +} + +type testFatalError struct{ error } + +func (e *testFatalError) IsTaskFatal() bool { return true } +func (e *testFatalError) Unwrap() error { return e.error } From 95719d0ddd3e93dd9f4524541673af6660ebed1f Mon Sep 17 00:00:00 2001 From: shpookas Date: Mon, 20 Jul 2026 12:01:21 +0200 Subject: [PATCH 15/40] fix(initializer): bound each auto-retry round by TaskTimeout (hung task defence) Upstream (erpc#973) still waits unbounded on WaitForTasks each retry round: one task hung inside its Fn (e.g. a client dial that ignores ctx and never returns) stays Running forever and blocks the retry loop for every other task. Bound each round's wait by TaskTimeout so a hung task only delays a round, never stops retries. Candidate for upstreaming. --- util/initializer.go | 8 +++++++- util/initializer_test.go | 38 ++++++++++++++++++++++++++++++++++++++ 2 files changed, 45 insertions(+), 1 deletion(-) diff --git a/util/initializer.go b/util/initializer.go index 52e717146..0c41fd013 100644 --- a/util/initializer.go +++ b/util/initializer.go @@ -602,7 +602,13 @@ func (i *Initializer) autoRetryLoop(ctx context.Context) { } i.attempts.Add(1) i.attemptRemainingTasks() - err := i.WaitForTasks(ctx) + // Bounded wait: a task hung inside its Fn (e.g. a client dial that + // ignores ctx and never returns) stays Running forever; an unbounded + // WaitForTasks would then block this loop and stop retries of every + // other task. + waitCtx, waitCancel := context.WithTimeout(ctx, i.conf.TaskTimeout) + err := i.WaitForTasks(waitCtx) + waitCancel() state := i.State() // Stop only once no task can benefit from another attempt (every task // succeeded or is fatal). Fatal tasks are skipped by diff --git a/util/initializer_test.go b/util/initializer_test.go index ef6ed6526..4465e658c 100644 --- a/util/initializer_test.go +++ b/util/initializer_test.go @@ -867,6 +867,44 @@ func TestInitializer_RetryLoopRestartsForNewFailedTasks(t *testing.T) { init.Stop(nil) } +// Regression: a task whose Fn hangs (ignores ctx) must not block the +// auto-retry loop for other, transiently-failing tasks. +func TestInitializer_HungTaskDoesNotBlockRetryOfOthers(t *testing.T) { + conf := &InitializerConfig{ + TaskTimeout: time.Millisecond * 200, + AutoRetry: true, + RetryMinDelay: time.Millisecond * 10, + RetryMaxDelay: time.Millisecond * 20, + RetryFactor: 1.2, + } + appCtx, cancel := context.WithCancel(context.Background()) + defer cancel() + init := setupInitializer(t, appCtx, conf) + + hungRelease := make(chan struct{}) + hungTask := NewBootstrapTask("hung-task", func(ctx context.Context) error { + <-hungRelease // ignores ctx: simulates a client dial with no deadline + return nil + }) + + var attempts atomic.Int32 + transientTask := NewBootstrapTask("transient-task", func(ctx context.Context) error { + if attempts.Add(1) < 3 { + return errors.New("transient failure") + } + return nil + }) + + go func() { _ = init.ExecuteTasks(appCtx, hungTask, transientTask) }() + + require.Eventually(t, func() bool { + return TaskState(transientTask.state.Load()) == TaskSucceeded + }, 5*time.Second, 20*time.Millisecond, "transient task should eventually succeed despite the hung sibling") + + close(hungRelease) + init.Stop(nil) +} + type testFatalError struct{ error } func (e *testFatalError) IsTaskFatal() bool { return true } From 42f6842572f5677807ef645c604ad98d2aa59946 Mon Sep 17 00:00:00 2001 From: shpookas Date: Mon, 20 Jul 2026 12:01:45 +0200 Subject: [PATCH 16/40] fix(initializer): stop the retry loop before taking tasksMu in Stop (lock-then-wait deadlock) Stop() held tasksMu while waiting for the auto-retry goroutine, which itself acquires tasksMu inside attemptRemainingTasks. If the loop was blocked acquiring the mutex when the cancel landed, Stop waited forever. Cancel-and-wait now happens before taking the mutex. Present upstream as well; candidate for upstreaming. --- util/initializer.go | 12 +++++++----- 1 file changed, 7 insertions(+), 5 deletions(-) diff --git a/util/initializer.go b/util/initializer.go index 0c41fd013..a34f79220 100644 --- a/util/initializer.go +++ b/util/initializer.go @@ -435,16 +435,18 @@ func (i *Initializer) MarkTaskAsFailed(name string, err error) { func (i *Initializer) Stop(destroyFn func() error) error { i.logger.Debug().Msg("stopping initializer") - i.tasksMu.Lock() - defer i.tasksMu.Unlock() - + // Cancel the auto-retry loop and wait for it to exit BEFORE taking + // tasksMu: the loop acquires tasksMu inside attemptRemainingTasks, so + // holding the mutex while waiting for the goroutine can deadlock if the + // loop is blocked on the mutex when the cancel lands. if cancel := i.cancelAutoRetry.Load(); cancel != nil { cancel.(context.CancelFunc)() } - - // Wait for auto-retry goroutine to finish i.autoRetryWg.Wait() + i.tasksMu.Lock() + defer i.tasksMu.Unlock() + // Now, wait for any tasks that might still be running to finish or fail. waitCtx, waitCancel := context.WithTimeout(i.appCtx, i.conf.TaskTimeout+100*time.Millisecond) defer waitCancel() From 1be9510fcbfb68389111afedfeb7e6bf14ec660c Mon Sep 17 00:00:00 2001 From: Jonny Date: Wed, 22 Jul 2026 12:58:10 +0100 Subject: [PATCH 17/40] fix(ws): advance network latest tip before newHeads fan-out Prevent HTTP "latest"/eth_blockNumber from regressing below a head already delivered on WS, which trips Chainlink MultiNode FinalizedBlockOutOfSync. Co-authored-by: Cursor --- erpc/networks.go | 38 +++++++ erpc/networks_ws_tip_test.go | 193 +++++++++++++++++++++++++++++++++++ erpc/subscription_manager.go | 18 ++-- indexer/ingress.go | 9 +- 4 files changed, 248 insertions(+), 10 deletions(-) create mode 100644 erpc/networks_ws_tip_test.go diff --git a/erpc/networks.go b/erpc/networks.go index 3101ef681..498712997 100644 --- a/erpc/networks.go +++ b/erpc/networks.go @@ -62,10 +62,48 @@ type Network struct { // value we WARN with the local/shared inputs so we can see whether the // regression originated in the per-upstream poller max, the cross-cluster // shared counter, or a race between them. + // + // lastReturnedLatestBlock is ALSO advanced by NoteObservedLatestBlock when + // a WS newHeads tip is about to be fan-out to clients — so HTTP "latest" + // cannot regress below a head we have already delivered on the same pod. lastReturnedLatestBlock atomic.Int64 lastReturnedFinalizedBlock atomic.Int64 } +// NoteObservedLatestBlock records that this Network has observed head +// blockNumber and is about to (or has) delivered it to clients via WS +// newHeads fan-out. It advances the cross-pod network latest counter and the +// process-local high-water mark used by EvmHighestLatestBlockNumber. +// +// Callers MUST invoke this before delivering the corresponding newHeads +// notification to any client. Otherwise a concurrent HTTP +// eth_getBlockByNumber("latest") / eth_blockNumber can race and return a +// lower tip — Chainlink MultiNode treats that 1-block regression as +// FinalizedBlockOutOfSync ("No live RPC nodes available"). +func (n *Network) NoteObservedLatestBlock(ctx context.Context, blockNumber int64) { + if n == nil || blockNumber <= 0 { + return + } + if ctx == nil { + ctx = n.appCtx + } + if ctx == nil { + ctx = context.Background() + } + if n.latestBlockShared != nil { + n.latestBlockShared.TryUpdate(ctx, blockNumber) + } + for { + cur := n.lastReturnedLatestBlock.Load() + if blockNumber <= cur { + return + } + if n.lastReturnedLatestBlock.CompareAndSwap(cur, blockNumber) { + return + } + } +} + // Bootstrap registers this network with the policy engine. The engine kicks // off the slot's ticker and runs an initial synchronous eval so request-path // reads through `policyEngine.GetOrdered` always see a populated cache. diff --git a/erpc/networks_ws_tip_test.go b/erpc/networks_ws_tip_test.go new file mode 100644 index 000000000..dc50084cc --- /dev/null +++ b/erpc/networks_ws_tip_test.go @@ -0,0 +1,193 @@ +package erpc + +import ( + "context" + "net/http" + "strings" + "testing" + "time" + + "github.com/erpc/erpc/common" + "github.com/erpc/erpc/data" + "github.com/erpc/erpc/health" + "github.com/erpc/erpc/thirdparty" + "github.com/erpc/erpc/upstream" + "github.com/erpc/erpc/util" + "github.com/h2non/gock" + "github.com/rs/zerolog/log" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +func init() { + util.ConfigureTestLogger() +} + +// Regression for Chainlink MultiNode FinalizedBlockOutOfSync flaps: +// once a WS newHeads tip N is observed (and about to be fan-out), HTTP +// tip resolution via EvmHighestLatestBlockNumber must not return < N — +// even if every local poller still reports N-1. +func TestNoteObservedLatestBlock_FloorsEvmHighestLatest(t *testing.T) { + util.ResetGock() + defer util.ResetGock() + util.SetupMocksForEvmStatePoller() + + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + + up := &common.UpstreamConfig{ + Type: common.UpstreamTypeEvm, + Id: "rpc1", + Endpoint: "http://rpc1.localhost", + Evm: &common.EvmUpstreamConfig{ChainId: 123}, + } + + gock.New("http://rpc1.localhost"). + Post(""). + Persist(). + Filter(func(r *http.Request) bool { + return strings.Contains(util.SafeReadBody(r), `eth_chainId`) + }). + Reply(200). + JSON([]byte(`{"result":"0x7b"}`)) + + rateLimitersRegistry, _ := upstream.NewRateLimitersRegistry(context.Background(), &common.RateLimiterConfig{}, &log.Logger) + metricsTracker := health.NewTracker(&log.Logger, "test", time.Minute) + + vr := thirdparty.NewVendorsRegistry() + pr, err := thirdparty.NewProvidersRegistry(&log.Logger, vr, []*common.ProviderConfig{}, nil) + require.NoError(t, err) + + ssr, err := data.NewSharedStateRegistry(ctx, &log.Logger, &common.SharedStateConfig{ + Connector: &common.ConnectorConfig{ + Driver: "memory", + Memory: &common.MemoryConnectorConfig{MaxItems: 100_000, MaxTotalSize: "1GB"}, + }, + }) + require.NoError(t, err) + + upstreamsRegistry := upstream.NewUpstreamsRegistry( + ctx, &log.Logger, "test", + []*common.UpstreamConfig{up}, ssr, rateLimitersRegistry, vr, pr, nil, + metricsTracker, nil, + ) + + networkConfig := &common.NetworkConfig{ + Architecture: common.ArchitectureEvm, + Evm: &common.EvmNetworkConfig{ChainId: 123}, + } + network, err := NewNetwork(ctx, &log.Logger, "test", networkConfig, + rateLimitersRegistry, upstreamsRegistry, metricsTracker, nil) + require.NoError(t, err) + + upstreamsRegistry.Bootstrap(ctx) + time.Sleep(200 * time.Millisecond) + require.NoError(t, upstreamsRegistry.GetInitializer().WaitForTasks(ctx)) + require.NoError(t, network.Bootstrap(ctx)) + time.Sleep(250 * time.Millisecond) + + upsList := upstreamsRegistry.GetNetworkUpstreams(ctx, util.EvmNetworkId(123)) + require.Len(t, upsList, 1) + u := upsList[0] + + u.EvmStatePoller().SuggestLatestBlock(1000) + time.Sleep(50 * time.Millisecond) + require.Equal(t, int64(1000), network.EvmHighestLatestBlockNumber(ctx)) + + // Simulate the WS ingest path: a head arrives that we are about to + // fan-out, but the lagging HTTP poller view is still at 1000. + network.NoteObservedLatestBlock(ctx, 1001) + + got := network.EvmHighestLatestBlockNumber(ctx) + assert.Equal(t, int64(1001), got, + "after WS tip observation, highest latest must be ≥ delivered head") + + // Even if the network shared counter is somehow still behind (or a + // local aggregator race computes 1000), the process-local high-water + // mark from NoteObservedLatestBlock must clamp the return. + require.NotNil(t, network.latestBlockShared) + // Shared already at 1001 from NoteObserved; verify lastReturned alone + // is enough by calling apply path with a lower computed tip via the + // monotonic guard — EvmHighest after noting must never go backwards. + assert.GreaterOrEqual(t, network.lastReturnedLatestBlock.Load(), int64(1001)) + assert.Equal(t, int64(1001), network.EvmHighestLatestBlockNumber(ctx)) +} + +// End-to-end through networkHandle.SuggestLatestBlock — the Indexer hook +// that runs before fan-out. +func TestNetworkHandle_SuggestLatestBlock_AdvancesNetworkTipBeforeFanOut(t *testing.T) { + util.ResetGock() + defer util.ResetGock() + util.SetupMocksForEvmStatePoller() + + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + + up := &common.UpstreamConfig{ + Type: common.UpstreamTypeEvm, + Id: "bor-1", + Endpoint: "http://bor1.localhost", + Evm: &common.EvmUpstreamConfig{ChainId: 123}, + } + + gock.New("http://bor1.localhost"). + Post(""). + Persist(). + Filter(func(r *http.Request) bool { + return strings.Contains(util.SafeReadBody(r), `eth_chainId`) + }). + Reply(200). + JSON([]byte(`{"result":"0x7b"}`)) + + rateLimitersRegistry, _ := upstream.NewRateLimitersRegistry(context.Background(), &common.RateLimiterConfig{}, &log.Logger) + metricsTracker := health.NewTracker(&log.Logger, "test", time.Minute) + + vr := thirdparty.NewVendorsRegistry() + pr, err := thirdparty.NewProvidersRegistry(&log.Logger, vr, []*common.ProviderConfig{}, nil) + require.NoError(t, err) + + ssr, err := data.NewSharedStateRegistry(ctx, &log.Logger, &common.SharedStateConfig{ + Connector: &common.ConnectorConfig{ + Driver: "memory", + Memory: &common.MemoryConnectorConfig{MaxItems: 100_000, MaxTotalSize: "1GB"}, + }, + }) + require.NoError(t, err) + + upstreamsRegistry := upstream.NewUpstreamsRegistry( + ctx, &log.Logger, "test", + []*common.UpstreamConfig{up}, ssr, rateLimitersRegistry, vr, pr, nil, + metricsTracker, nil, + ) + + networkConfig := &common.NetworkConfig{ + Architecture: common.ArchitectureEvm, + Evm: &common.EvmNetworkConfig{ChainId: 123}, + } + network, err := NewNetwork(ctx, &log.Logger, "test", networkConfig, + rateLimitersRegistry, upstreamsRegistry, metricsTracker, nil) + require.NoError(t, err) + + upstreamsRegistry.Bootstrap(ctx) + time.Sleep(200 * time.Millisecond) + require.NoError(t, upstreamsRegistry.GetInitializer().WaitForTasks(ctx)) + require.NoError(t, network.Bootstrap(ctx)) + time.Sleep(250 * time.Millisecond) + + upsList := upstreamsRegistry.GetNetworkUpstreams(ctx, util.EvmNetworkId(123)) + require.Len(t, upsList, 1) + upsList[0].EvmStatePoller().SuggestLatestBlock(90677358) + time.Sleep(50 * time.Millisecond) + require.Equal(t, int64(90677358), network.EvmHighestLatestBlockNumber(ctx)) + + handle := &networkHandle{nw: network} + // Mirrors indexer.Ingest ordering: SuggestLatestBlock then fan-out. + handle.SuggestLatestBlock("ws:bor-1", 90677359) + + assert.Equal(t, int64(90677359), upsList[0].EvmStatePoller().LatestBlock(), + "per-upstream poller must advance") + assert.Equal(t, int64(90677359), network.EvmHighestLatestBlockNumber(ctx), + "network tip must advance before any client would see the WS head") + assert.GreaterOrEqual(t, network.lastReturnedLatestBlock.Load(), int64(90677359), + "process-local high-water mark must cover the delivered WS tip") +} diff --git a/erpc/subscription_manager.go b/erpc/subscription_manager.go index 54b78f152..1c9823a04 100644 --- a/erpc/subscription_manager.go +++ b/erpc/subscription_manager.go @@ -569,8 +569,13 @@ func (h *networkHandle) FinalityDepth() int64 { } // SuggestLatestBlock routes a per-source block observation to the -// upstream's state poller. sourceId is the ingress adapter's Name(), -// which for wsupstream.Adapter is "ws:". +// upstream's state poller, then advances the network-level latest tip. +// sourceId is the ingress adapter's Name(), which for wsupstream.Adapter +// is "ws:". +// +// Ordering matters: Indexer.Ingest calls this BEFORE fan-out, so by the +// time any client sees head N on WS, EvmHighestLatestBlockNumber on this +// pod is already ≥ N (see Network.NoteObservedLatestBlock). func (h *networkHandle) SuggestLatestBlock(sourceId string, blockNumber int64) { const prefix = "ws:" if !strings.HasPrefix(sourceId, prefix) { @@ -582,12 +587,12 @@ func (h *networkHandle) SuggestLatestBlock(sourceId string, blockNumber int64) { continue } poller := u.EvmStatePoller() - if poller == nil || poller.IsObjectNull() { - return + if poller != nil && !poller.IsObjectNull() { + poller.SuggestLatestBlock(blockNumber) } - poller.SuggestLatestBlock(blockNumber) - return + break } + h.nw.NoteObservedLatestBlock(h.nw.appCtx, blockNumber) } // Interface checks: fail the build if either contract drifts. @@ -595,4 +600,3 @@ var ( _ wsclient.NotificationWriter = (*WsConnection)(nil) _ indexer.NetworkHandle = (*networkHandle)(nil) ) - diff --git a/indexer/ingress.go b/indexer/ingress.go index 8026b7675..d33f57705 100644 --- a/indexer/ingress.go +++ b/indexer/ingress.go @@ -25,9 +25,12 @@ type NetworkHandle interface { // indexer when tagging IndexedEvent.Lifecycle. FinalityDepth() int64 // SuggestLatestBlock advances the per-source latest-block tracker - // before the indexer dedupes. Preserving "update-before-dedup" - // ordering is critical — the state poller needs to see every - // observation, even ones we'll drop in the fan-out stage. + // (and the network-level latest tip) before the indexer dedupes / + // fans out. Preserving "update-before-dedup" and + // "tip-before-fanout" ordering is critical — the state poller needs + // to see every observation (even ones we drop), and HTTP "latest" + // must not be allowed to regress below a head we are about to + // deliver on WS. SuggestLatestBlock(sourceId string, blockNumber int64) } From 0d8003e8a27e93e1aa9d5bb65b078c423993921c Mon Sep 17 00:00:00 2001 From: Jonny Date: Wed, 22 Jul 2026 12:59:13 +0100 Subject: [PATCH 18/40] docs: drop consumer-specific wording from WS tip comments Co-authored-by: Cursor --- erpc/networks.go | 3 +-- erpc/networks_ws_tip_test.go | 3 +-- 2 files changed, 2 insertions(+), 4 deletions(-) diff --git a/erpc/networks.go b/erpc/networks.go index 498712997..6a88d3241 100644 --- a/erpc/networks.go +++ b/erpc/networks.go @@ -78,8 +78,7 @@ type Network struct { // Callers MUST invoke this before delivering the corresponding newHeads // notification to any client. Otherwise a concurrent HTTP // eth_getBlockByNumber("latest") / eth_blockNumber can race and return a -// lower tip — Chainlink MultiNode treats that 1-block regression as -// FinalizedBlockOutOfSync ("No live RPC nodes available"). +// lower tip than a head already (or about to be) served on WS. func (n *Network) NoteObservedLatestBlock(ctx context.Context, blockNumber int64) { if n == nil || blockNumber <= 0 { return diff --git a/erpc/networks_ws_tip_test.go b/erpc/networks_ws_tip_test.go index dc50084cc..d4cf18009 100644 --- a/erpc/networks_ws_tip_test.go +++ b/erpc/networks_ws_tip_test.go @@ -23,8 +23,7 @@ func init() { util.ConfigureTestLogger() } -// Regression for Chainlink MultiNode FinalizedBlockOutOfSync flaps: -// once a WS newHeads tip N is observed (and about to be fan-out), HTTP +// Once a WS newHeads tip N is observed (and about to be fan-out), HTTP // tip resolution via EvmHighestLatestBlockNumber must not return < N — // even if every local poller still reports N-1. func TestNoteObservedLatestBlock_FloorsEvmHighestLatest(t *testing.T) { From c1c3404419379e06fb3f7147b67ad40230d53aa2 Mon Sep 17 00:00:00 2001 From: Jonny Date: Wed, 22 Jul 2026 13:11:13 +0100 Subject: [PATCH 19/40] fix(initializer): reap hung tasks, fix State/Wait tip races MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Reap Running tasks past TaskTimeout to TimedOut so ignore-ctx Fns cannot wedge hasPendingWork forever; completion uses attempt-ID + CAS so a late return cannot clobber a newer retry. - State(): one fatal sibling no longer maps the whole Initializer to Fatal while others succeeded/recover — mix is Partial; all-fatal stays Fatal. - waitForTasks waits in parallel and distinguishes wait-context abort from a task that already finished as TimedOut; Wait surfaces TimedOut/Fatal errors via lastErr. Co-authored-by: Cursor --- util/initializer.go | 217 ++++++++++++++++++++++++++++----------- util/initializer_test.go | 103 ++++++++++++++++++- 2 files changed, 256 insertions(+), 64 deletions(-) diff --git a/util/initializer.go b/util/initializer.go index a34f79220..41870027e 100644 --- a/util/initializer.go +++ b/util/initializer.go @@ -128,15 +128,11 @@ func (t *BootstrapTask) Wait(ctx context.Context) error { case <-ctx.Done(): return ctx.Err() case <-ch.(chan struct{}): - // The attempt ended. Check if we failed. - if TaskState(t.state.Load()) == TaskFailed { - wr, _ := t.lastErr.Load().(wrappedError) - if wr.err == nil { - t.lastErr.Store(wrappedError{err: errors.New("task failed without specific error")}) - } - return wr.err - } - return nil // Succeeded or otherwise finished + // Attempt ended — loop back to the terminal-state check so + // TimedOut/Fatal/Failed all surface lastErr consistently + // (previously only TaskFailed was handled here, so a deadline + // TimedOut incorrectly returned nil). + continue } } } @@ -224,23 +220,53 @@ func (i *Initializer) WaitForTasks(ctx context.Context) error { } // Wait for a set of tasks to complete or ctx to expire. +// +// Waits run in parallel so one slow/hung task cannot serialize the wait +// budget across siblings (previously a single hung task at the front of +// sync.Map iteration burned the whole timeout before others were observed). func (i *Initializer) waitForTasks(ctx context.Context, tasks ...*BootstrapTask) error { - var errs []error + if len(tasks) == 0 { + return nil + } + + type waitResult struct { + task *BootstrapTask + err error + } + errCh := make(chan waitResult, len(tasks)) for _, task := range tasks { - if err := task.Wait(ctx); err != nil { - // If context was canceled, likely best to just return. - return err + go func(task *BootstrapTask) { + errCh <- waitResult{task: task, err: task.Wait(ctx)} + }(task) + } + + var errs []error + var ctxErr error + for range tasks { + res := <-errCh + if res.err == nil { + continue } - // If task is failed, record that error - state := TaskState(task.state.Load()) - if state == TaskFailed && task.Error() != nil { - errs = append(errs, task.Error().Err) + st := TaskState(res.task.state.Load()) + // Wait-context abort: task still in-flight when Wait returned a + // context error. A task that already finished as TimedOut also + // surfaces DeadlineExceeded via lastErr — that is a task failure. + if (errors.Is(res.err, context.Canceled) || errors.Is(res.err, context.DeadlineExceeded)) && + (st == TaskPending || st == TaskRunning) { + if ctxErr == nil { + ctxErr = res.err + } + continue } + errs = append(errs, res.err) + } + if ctxErr != nil { + return ctxErr } if len(errs) > 0 { total := len(tasks) i.logger.Warn().Errs("tasks", errs).Msgf("initialization failed: %d/%d tasks failed", len(errs), total) - return fmt.Errorf("initialization failed: %d/%d tasks failed: %v", len(errs), total, errs) + return fmt.Errorf("initialization failed: %d/%d tasks failed: %w", len(errs), total, errors.Join(errs...)) } return nil } @@ -266,21 +292,40 @@ func (i *Initializer) attemptRemainingTasks() { // #nosec G115 - We know TaskState is small enough that int->int32 won't overflow if t.state.CompareAndSwap(int32(state), int32(TaskRunning)) { t.beginAttempt() + attemptID := t.attempts.Load() t.lastErr.Store(wrappedError{err: nil}) // Create a fresh done channel to signal this attempt's completion doneCh := t.createNewDoneChannel() tasksToRun = append(tasksToRun, t) - go func(bt *BootstrapTask, doneCh chan struct{}) { - // Close the channel when the function finishes - // The CompareAndSwap will ensure we always and only close the channel once for each attempt + go func(bt *BootstrapTask, doneCh chan struct{}, attemptID int32) { + // Close the channel when the function finishes. defer close(doneCh) + // finishAttempt applies terminalState only if this goroutine + // still owns the attempt (same attemptID and still Running). + // Prevents a reaped/superseded hung Fn from clobbering a + // later retry's state. + finishAttempt := func(terminal TaskState, err error) bool { + if bt.attempts.Load() != attemptID { + return false + } + if !bt.state.CompareAndSwap(int32(TaskRunning), int32(terminal)) { + return false + } + if err != nil { + bt.lastErr.Store(wrappedError{err: err}) + } else { + bt.lastErr.Store(wrappedError{err: nil}) + } + return true + } + if i.appCtx.Err() != nil { - bt.lastErr.Store(wrappedError{err: i.appCtx.Err()}) - bt.state.Store(int32(TaskFailed)) - i.logger.Warn().Str("task", bt.Name).Err(i.appCtx.Err()).Msg("initialization task context error") + if finishAttempt(TaskFailed, i.appCtx.Err()) { + i.logger.Warn().Str("task", bt.Name).Err(i.appCtx.Err()).Msg("initialization task context error") + } return } @@ -297,36 +342,39 @@ func (i *Initializer) attemptRemainingTasks() { // Detect fatal control errors without importing the common package to avoid cycles var fatal interface{ IsTaskFatal() bool } if errors.As(err, &fatal) { - // Fatal errors should stop retries - // Unwrap underlying error if available underlying := err if uw, ok := err.(interface{ Unwrap() error }); ok && uw.Unwrap() != nil { underlying = uw.Unwrap() } - bt.lastErr.Store(wrappedError{err: underlying}) - bt.state.Store(int32(TaskFatal)) - // Log the underlying fatal error - i.logger.Error().Str("task", bt.Name).Err(underlying).Msg("initialization task fatal error") + if finishAttempt(TaskFatal, underlying) { + i.logger.Error().Str("task", bt.Name).Err(underlying).Msg("initialization task fatal error") + } return } - // If context is cancelled there will be a reason already set for it on lastErr if !errors.Is(err, context.Canceled) { if cause := context.Cause(tctx); cause != nil { err = cause } - bt.lastErr.Store(wrappedError{err: err}) } else { - bt.lastErr.CompareAndSwap(nil, wrappedError{err: err}) + // Preserve a reason already set (e.g. by reap) when canceled. + if wr, ok := bt.lastErr.Load().(wrappedError); ok && wr.err != nil { + err = wr.err + } + } + terminal := TaskFailed + if errors.Is(err, context.DeadlineExceeded) { + terminal = TaskTimedOut + } + if finishAttempt(terminal, err) { + i.logger.Warn().Str("task", bt.Name).Err(err).Str("state", terminal.String()).Msg("initialization task failed") } - bt.state.Store(int32(TaskFailed)) - i.logger.Warn().Str("task", bt.Name).Err(err).Msg("initialization task failed") } else { - bt.lastErr.Store(wrappedError{err: nil}) - bt.state.Store(int32(TaskSucceeded)) - lastAttempt, _ := bt.lastAttempt.Load().(time.Time) - i.logger.Info().Str("task", bt.Name).Dur("durationMs", time.Since(lastAttempt)).Msg("initialization task succeeded") + if finishAttempt(TaskSucceeded, nil) { + lastAttempt, _ := bt.lastAttempt.Load().(time.Time) + i.logger.Info().Str("task", bt.Name).Dur("durationMs", time.Since(lastAttempt)).Msg("initialization task succeeded") + } } - }(t, doneCh) + }(t, doneCh, attemptID) } else { wg.Done() } @@ -341,7 +389,7 @@ func (i *Initializer) attemptRemainingTasks() { } func (i *Initializer) State() InitializationState { - var total, pending, running, succeeded, failed, fatal int + var total, pending, running, succeeded, failed, timedOut, fatal int i.tasks.Range(func(key, value interface{}) bool { t := value.(*BootstrapTask) state := TaskState(t.state.Load()) @@ -354,6 +402,8 @@ func (i *Initializer) State() InitializationState { succeeded++ case TaskFailed: failed++ + case TaskTimedOut: + timedOut++ case TaskFatal: fatal++ } @@ -367,31 +417,81 @@ func (i *Initializer) State() InitializationState { Int("running", running). Int("succeeded", succeeded). Int("failed", failed). + Int("timedOut", timedOut). Int("fatal", fatal). Msg("calculating initialization state") + if total == 0 { + return StateUninitialized + } if total == succeeded { return StateReady } - // If any fatal exists, prefer Fatal state - if fatal > 0 { - return StateFatal - } - // If all tasks are done (some are failed, none running or pending), it's a "Failed" state - if failed > 0 && (pending+running+succeeded == 0) { + + // failed + timedOut are retryable; pending/running are in-flight. + // Do NOT map "any fatal" → StateFatal while siblings can still recover — + // one permanently-misconfigured upstream must not mark a shared + // Initializer (dozens of networks) as wholly fatal. + retryable := failed + timedOut + inFlight := pending + running + nonTerminal := inFlight + retryable + + if nonTerminal > 0 { + atp := i.attempts.Load() + if atp > 1 { + return StateRetrying + } + if inFlight > 0 { + return StateInitializing + } + // Only retryable tasks left (awaiting auto-retry), nothing in-flight. + if succeeded > 0 || fatal > 0 { + return StatePartial + } return StateFailed } - if failed > 0 && (pending+running == 0) { - return StatePartial + + // All terminal: succeeded and/or fatal only. + if fatal == total { + return StateFatal } - // If we've tried multiple times but still have tasks not succeeded - atp := i.attempts.Load() - if atp > 1 && (pending > 0 || running > 0) { - return StateRetrying + if fatal > 0 { + return StatePartial } return StateInitializing } +// reapOverdueRunningTasks force-transitions Running tasks whose attempt has +// exceeded TaskTimeout to TaskTimedOut. Used after a bounded WaitForTasks so a +// Fn that ignores ctx cannot keep the auto-retry loop's hasPendingWork true +// forever as TaskRunning (and cannot block Stop on that goroutine forever — +// Stop still may time out waiting for the leaked Fn, but the task is +// retryable again). +func (i *Initializer) reapOverdueRunningTasks() { + now := time.Now() + i.tasks.Range(func(_, value interface{}) bool { + t := value.(*BootstrapTask) + if TaskState(t.state.Load()) != TaskRunning { + return true + } + lastAttempt, _ := t.lastAttempt.Load().(time.Time) + if lastAttempt.IsZero() || now.Sub(lastAttempt) < i.conf.TaskTimeout { + return true + } + if cancel, ok := t.ctxCancel.Load().(context.CancelFunc); ok && cancel != nil { + cancel() + } + if t.state.CompareAndSwap(int32(TaskRunning), int32(TaskTimedOut)) { + t.lastErr.Store(wrappedError{err: context.DeadlineExceeded}) + i.logger.Warn(). + Str("task", t.Name). + Dur("runningFor", now.Sub(lastAttempt)). + Msg("initialization task timed out while still running; reaped for retry") + } + return true + }) +} + func (i *Initializer) Status() *InitializerStatus { state := i.State() return &InitializerStatus{ @@ -561,12 +661,8 @@ func (i *Initializer) ensureAutoRetryIfEnabled() { // from another attempt. Only succeeded and fatal tasks are terminal. // // This is the auto-retry loop's stop condition. It deliberately does NOT key off -// State(): State() returns StateFatal as soon as ANY task is fatal, so keying -// the loop off it makes a single permanently-failing task (e.g. an upstream with -// a chainId mismatch) end auto-retry for every sibling task in the same -// initializer. Because one Initializer is shared across many independent -// resources (one bootstrap task per network/upstream), that stranded every -// not-yet-initialized resource until the process restarted. +// State() alone: a single fatal sibling must not end retries for recoverable +// tasks in the same shared Initializer. func (i *Initializer) hasPendingWork() bool { pending := false i.tasks.Range(func(_, value interface{}) bool { @@ -611,6 +707,9 @@ func (i *Initializer) autoRetryLoop(ctx context.Context) { waitCtx, waitCancel := context.WithTimeout(ctx, i.conf.TaskTimeout) err := i.WaitForTasks(waitCtx) waitCancel() + // Reap Fns that ignored their deadline so they become TaskTimedOut + // (retryable) instead of wedging hasPendingWork as TaskRunning forever. + i.reapOverdueRunningTasks() state := i.State() // Stop only once no task can benefit from another attempt (every task // succeeded or is fatal). Fatal tasks are skipped by diff --git a/util/initializer_test.go b/util/initializer_test.go index 4465e658c..630ac3252 100644 --- a/util/initializer_test.go +++ b/util/initializer_test.go @@ -259,7 +259,7 @@ func TestInitializer_TaskTimeout(t *testing.T) { defer init.Stop(nil) require.Error(t, err) assert.Equal(t, StateFailed, init.State()) - assert.Equal(t, TaskFailed, TaskState(task.state.Load())) + assert.Equal(t, TaskTimedOut, TaskState(task.state.Load())) assert.ErrorIs(t, task.Error().Err, context.DeadlineExceeded) } @@ -362,11 +362,13 @@ func TestInitializer_MultipleRapidFailures(t *testing.T) { // Check we tried multiple times (rapidly) assert.True(t, attempts > 1, "should attempt multiple times in quick succession") - // Check final State is either partial or failed + // Auto-retry is still armed, so aggregate state is Retrying while the + // sole task keeps failing between attempts (Failed only once the loop + // has stopped and nothing remains in-flight). state := init.State() assert.True( t, - state == StateFailed, + state == StateFailed || state == StateRetrying, "final state should reflect the repeated failures, got %v", state, ) @@ -404,9 +406,9 @@ func TestInitializer_ForcedCancellationMidTask(t *testing.T) { require.Error(t, err, "should fail or be canceled") assert.ErrorIs(t, err, context.DeadlineExceeded) - // Check task state is failed (or timed out) after forced cancel + // ExecuteTasks already ran until TaskTimeout; deadline surfaces as TimedOut. st := TaskState(task.state.Load()) - assert.True(t, st == TaskFailed, "task should show failed or timed out, got %d", st) + assert.True(t, st == TaskFailed || st == TaskTimedOut, "task should show failed or timed out, got %d", st) } func TestInitializer_MarkTaskAsFailedMidRun(t *testing.T) { @@ -828,6 +830,9 @@ func TestInitializer_FatalTaskDoesNotStopRetryOfOthers(t *testing.T) { }, 5*time.Second, 20*time.Millisecond, "transient task should eventually succeed despite the fatal sibling") assert.Equal(t, TaskFatal, TaskState(fatalTask.state.Load())) + // Aggregate state must be Partial (some OK, some permanently dead) — not + // Fatal, which would imply the whole shared Initializer is unusable. + assert.Equal(t, StatePartial, init.State()) init.Stop(nil) } @@ -901,10 +906,98 @@ func TestInitializer_HungTaskDoesNotBlockRetryOfOthers(t *testing.T) { return TaskState(transientTask.state.Load()) == TaskSucceeded }, 5*time.Second, 20*time.Millisecond, "transient task should eventually succeed despite the hung sibling") + // Hung attempt must be reaped to TimedOut (retryable), not left Running forever. + require.Eventually(t, func() bool { + st := TaskState(hungTask.state.Load()) + return st == TaskTimedOut || st == TaskRunning // Running only during a brief retry window + }, 2*time.Second, 20*time.Millisecond) + + // Observe at least one TimedOut transition (reap happened). + require.Eventually(t, func() bool { + return TaskState(hungTask.state.Load()) == TaskTimedOut || + hungTask.attempts.Load() >= 2 // reaped and retried + }, 3*time.Second, 20*time.Millisecond, "hung task should be reaped (TimedOut) and/or retried") + close(hungRelease) init.Stop(nil) } +func TestInitializer_State_FatalSiblingWithSuccessIsPartial(t *testing.T) { + conf := &InitializerConfig{ + TaskTimeout: time.Second, + AutoRetry: false, + } + appCtx, cancel := context.WithCancel(context.Background()) + defer cancel() + init := setupInitializer(t, appCtx, conf) + + fatalTask := NewBootstrapTask("fatal-task", func(ctx context.Context) error { + return &testFatalError{errors.New("permanent misconfiguration")} + }) + okTask := NewBootstrapTask("ok-task", func(ctx context.Context) error { return nil }) + + _ = init.ExecuteTasks(appCtx, fatalTask, okTask) + require.Equal(t, TaskFatal, TaskState(fatalTask.state.Load())) + require.Equal(t, TaskSucceeded, TaskState(okTask.state.Load())) + assert.Equal(t, StatePartial, init.State(), + "one fatal sibling must not mark the whole initializer Fatal when others succeeded") + init.Stop(nil) +} + +func TestInitializer_State_AllFatalIsFatal(t *testing.T) { + conf := &InitializerConfig{ + TaskTimeout: time.Second, + AutoRetry: false, + } + appCtx, cancel := context.WithCancel(context.Background()) + defer cancel() + init := setupInitializer(t, appCtx, conf) + + a := NewBootstrapTask("fatal-a", func(ctx context.Context) error { + return &testFatalError{errors.New("bad a")} + }) + b := NewBootstrapTask("fatal-b", func(ctx context.Context) error { + return &testFatalError{errors.New("bad b")} + }) + _ = init.ExecuteTasks(appCtx, a, b) + assert.Equal(t, StateFatal, init.State()) + init.Stop(nil) +} + +func TestInitializer_WaitForTasks_ParallelDoesNotSerializeOnSlowSibling(t *testing.T) { + conf := &InitializerConfig{ + TaskTimeout: time.Second, + AutoRetry: false, + } + appCtx, cancel := context.WithCancel(context.Background()) + defer cancel() + init := setupInitializer(t, appCtx, conf) + + slowStarted := make(chan struct{}) + slow := NewBootstrapTask("slow", func(ctx context.Context) error { + close(slowStarted) + select { + case <-ctx.Done(): + return ctx.Err() + case <-time.After(time.Second): + return nil + } + }) + fast := NewBootstrapTask("fast", func(ctx context.Context) error { return nil }) + + go func() { _ = init.ExecuteTasks(appCtx, slow, fast) }() + <-slowStarted + + // Short wait: with sequential Wait, a slow-first Range order could burn the + // whole budget before observing fast. Parallel wait must see fast succeed. + waitCtx, waitCancel := context.WithTimeout(appCtx, 100*time.Millisecond) + defer waitCancel() + _ = init.WaitForTasks(waitCtx) + + assert.Equal(t, TaskSucceeded, TaskState(fast.state.Load()), "fast task must complete even while slow is in-flight") + init.Stop(nil) +} + type testFatalError struct{ error } func (e *testFatalError) IsTaskFatal() bool { return true } From 8f4a725d699b4af1137cabfbadda34fb51f1359d Mon Sep 17 00:00:00 2001 From: Jonny Date: Wed, 22 Jul 2026 14:33:12 +0100 Subject: [PATCH 20/40] fix(ws): serve cached newHeads header when HTTP latest lags tip TipHW alone still allowed eth_getBlockByNumber("latest") to fail-open to a stale upstream block when re-fetch of the WS tip missed. Cache the newHeads header before fan-out and prefer it in EnforceHighestBlock for header-only requests so HTTP cannot regress below a tip already delivered on the pod. Co-authored-by: Cursor --- architecture/evm/eth_getBlockByNumber.go | 60 ++++++++++- erpc/http_server_ws_tip_floor_test.go | 99 +++++++++++++++++++ erpc/networks.go | 61 ++++++++++++ erpc/networks_ws_tip_test.go | 90 ++++++++++++++++- erpc/subscription_manager.go | 10 +- .../wsupstream/adapter_reconnect_test.go | 2 +- indexer/indexer.go | 2 +- indexer/indexer_test.go | 2 +- indexer/ingress.go | 13 +-- indexer/integration_test.go | 2 +- 10 files changed, 325 insertions(+), 16 deletions(-) create mode 100644 erpc/http_server_ws_tip_floor_test.go diff --git a/architecture/evm/eth_getBlockByNumber.go b/architecture/evm/eth_getBlockByNumber.go index b70c0f370..fe522387c 100644 --- a/architecture/evm/eth_getBlockByNumber.go +++ b/architecture/evm/eth_getBlockByNumber.go @@ -108,6 +108,39 @@ func networkPostForward_eth_getBlockByNumber(ctx context.Context, network common return enforceNonNullBlock(nq, nr) } +// observedLatestHeadProvider is implemented by networks that cache the last +// WS newHeads header before fan-out (erpc.Network). Used to floor HTTP +// "latest" responses when upstream re-fetch of the tip would fail open. +type observedLatestHeadProvider interface { + LastObservedLatestHead() (blockNumber int64, payload []byte) +} + +// responseFromObservedLatestHead builds an eth_getBlockByNumber response from +// the cached WS newHeads header when it matches expectedTip. ok is false when +// the cache is missing, behind, or the network does not expose a tip cache. +func responseFromObservedLatestHead(network common.Network, nq *common.NormalizedRequest, expectedTip int64) (*common.NormalizedResponse, bool) { + provider, ok := network.(observedLatestHeadProvider) + if !ok || expectedTip <= 0 { + return nil, false + } + cachedNumber, payload := provider.LastObservedLatestHead() + if cachedNumber != expectedTip || len(payload) == 0 { + return nil, false + } + idBytes, err := common.SonicCfg.Marshal(nq.ID()) + if err != nil { + return nil, false + } + jrr, err := common.NewJsonRpcResponseFromBytes(idBytes, append([]byte(nil), payload...), nil) + if err != nil { + return nil, false + } + resp := common.NewNormalizedResponse(). + WithRequest(nq). + WithJsonRpcResponse(jrr) + return resp, true +} + func enforceHighestBlock(ctx context.Context, network common.Network, nq *common.NormalizedRequest, nr *common.NormalizedResponse, re error) (*common.NormalizedResponse, error) { if re != nil { return nr, re @@ -182,6 +215,19 @@ func enforceHighestBlock(ctx context.Context, network common.Network, nq *common if len(rqj.Params) > 1 { itx, _ = rqj.Params[1].(bool) } + + // Prefer a cached WS newHeads header when the HTTP tip lags. Re-fetching + // the concrete tip often fails when only the WS upstream has seen it yet; + // pickHighestBlock would then fail-open to the stale response. + if !itx { + if cached, ok := responseFromObservedLatestHead(network, nq, highestBlockNumber); ok { + if nr != nil { + nr.Release() + } + return cached, nil + } + } + request, err := BuildGetBlockByNumberRequest(highestBlockNumber, itx) if err != nil { return nil, err @@ -208,7 +254,19 @@ func enforceHighestBlock(ctx context.Context, network common.Network, nq *common nnr, err := network.Forward(ctx, newReq) // This is needed in case highest block number is corrupted somehow and for example // it is requesting a very high non-existent block number. - return pickHighestBlock(ctx, nnr, nr, err) + picked, pickErr := pickHighestBlock(ctx, nnr, nr, err) + // If re-fetch still lost to the stale tip, try the WS-cached header once more + // (TipHW may have advanced mid-flight after the first cache check). + if !itx && pickErr == nil && picked != nil { + _, pickedNumber, refErr := ExtractBlockReferenceFromResponse(ctx, picked) + if refErr == nil && pickedNumber < highestBlockNumber { + if cached, ok := responseFromObservedLatestHead(network, nq, highestBlockNumber); ok { + picked.Release() + return cached, nil + } + } + } + return picked, pickErr } else { return nr, re } diff --git a/erpc/http_server_ws_tip_floor_test.go b/erpc/http_server_ws_tip_floor_test.go new file mode 100644 index 000000000..29f538334 --- /dev/null +++ b/erpc/http_server_ws_tip_floor_test.go @@ -0,0 +1,99 @@ +package erpc + +import ( + "context" + "net/http" + "testing" + "time" + + "github.com/bytedance/sonic" + "github.com/erpc/erpc/common" + "github.com/erpc/erpc/internal/policy" + "github.com/erpc/erpc/util" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +func init() { + util.ConfigureTestLogger() +} + +// Reproduces the tip race that trips consumers when HTTP "latest" lags a +// WS newHeads tip already delivered on the same pod: the only HTTP upstream +// still serves N, TipHW/cache is N+1 from WS, and re-fetch of N+1 would fail. +// eth_getBlockByNumber("latest", false) must return the cached WS header. +func TestHttpServer_GetBlockByNumberLatest_UsesCachedWsTipWhenHttpLags(t *testing.T) { + util.ResetGock() + defer util.ResetGock() + util.SetupMocksForEvmStatePoller() + defer util.AssertNoPendingMocks(t, 0) + + cfg := &common.Config{ + Server: &common.ServerConfig{ + MaxTimeout: common.Duration(100 * time.Second).Ptr(), + }, + Projects: []*common.ProjectConfig{ + { + Id: "test_project", + Networks: []*common.NetworkConfig{ + { + Architecture: "evm", + Evm: &common.EvmNetworkConfig{ + ChainId: 123, + Integrity: &common.EvmIntegrityConfig{ + EnforceHighestBlock: util.BoolPtr(true), + }, + }, + }, + }, + Upstreams: []*common.UpstreamConfig{ + { + Id: "rpc1", + Endpoint: "http://rpc1.localhost", + Type: common.UpstreamTypeEvm, + Evm: &common.EvmUpstreamConfig{ + ChainId: 123, + StatePollerInterval: common.Duration(10 * time.Second), + }, + }, + }, + }, + }, + } + + sendRequest, _, _, shutdown, erpcInstance := createServerTestFixtures(cfg, t) + defer shutdown() + + prj, err := erpcInstance.GetProject("test_project") + require.NoError(t, err) + policy.OverrideAllForTest(prj.policyEngine) + + // Let state poller settle at 0x11118888 (SetupMocksForEvmStatePoller). + time.Sleep(500 * time.Millisecond) + + nw, err := prj.GetNetwork(context.Background(), "evm:123") + require.NoError(t, err) + + // WS path noted tip N+1 before fan-out; HTTP upstream still only has N. + const tip = int64(0x11118889) + wsHeader := []byte(`{"number":"0x11118889","hash":"0xwshead","parentHash":"0xwsparent","timestamp":"0x6702a8f1"}`) + nw.NoteObservedLatestHead(context.Background(), tip, wsHeader) + require.Equal(t, tip, nw.EvmHighestLatestBlockNumber(context.Background())) + + statusCode, _, body := sendRequest(`{ + "jsonrpc": "2.0", + "id": 1, + "method": "eth_getBlockByNumber", + "params": ["latest", false] + }`, nil, nil) + + require.Equal(t, http.StatusOK, statusCode) + + var respObject map[string]interface{} + require.NoError(t, sonic.UnmarshalString(body, &respObject)) + result, ok := respObject["result"].(map[string]interface{}) + require.True(t, ok, "response should have a result object, got: %s", body) + assert.Equal(t, "0x11118889", result["number"], + "HTTP latest must not regress below the WS tip already noted on this pod") + assert.Equal(t, "0xwshead", result["hash"]) +} diff --git a/erpc/networks.go b/erpc/networks.go index 6a88d3241..77c820f92 100644 --- a/erpc/networks.go +++ b/erpc/networks.go @@ -2,6 +2,7 @@ package erpc import ( "context" + "encoding/json" "errors" "fmt" "math" @@ -68,6 +69,18 @@ type Network struct { // cannot regress below a head we have already delivered on the same pod. lastReturnedLatestBlock atomic.Int64 lastReturnedFinalizedBlock atomic.Int64 + + // lastObservedLatestHead caches the most recent newHeads header payload + // (eth_subscription result) whose tip we have already committed to TipHW. + // enforceHighestBlock serves this when HTTP upstreams lag the WS tip and + // a re-fetch of the concrete tip block would fail open to a stale response. + lastObservedLatestHead atomic.Pointer[observedLatestHead] +} + +// observedLatestHead is the last WS newHeads header we noted before fan-out. +type observedLatestHead struct { + number int64 + payload json.RawMessage } // NoteObservedLatestBlock records that this Network has observed head @@ -103,6 +116,54 @@ func (n *Network) NoteObservedLatestBlock(ctx context.Context, blockNumber int64 } } +// NoteObservedLatestHead records a WS newHeads tip together with its header +// payload. It advances TipHW via NoteObservedLatestBlock and caches the +// header so eth_getBlockByNumber("latest", false) can return the same tip +// when HTTP upstreams are still behind. +// +// Callers MUST invoke this before delivering the corresponding newHeads +// notification to any client. Empty payloads still advance TipHW. +func (n *Network) NoteObservedLatestHead(ctx context.Context, blockNumber int64, payload json.RawMessage) { + n.NoteObservedLatestBlock(ctx, blockNumber) + if n == nil || blockNumber <= 0 || len(payload) == 0 { + return + } + // Copy so callers can reuse/recycle their buffer. + stored := observedLatestHead{ + number: blockNumber, + payload: append(json.RawMessage(nil), payload...), + } + for { + cur := n.lastObservedLatestHead.Load() + if cur != nil && blockNumber < cur.number { + return + } + // Same height with a new hash (reorg): replace. Lower heights: reject. + if cur != nil && blockNumber == cur.number { + if n.lastObservedLatestHead.CompareAndSwap(cur, &stored) { + return + } + continue + } + if n.lastObservedLatestHead.CompareAndSwap(cur, &stored) { + return + } + } +} + +// LastObservedLatestHead returns the cached newHeads header for the highest +// tip we have noted on this pod, or (0, nil) if none. +func (n *Network) LastObservedLatestHead() (blockNumber int64, payload []byte) { + if n == nil { + return 0, nil + } + cur := n.lastObservedLatestHead.Load() + if cur == nil || cur.number <= 0 || len(cur.payload) == 0 { + return 0, nil + } + return cur.number, append([]byte(nil), cur.payload...) +} + // Bootstrap registers this network with the policy engine. The engine kicks // off the slot's ticker and runs an initial synchronous eval so request-path // reads through `policyEngine.GetOrdered` always see a populated cache. diff --git a/erpc/networks_ws_tip_test.go b/erpc/networks_ws_tip_test.go index d4cf18009..75151aa50 100644 --- a/erpc/networks_ws_tip_test.go +++ b/erpc/networks_ws_tip_test.go @@ -181,7 +181,7 @@ func TestNetworkHandle_SuggestLatestBlock_AdvancesNetworkTipBeforeFanOut(t *test handle := &networkHandle{nw: network} // Mirrors indexer.Ingest ordering: SuggestLatestBlock then fan-out. - handle.SuggestLatestBlock("ws:bor-1", 90677359) + handle.SuggestLatestBlock("ws:bor-1", 90677359, []byte(`{"number":"0x56789cf","hash":"0xabc"}`)) assert.Equal(t, int64(90677359), upsList[0].EvmStatePoller().LatestBlock(), "per-upstream poller must advance") @@ -189,4 +189,92 @@ func TestNetworkHandle_SuggestLatestBlock_AdvancesNetworkTipBeforeFanOut(t *test "network tip must advance before any client would see the WS head") assert.GreaterOrEqual(t, network.lastReturnedLatestBlock.Load(), int64(90677359), "process-local high-water mark must cover the delivered WS tip") + cachedNum, cachedPayload := network.LastObservedLatestHead() + assert.Equal(t, int64(90677359), cachedNum) + assert.Contains(t, string(cachedPayload), `"0x56789cf"`) +} + +func TestNoteObservedLatestHead_CachesPayloadAndAdvancesTip(t *testing.T) { + util.ResetGock() + defer util.ResetGock() + util.SetupMocksForEvmStatePoller() + + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + + up := &common.UpstreamConfig{ + Type: common.UpstreamTypeEvm, + Id: "rpc1", + Endpoint: "http://rpc1.localhost", + Evm: &common.EvmUpstreamConfig{ChainId: 123}, + } + + gock.New("http://rpc1.localhost"). + Post(""). + Persist(). + Filter(func(r *http.Request) bool { + return strings.Contains(util.SafeReadBody(r), `eth_chainId`) + }). + Reply(200). + JSON([]byte(`{"result":"0x7b"}`)) + + rateLimitersRegistry, _ := upstream.NewRateLimitersRegistry(context.Background(), &common.RateLimiterConfig{}, &log.Logger) + metricsTracker := health.NewTracker(&log.Logger, "test", time.Minute) + + vr := thirdparty.NewVendorsRegistry() + pr, err := thirdparty.NewProvidersRegistry(&log.Logger, vr, []*common.ProviderConfig{}, nil) + require.NoError(t, err) + + ssr, err := data.NewSharedStateRegistry(ctx, &log.Logger, &common.SharedStateConfig{ + Connector: &common.ConnectorConfig{ + Driver: "memory", + Memory: &common.MemoryConnectorConfig{MaxItems: 100_000, MaxTotalSize: "1GB"}, + }, + }) + require.NoError(t, err) + + upstreamsRegistry := upstream.NewUpstreamsRegistry( + ctx, &log.Logger, "test", + []*common.UpstreamConfig{up}, ssr, rateLimitersRegistry, vr, pr, nil, + metricsTracker, nil, + ) + + networkConfig := &common.NetworkConfig{ + Architecture: common.ArchitectureEvm, + Evm: &common.EvmNetworkConfig{ChainId: 123}, + } + network, err := NewNetwork(ctx, &log.Logger, "test", networkConfig, + rateLimitersRegistry, upstreamsRegistry, metricsTracker, nil) + require.NoError(t, err) + + upstreamsRegistry.Bootstrap(ctx) + time.Sleep(200 * time.Millisecond) + require.NoError(t, upstreamsRegistry.GetInitializer().WaitForTasks(ctx)) + require.NoError(t, network.Bootstrap(ctx)) + time.Sleep(250 * time.Millisecond) + + upsList := upstreamsRegistry.GetNetworkUpstreams(ctx, util.EvmNetworkId(123)) + require.Len(t, upsList, 1) + upsList[0].EvmStatePoller().SuggestLatestBlock(1000) + time.Sleep(50 * time.Millisecond) + + payload := []byte(`{"number":"0x3e9","hash":"0xdead","parentHash":"0xbeef"}`) + network.NoteObservedLatestHead(ctx, 1001, payload) + + assert.Equal(t, int64(1001), network.EvmHighestLatestBlockNumber(ctx)) + gotNum, gotPayload := network.LastObservedLatestHead() + assert.Equal(t, int64(1001), gotNum) + assert.JSONEq(t, string(payload), string(gotPayload)) + + // Empty payload still advances tip but does not clobber a good cache. + network.NoteObservedLatestHead(ctx, 1002, nil) + assert.Equal(t, int64(1002), network.EvmHighestLatestBlockNumber(ctx)) + gotNum, gotPayload = network.LastObservedLatestHead() + assert.Equal(t, int64(1001), gotNum, "empty payload must not replace cached head") + assert.JSONEq(t, string(payload), string(gotPayload)) + + // Lower tip must not regress the cache. + network.NoteObservedLatestHead(ctx, 999, []byte(`{"number":"0x3e7","hash":"0xold"}`)) + gotNum, _ = network.LastObservedLatestHead() + assert.Equal(t, int64(1001), gotNum) } diff --git a/erpc/subscription_manager.go b/erpc/subscription_manager.go index 1c9823a04..96b2208b2 100644 --- a/erpc/subscription_manager.go +++ b/erpc/subscription_manager.go @@ -569,14 +569,16 @@ func (h *networkHandle) FinalityDepth() int64 { } // SuggestLatestBlock routes a per-source block observation to the -// upstream's state poller, then advances the network-level latest tip. +// upstream's state poller, then advances the network-level latest tip +// and caches the newHeads header payload for HTTP tip enforcement. // sourceId is the ingress adapter's Name(), which for wsupstream.Adapter // is "ws:". // // Ordering matters: Indexer.Ingest calls this BEFORE fan-out, so by the // time any client sees head N on WS, EvmHighestLatestBlockNumber on this -// pod is already ≥ N (see Network.NoteObservedLatestBlock). -func (h *networkHandle) SuggestLatestBlock(sourceId string, blockNumber int64) { +// pod is already ≥ N and LastObservedLatestHead can serve that header +// (see Network.NoteObservedLatestHead). +func (h *networkHandle) SuggestLatestBlock(sourceId string, blockNumber int64, headPayload []byte) { const prefix = "ws:" if !strings.HasPrefix(sourceId, prefix) { return @@ -592,7 +594,7 @@ func (h *networkHandle) SuggestLatestBlock(sourceId string, blockNumber int64) { } break } - h.nw.NoteObservedLatestBlock(h.nw.appCtx, blockNumber) + h.nw.NoteObservedLatestHead(h.nw.appCtx, blockNumber, headPayload) } // Interface checks: fail the build if either contract drifts. diff --git a/indexer/adapters/wsupstream/adapter_reconnect_test.go b/indexer/adapters/wsupstream/adapter_reconnect_test.go index f07475b0b..d56ae4f39 100644 --- a/indexer/adapters/wsupstream/adapter_reconnect_test.go +++ b/indexer/adapters/wsupstream/adapter_reconnect_test.go @@ -28,7 +28,7 @@ type fakeNetworkHandle struct{} func (fakeNetworkHandle) Id() string { return "evm:324" } func (fakeNetworkHandle) FinalityDepth() int64 { return 0 } -func (fakeNetworkHandle) SuggestLatestBlock(string, int64) {} +func (fakeNetworkHandle) SuggestLatestBlock(string, int64, []byte) {} type fakeSink struct { events chan indexer.StreamEvent diff --git a/indexer/indexer.go b/indexer/indexer.go index a686fecad..f9d060e4c 100644 --- a/indexer/indexer.go +++ b/indexer/indexer.go @@ -361,7 +361,7 @@ func (i *Indexer) Ingest(ev StreamEvent) { // the indexer level — otherwise a lagging source's state poller // stalls on the first dup. if ev.Kind == KindNewHead && !ev.Block.Zero() && ev.SourceId != "" { - ns.handle.SuggestLatestBlock(ev.SourceId, ev.Block.Number) + ns.handle.SuggestLatestBlock(ev.SourceId, ev.Block.Number, ev.Payload) } // Dedup. diff --git a/indexer/indexer_test.go b/indexer/indexer_test.go index b586fb935..d965d9590 100644 --- a/indexer/indexer_test.go +++ b/indexer/indexer_test.go @@ -31,7 +31,7 @@ func newFakeNetwork(id string, depth int64) *fakeNetwork { func (n *fakeNetwork) Id() string { return n.id } func (n *fakeNetwork) FinalityDepth() int64 { return n.finalityDepth } -func (n *fakeNetwork) SuggestLatestBlock(sourceId string, block int64) { +func (n *fakeNetwork) SuggestLatestBlock(sourceId string, block int64, _ []byte) { n.mu.Lock() n.suggestedBy[sourceId] = append(n.suggestedBy[sourceId], block) n.mu.Unlock() diff --git a/indexer/ingress.go b/indexer/ingress.go index d33f57705..80bd54657 100644 --- a/indexer/ingress.go +++ b/indexer/ingress.go @@ -26,12 +26,13 @@ type NetworkHandle interface { FinalityDepth() int64 // SuggestLatestBlock advances the per-source latest-block tracker // (and the network-level latest tip) before the indexer dedupes / - // fans out. Preserving "update-before-dedup" and - // "tip-before-fanout" ordering is critical — the state poller needs - // to see every observation (even ones we drop), and HTTP "latest" - // must not be allowed to regress below a head we are about to - // deliver on WS. - SuggestLatestBlock(sourceId string, blockNumber int64) + // fans out. headPayload is the newHeads eth_subscription result + // (block header JSON); it may be nil when the caller only has a + // number. Preserving "update-before-dedup" and "tip-before-fanout" + // ordering is critical — the state poller needs to see every + // observation (even ones we drop), and HTTP "latest" must not be + // allowed to regress below a head we are about to deliver on WS. + SuggestLatestBlock(sourceId string, blockNumber int64, headPayload []byte) } // EventIngress is an adapter that converts some transport-specific diff --git a/indexer/integration_test.go b/indexer/integration_test.go index 3eec40b0a..9be2c01fb 100644 --- a/indexer/integration_test.go +++ b/indexer/integration_test.go @@ -107,7 +107,7 @@ type stubNetwork struct { func (s *stubNetwork) Id() string { return s.id } func (s *stubNetwork) FinalityDepth() int64 { return s.finality } -func (s *stubNetwork) SuggestLatestBlock(sourceID string, block int64) { +func (s *stubNetwork) SuggestLatestBlock(sourceID string, block int64, _ []byte) { s.mu.Lock() if s.suggestions == nil { s.suggestions = make(map[string][]int64) From eb99874c3530dc3abf0101a580db8044b9c13c06 Mon Sep 17 00:00:00 2001 From: Jonny Date: Wed, 22 Jul 2026 15:43:25 +0100 Subject: [PATCH 21/40] fix(ws): route tip reads to upstreams that already have the block Record the tip-source upstream id with the cached newHeads head, partition eth_getBlockByNumber("latest") using TipHW so WS upstreams are tried first, and pin EnforceHighestBlock re-fetch to that tip source instead of only excluding the stale HTTP responder. Co-authored-by: Cursor --- architecture/evm/eth_getBlockByNumber.go | 22 ++-- erpc/http_server_ws_tip_floor_test.go | 125 ++++++++++++++++++++- erpc/networks.go | 134 ++++++++++++++++++----- erpc/networks_block_partition_test.go | 50 +++++++++ erpc/networks_ws_tip_test.go | 24 ++-- erpc/subscription_manager.go | 2 +- 6 files changed, 314 insertions(+), 43 deletions(-) diff --git a/architecture/evm/eth_getBlockByNumber.go b/architecture/evm/eth_getBlockByNumber.go index fe522387c..518af2013 100644 --- a/architecture/evm/eth_getBlockByNumber.go +++ b/architecture/evm/eth_getBlockByNumber.go @@ -110,9 +110,10 @@ func networkPostForward_eth_getBlockByNumber(ctx context.Context, network common // observedLatestHeadProvider is implemented by networks that cache the last // WS newHeads header before fan-out (erpc.Network). Used to floor HTTP -// "latest" responses when upstream re-fetch of the tip would fail open. +// "latest" responses when upstream re-fetch of the tip would fail open, and +// to pin tip re-fetches to the upstream that delivered the head. type observedLatestHeadProvider interface { - LastObservedLatestHead() (blockNumber int64, payload []byte) + LastObservedLatestHead() (blockNumber int64, payload []byte, upstreamId string) } // responseFromObservedLatestHead builds an eth_getBlockByNumber response from @@ -123,7 +124,7 @@ func responseFromObservedLatestHead(network common.Network, nq *common.Normalize if !ok || expectedTip <= 0 { return nil, false } - cachedNumber, payload := provider.LastObservedLatestHead() + cachedNumber, payload, _ := provider.LastObservedLatestHead() if cachedNumber != expectedTip || len(payload) == 0 { return nil, false } @@ -239,10 +240,17 @@ func enforceHighestBlock(ctx context.Context, network common.Network, nq *common newReq := common.NewNormalizedRequestFromJsonRpcRequest(request) dr := nq.Directives().Clone() dr.SkipCacheRead = "true" - // In case a block number is extracted, it means the node actually has an older latest block. - // Therefore we exclude the current upstream from the request (as high likely it doesn't have this block). - // Otherwise we still allow the current upstream to be used in case json-rpc error was an intermittent issue. - if respBlockNumber > 0 { + // Prefer the upstream that already delivered this tip (typically the + // WS ingress). Falling back to excluding the stale responder keeps + // prior behaviour when we have no tip-source id. + if provider, ok := network.(observedLatestHeadProvider); ok { + cachedNum, _, tipSourceId := provider.LastObservedLatestHead() + if tipSourceId != "" && cachedNum == highestBlockNumber { + dr.UseUpstream = tipSourceId + } else if respBlockNumber > 0 { + dr.UseUpstream = fmt.Sprintf("!%s", nr.UpstreamId()) + } + } else if respBlockNumber > 0 { dr.UseUpstream = fmt.Sprintf("!%s", nr.UpstreamId()) } newReq.SetDirectives(dr) diff --git a/erpc/http_server_ws_tip_floor_test.go b/erpc/http_server_ws_tip_floor_test.go index 29f538334..9c7656574 100644 --- a/erpc/http_server_ws_tip_floor_test.go +++ b/erpc/http_server_ws_tip_floor_test.go @@ -3,6 +3,8 @@ package erpc import ( "context" "net/http" + "strings" + "sync/atomic" "testing" "time" @@ -10,6 +12,7 @@ import ( "github.com/erpc/erpc/common" "github.com/erpc/erpc/internal/policy" "github.com/erpc/erpc/util" + "github.com/h2non/gock" "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" ) @@ -77,7 +80,7 @@ func TestHttpServer_GetBlockByNumberLatest_UsesCachedWsTipWhenHttpLags(t *testin // WS path noted tip N+1 before fan-out; HTTP upstream still only has N. const tip = int64(0x11118889) wsHeader := []byte(`{"number":"0x11118889","hash":"0xwshead","parentHash":"0xwsparent","timestamp":"0x6702a8f1"}`) - nw.NoteObservedLatestHead(context.Background(), tip, wsHeader) + nw.NoteObservedLatestHead(context.Background(), tip, wsHeader, "rpc1") require.Equal(t, tip, nw.EvmHighestLatestBlockNumber(context.Background())) statusCode, _, body := sendRequest(`{ @@ -97,3 +100,123 @@ func TestHttpServer_GetBlockByNumberLatest_UsesCachedWsTipWhenHttpLags(t *testin "HTTP latest must not regress below the WS tip already noted on this pod") assert.Equal(t, "0xwshead", result["hash"]) } + +// Tip-aware routing: after WS ingest advances TipHW + the tip-source poller, +// eth_getBlockByNumber("latest") must prefer that tip-source upstream over a +// lagging HTTP sibling (partition), and EnforceHighestBlock must pin the +// concrete tip re-fetch to it when the first response is still stale. +func TestHttpServer_GetBlockByNumberLatest_PinsReFetchToTipSourceUpstream(t *testing.T) { + util.ResetGock() + defer util.ResetGock() + util.SetupMocksForEvmStatePoller() + defer util.AssertNoPendingMocks(t, 0) + + const tip = int64(0x22228889) + tipHex := "0x22228889" + var tipSourceHits atomic.Int64 + + gock.New("http://rpc2.localhost"). + Post(""). + Filter(func(r *http.Request) bool { + body := util.SafeReadBody(r) + if !strings.Contains(body, "eth_getBlockByNumber") || !strings.Contains(body, tipHex) { + return false + } + tipSourceHits.Add(1) + return true + }). + Reply(200). + JSON([]byte(`{"result":{"number":"0x22228889","hash":"0xtipsrc","parentHash":"0xparent","timestamp":"0x6702a8f1"}}`)) + + cfg := &common.Config{ + Server: &common.ServerConfig{ + MaxTimeout: common.Duration(100 * time.Second).Ptr(), + }, + Projects: []*common.ProjectConfig{ + { + Id: "test_project", + Networks: []*common.NetworkConfig{ + { + Architecture: "evm", + Evm: &common.EvmNetworkConfig{ + ChainId: 123, + Integrity: &common.EvmIntegrityConfig{ + EnforceHighestBlock: util.BoolPtr(true), + }, + }, + Failsafe: []*common.FailsafeConfig{ + { + Retry: &common.RetryPolicyConfig{MaxAttempts: 3}, + }, + }, + }, + }, + Upstreams: []*common.UpstreamConfig{ + { + Id: "rpc1", + Endpoint: "http://rpc1.localhost", + Type: common.UpstreamTypeEvm, + Evm: &common.EvmUpstreamConfig{ + ChainId: 123, + StatePollerInterval: common.Duration(10 * time.Second), + }, + }, + { + Id: "rpc2", + Endpoint: "http://rpc2.localhost", + Type: common.UpstreamTypeEvm, + Evm: &common.EvmUpstreamConfig{ + ChainId: 123, + StatePollerInterval: common.Duration(10 * time.Second), + }, + }, + }, + }, + }, + } + + sendRequest, _, _, shutdown, erpcInstance := createServerTestFixtures(cfg, t) + defer shutdown() + + prj, err := erpcInstance.GetProject("test_project") + require.NoError(t, err) + policy.OverrideAllForTest(prj.policyEngine) + // Prefer lagging HTTP first so EnforceHighestBlock re-fetch pin is exercised + // even when partition would otherwise put the tip-source first. + policy.OverrideOrderForTest(prj.policyEngine, "evm:123", "rpc1", "rpc2") + + time.Sleep(500 * time.Millisecond) + + nw, err := prj.GetNetwork(context.Background(), "evm:123") + require.NoError(t, err) + + // Advance TipHW + tip-source id WITHOUT a cached header and WITHOUT + // bumping rpc2's poller. That forces EnforceHighestBlock to pin the + // concrete tip re-fetch to rpc2 (cache cannot short-circuit; partition + // cannot reorder rpc2 ahead). + nw.NoteObservedLatestHead(context.Background(), tip, nil, "rpc2") + + require.Equal(t, tip, nw.EvmHighestLatestBlockNumber(context.Background())) + cachedNum, cachedPayload, tipSrc := nw.LastObservedLatestHead() + require.Equal(t, tip, cachedNum) + require.Empty(t, cachedPayload) + require.Equal(t, "rpc2", tipSrc) + + statusCode, _, body := sendRequest(`{ + "jsonrpc": "2.0", + "id": 1, + "method": "eth_getBlockByNumber", + "params": ["latest", false] + }`, nil, nil) + + require.Equal(t, http.StatusOK, statusCode) + + var respObject map[string]interface{} + require.NoError(t, sonic.UnmarshalString(body, &respObject)) + result, ok := respObject["result"].(map[string]interface{}) + require.True(t, ok, "response should have a result object, got: %s", body) + assert.Equal(t, tipHex, result["number"]) + assert.Equal(t, "0xtipsrc", result["hash"]) + assert.GreaterOrEqual(t, tipSourceHits.Load(), int64(1), + "EnforceHighestBlock must pin the concrete tip re-fetch to the tip-source upstream") +} diff --git a/erpc/networks.go b/erpc/networks.go index 77c820f92..89c01c8c5 100644 --- a/erpc/networks.go +++ b/erpc/networks.go @@ -79,8 +79,9 @@ type Network struct { // observedLatestHead is the last WS newHeads header we noted before fan-out. type observedLatestHead struct { - number int64 - payload json.RawMessage + number int64 + payload json.RawMessage + upstreamId string // upstream that delivered this tip (e.g. WS ingress source) } // NoteObservedLatestBlock records that this Network has observed head @@ -117,51 +118,78 @@ func (n *Network) NoteObservedLatestBlock(ctx context.Context, blockNumber int64 } // NoteObservedLatestHead records a WS newHeads tip together with its header -// payload. It advances TipHW via NoteObservedLatestBlock and caches the -// header so eth_getBlockByNumber("latest", false) can return the same tip -// when HTTP upstreams are still behind. +// payload and the upstream that delivered it. It advances TipHW via +// NoteObservedLatestBlock, caches the header for HTTP tip flooring, and +// records upstreamId so tip re-fetches can pin to that upstream. // // Callers MUST invoke this before delivering the corresponding newHeads -// notification to any client. Empty payloads still advance TipHW. -func (n *Network) NoteObservedLatestHead(ctx context.Context, blockNumber int64, payload json.RawMessage) { +// notification to any client. Empty payloads still advance TipHW. When +// upstreamId is set without a payload, a tip-source marker is stored for +// UseUpstream pin only if there is no newer cached head (and a prior +// header at a lower tip is left untouched). +func (n *Network) NoteObservedLatestHead(ctx context.Context, blockNumber int64, payload json.RawMessage, upstreamId string) { n.NoteObservedLatestBlock(ctx, blockNumber) - if n == nil || blockNumber <= 0 || len(payload) == 0 { + if n == nil || blockNumber <= 0 { return } - // Copy so callers can reuse/recycle their buffer. + + if len(payload) == 0 { + if upstreamId == "" { + return + } + for { + cur := n.lastObservedLatestHead.Load() + if cur != nil && blockNumber < cur.number { + return + } + if cur != nil && blockNumber > cur.number { + // Tip advanced without a header — do not clobber an older cached head. + return + } + stored := observedLatestHead{ + number: blockNumber, + upstreamId: upstreamId, + } + if cur != nil && len(cur.payload) > 0 { + stored.payload = append(json.RawMessage(nil), cur.payload...) + } + if n.lastObservedLatestHead.CompareAndSwap(cur, &stored) { + return + } + } + } + stored := observedLatestHead{ - number: blockNumber, - payload: append(json.RawMessage(nil), payload...), + number: blockNumber, + payload: append(json.RawMessage(nil), payload...), + upstreamId: upstreamId, } for { cur := n.lastObservedLatestHead.Load() if cur != nil && blockNumber < cur.number { return } - // Same height with a new hash (reorg): replace. Lower heights: reject. - if cur != nil && blockNumber == cur.number { - if n.lastObservedLatestHead.CompareAndSwap(cur, &stored) { - return - } - continue - } if n.lastObservedLatestHead.CompareAndSwap(cur, &stored) { return } } } -// LastObservedLatestHead returns the cached newHeads header for the highest -// tip we have noted on this pod, or (0, nil) if none. -func (n *Network) LastObservedLatestHead() (blockNumber int64, payload []byte) { +// LastObservedLatestHead returns the cached newHeads header and tip-source +// upstream id for the highest tip we have noted on this pod, or zeros if none. +func (n *Network) LastObservedLatestHead() (blockNumber int64, payload []byte, upstreamId string) { if n == nil { - return 0, nil + return 0, nil, "" } cur := n.lastObservedLatestHead.Load() - if cur == nil || cur.number <= 0 || len(cur.payload) == 0 { - return 0, nil + if cur == nil || cur.number <= 0 { + return 0, nil, "" + } + var p []byte + if len(cur.payload) > 0 { + p = append([]byte(nil), cur.payload...) } - return cur.number, append([]byte(nil), cur.payload...) + return cur.number, p, cur.upstreamId } // Bootstrap registers this network with the policy engine. The engine kicks @@ -735,7 +763,8 @@ func (n *Network) Forward(ctx context.Context, req *common.NormalizedRequest) (* } // Block-availability-aware routing: when the request targets a specific - // block, prefer upstreams whose state poller has already observed it. + // block (or tip-tagged "latest" with a TipHW already noted from WS), + // prefer upstreams whose state poller has already observed it. // Without this, requests for a block we just delivered to a client via // WS would still get routed to an HTTP-only sibling whose own polling // loop hasn't caught up — checkUpstreamBlockAvailability rejects with @@ -745,7 +774,7 @@ func (n *Network) Forward(ctx context.Context, req *common.NormalizedRequest) (* // is stable so it composes with the tier and score orderings layered // on top. if n.Architecture() == common.ArchitectureEvm { - if bn := requestBlockNumber(ctx, req); bn > 0 { + if bn := n.routingBlockNumber(ctx, req); bn > 0 { upsList = partitionUpstreamsByLatestBlock(upsList, bn) } } @@ -2234,3 +2263,54 @@ func requestBlockNumber(ctx context.Context, req *common.NormalizedRequest) int6 } return 0 } + +// routingBlockNumber is the block height used for tip-aware upstream +// partitioning. Concrete numeric targets use requestBlockNumber; tip-tagged +// "latest" reads (eth_getBlockByNumber / eth_blockNumber) use TipHW so WS +// upstreams that already observed the head are tried before lagging HTTP +// siblings. +func (n *Network) routingBlockNumber(ctx context.Context, req *common.NormalizedRequest) int64 { + if bn := requestBlockNumber(ctx, req); bn > 0 { + return bn + } + if n == nil || req == nil { + return 0 + } + method, err := req.Method() + if err != nil { + return 0 + } + switch method { + case "eth_getBlockByNumber", "eth_blockNumber": + if !requestTargetsLatestTip(ctx, req, method) { + return 0 + } + return n.lastReturnedLatestBlock.Load() + default: + return 0 + } +} + +// requestTargetsLatestTip reports whether the request is a tip-tagged +// "latest" read (as opposed to a concrete hex / finalized / safe tag). +func requestTargetsLatestTip(ctx context.Context, req *common.NormalizedRequest, method string) bool { + if method == "eth_blockNumber" { + return true + } + if ref := req.EvmBlockRef(); ref != nil { + if s, ok := ref.(string); ok { + return s == "latest" + } + } + jrq, err := req.JsonRpcRequest(ctx) + if err != nil || jrq == nil { + return false + } + jrq.RLock() + defer jrq.RUnlock() + if len(jrq.Params) == 0 { + return false + } + s, ok := jrq.Params[0].(string) + return ok && s == "latest" +} diff --git a/erpc/networks_block_partition_test.go b/erpc/networks_block_partition_test.go index 34186179c..d506eb0f0 100644 --- a/erpc/networks_block_partition_test.go +++ b/erpc/networks_block_partition_test.go @@ -1,10 +1,12 @@ package erpc import ( + "context" "testing" "github.com/erpc/erpc/common" "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" ) // upstream helper: a FakeUpstream wired with a FakeEvmStatePoller at the @@ -99,3 +101,51 @@ func TestPartitionUpstreamsByLatestBlock_SingleUpstreamIsNoOp(t *testing.T) { got := partitionUpstreamsByLatestBlock(in, 100) assert.Equal(t, in, got, "no other upstream to prefer; partition is a no-op") } + +func TestRoutingBlockNumber_LatestUsesTipHW(t *testing.T) { + n := &Network{} + n.lastReturnedLatestBlock.Store(1001) + + latestReq := common.NewNormalizedRequest([]byte( + `{"jsonrpc":"2.0","id":1,"method":"eth_getBlockByNumber","params":["latest",false]}`, + )) + assert.Equal(t, int64(1001), n.routingBlockNumber(context.Background(), latestReq)) + + blockNumberReq := common.NewNormalizedRequest([]byte( + `{"jsonrpc":"2.0","id":1,"method":"eth_blockNumber","params":[]}`, + )) + assert.Equal(t, int64(1001), n.routingBlockNumber(context.Background(), blockNumberReq)) + + concreteReq := common.NewNormalizedRequest([]byte( + `{"jsonrpc":"2.0","id":1,"method":"eth_getBlockByNumber","params":["0x64",false]}`, + )) + // ExtractBlockReferenceFromRequest may set EvmBlockNumber during extract; + // either way concrete hex must win over TipHW. + bn := n.routingBlockNumber(context.Background(), concreteReq) + assert.Equal(t, int64(0x64), bn) + + finalizedReq := common.NewNormalizedRequest([]byte( + `{"jsonrpc":"2.0","id":1,"method":"eth_getBlockByNumber","params":["finalized",false]}`, + )) + assert.Equal(t, int64(0), n.routingBlockNumber(context.Background(), finalizedReq), + "finalized tag must not use TipHW partition") +} + +func TestRoutingBlockNumber_LatestPartitionsWsAheadOfHttp(t *testing.T) { + n := &Network{} + n.lastReturnedLatestBlock.Store(1001) + + latestReq := common.NewNormalizedRequest([]byte( + `{"jsonrpc":"2.0","id":1,"method":"eth_getBlockByNumber","params":["latest",false]}`, + )) + bn := n.routingBlockNumber(context.Background(), latestReq) + require.Equal(t, int64(1001), bn) + + in := []common.Upstream{ + upWithLatest("http-lagging", 1000), + upWithLatest("ws-tip", 1001), + } + got := partitionUpstreamsByLatestBlock(in, bn) + assert.Equal(t, []string{"ws-tip", "http-lagging"}, ids(got), + "WS upstream that already observed TipHW must be tried before lagging HTTP") +} diff --git a/erpc/networks_ws_tip_test.go b/erpc/networks_ws_tip_test.go index 75151aa50..ca45fd723 100644 --- a/erpc/networks_ws_tip_test.go +++ b/erpc/networks_ws_tip_test.go @@ -189,9 +189,10 @@ func TestNetworkHandle_SuggestLatestBlock_AdvancesNetworkTipBeforeFanOut(t *test "network tip must advance before any client would see the WS head") assert.GreaterOrEqual(t, network.lastReturnedLatestBlock.Load(), int64(90677359), "process-local high-water mark must cover the delivered WS tip") - cachedNum, cachedPayload := network.LastObservedLatestHead() + cachedNum, cachedPayload, cachedUps := network.LastObservedLatestHead() assert.Equal(t, int64(90677359), cachedNum) assert.Contains(t, string(cachedPayload), `"0x56789cf"`) + assert.Equal(t, "bor-1", cachedUps) } func TestNoteObservedLatestHead_CachesPayloadAndAdvancesTip(t *testing.T) { @@ -258,23 +259,32 @@ func TestNoteObservedLatestHead_CachesPayloadAndAdvancesTip(t *testing.T) { upsList[0].EvmStatePoller().SuggestLatestBlock(1000) time.Sleep(50 * time.Millisecond) + // Empty payload + tip-source id with no prior cache stores a pin-only marker. + network.NoteObservedLatestHead(ctx, 1000, nil, "rpc2") + gotNum, gotPayload, gotUps := network.LastObservedLatestHead() + assert.Equal(t, int64(1000), gotNum) + assert.Empty(t, gotPayload) + assert.Equal(t, "rpc2", gotUps) + payload := []byte(`{"number":"0x3e9","hash":"0xdead","parentHash":"0xbeef"}`) - network.NoteObservedLatestHead(ctx, 1001, payload) + network.NoteObservedLatestHead(ctx, 1001, payload, "rpc1") assert.Equal(t, int64(1001), network.EvmHighestLatestBlockNumber(ctx)) - gotNum, gotPayload := network.LastObservedLatestHead() + gotNum, gotPayload, gotUps = network.LastObservedLatestHead() assert.Equal(t, int64(1001), gotNum) assert.JSONEq(t, string(payload), string(gotPayload)) + assert.Equal(t, "rpc1", gotUps) // Empty payload still advances tip but does not clobber a good cache. - network.NoteObservedLatestHead(ctx, 1002, nil) + network.NoteObservedLatestHead(ctx, 1002, nil, "rpc1") assert.Equal(t, int64(1002), network.EvmHighestLatestBlockNumber(ctx)) - gotNum, gotPayload = network.LastObservedLatestHead() + gotNum, gotPayload, gotUps = network.LastObservedLatestHead() assert.Equal(t, int64(1001), gotNum, "empty payload must not replace cached head") assert.JSONEq(t, string(payload), string(gotPayload)) + assert.Equal(t, "rpc1", gotUps) // Lower tip must not regress the cache. - network.NoteObservedLatestHead(ctx, 999, []byte(`{"number":"0x3e7","hash":"0xold"}`)) - gotNum, _ = network.LastObservedLatestHead() + network.NoteObservedLatestHead(ctx, 999, []byte(`{"number":"0x3e7","hash":"0xold"}`), "rpc1") + gotNum, _, _ = network.LastObservedLatestHead() assert.Equal(t, int64(1001), gotNum) } diff --git a/erpc/subscription_manager.go b/erpc/subscription_manager.go index 96b2208b2..7a03584fa 100644 --- a/erpc/subscription_manager.go +++ b/erpc/subscription_manager.go @@ -594,7 +594,7 @@ func (h *networkHandle) SuggestLatestBlock(sourceId string, blockNumber int64, h } break } - h.nw.NoteObservedLatestHead(h.nw.appCtx, blockNumber, headPayload) + h.nw.NoteObservedLatestHead(h.nw.appCtx, blockNumber, headPayload, upstreamID) } // Interface checks: fail the build if either contract drifts. From 961d3f916022de44c0f74c2cbbf794e68bc5d1a1 Mon Sep 17 00:00:00 2001 From: Jonny Date: Wed, 22 Jul 2026 15:48:14 +0100 Subject: [PATCH 22/40] fix(ws): drop parallel tip-source id; use poller partition for latest Tip ownership already lives on per-upstream SuggestLatestBlock/LatestBlock and partitionUpstreamsByLatestBlock. Keep TipHW partitioning for "latest" plus the header cache floor; remove the redundant upstreamId pin registry. Co-authored-by: Cursor --- architecture/evm/eth_getBlockByNumber.go | 22 ++-- erpc/http_server_ws_tip_floor_test.go | 125 +---------------------- erpc/networks.go | 80 +++++---------- erpc/networks_ws_tip_test.go | 24 ++--- erpc/subscription_manager.go | 2 +- 5 files changed, 43 insertions(+), 210 deletions(-) diff --git a/architecture/evm/eth_getBlockByNumber.go b/architecture/evm/eth_getBlockByNumber.go index 518af2013..93c24a4ec 100644 --- a/architecture/evm/eth_getBlockByNumber.go +++ b/architecture/evm/eth_getBlockByNumber.go @@ -110,10 +110,9 @@ func networkPostForward_eth_getBlockByNumber(ctx context.Context, network common // observedLatestHeadProvider is implemented by networks that cache the last // WS newHeads header before fan-out (erpc.Network). Used to floor HTTP -// "latest" responses when upstream re-fetch of the tip would fail open, and -// to pin tip re-fetches to the upstream that delivered the head. +// "latest" responses when upstream re-fetch of the tip would fail open. type observedLatestHeadProvider interface { - LastObservedLatestHead() (blockNumber int64, payload []byte, upstreamId string) + LastObservedLatestHead() (blockNumber int64, payload []byte) } // responseFromObservedLatestHead builds an eth_getBlockByNumber response from @@ -124,7 +123,7 @@ func responseFromObservedLatestHead(network common.Network, nq *common.Normalize if !ok || expectedTip <= 0 { return nil, false } - cachedNumber, payload, _ := provider.LastObservedLatestHead() + cachedNumber, payload := provider.LastObservedLatestHead() if cachedNumber != expectedTip || len(payload) == 0 { return nil, false } @@ -240,17 +239,10 @@ func enforceHighestBlock(ctx context.Context, network common.Network, nq *common newReq := common.NewNormalizedRequestFromJsonRpcRequest(request) dr := nq.Directives().Clone() dr.SkipCacheRead = "true" - // Prefer the upstream that already delivered this tip (typically the - // WS ingress). Falling back to excluding the stale responder keeps - // prior behaviour when we have no tip-source id. - if provider, ok := network.(observedLatestHeadProvider); ok { - cachedNum, _, tipSourceId := provider.LastObservedLatestHead() - if tipSourceId != "" && cachedNum == highestBlockNumber { - dr.UseUpstream = tipSourceId - } else if respBlockNumber > 0 { - dr.UseUpstream = fmt.Sprintf("!%s", nr.UpstreamId()) - } - } else if respBlockNumber > 0 { + // Exclude the stale responder. Upstreams that already have the tip + // (via SuggestLatestBlock → LatestBlock) are preferred by + // partitionUpstreamsByLatestBlock on the concrete re-fetch. + if respBlockNumber > 0 { dr.UseUpstream = fmt.Sprintf("!%s", nr.UpstreamId()) } newReq.SetDirectives(dr) diff --git a/erpc/http_server_ws_tip_floor_test.go b/erpc/http_server_ws_tip_floor_test.go index 9c7656574..29f538334 100644 --- a/erpc/http_server_ws_tip_floor_test.go +++ b/erpc/http_server_ws_tip_floor_test.go @@ -3,8 +3,6 @@ package erpc import ( "context" "net/http" - "strings" - "sync/atomic" "testing" "time" @@ -12,7 +10,6 @@ import ( "github.com/erpc/erpc/common" "github.com/erpc/erpc/internal/policy" "github.com/erpc/erpc/util" - "github.com/h2non/gock" "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" ) @@ -80,7 +77,7 @@ func TestHttpServer_GetBlockByNumberLatest_UsesCachedWsTipWhenHttpLags(t *testin // WS path noted tip N+1 before fan-out; HTTP upstream still only has N. const tip = int64(0x11118889) wsHeader := []byte(`{"number":"0x11118889","hash":"0xwshead","parentHash":"0xwsparent","timestamp":"0x6702a8f1"}`) - nw.NoteObservedLatestHead(context.Background(), tip, wsHeader, "rpc1") + nw.NoteObservedLatestHead(context.Background(), tip, wsHeader) require.Equal(t, tip, nw.EvmHighestLatestBlockNumber(context.Background())) statusCode, _, body := sendRequest(`{ @@ -100,123 +97,3 @@ func TestHttpServer_GetBlockByNumberLatest_UsesCachedWsTipWhenHttpLags(t *testin "HTTP latest must not regress below the WS tip already noted on this pod") assert.Equal(t, "0xwshead", result["hash"]) } - -// Tip-aware routing: after WS ingest advances TipHW + the tip-source poller, -// eth_getBlockByNumber("latest") must prefer that tip-source upstream over a -// lagging HTTP sibling (partition), and EnforceHighestBlock must pin the -// concrete tip re-fetch to it when the first response is still stale. -func TestHttpServer_GetBlockByNumberLatest_PinsReFetchToTipSourceUpstream(t *testing.T) { - util.ResetGock() - defer util.ResetGock() - util.SetupMocksForEvmStatePoller() - defer util.AssertNoPendingMocks(t, 0) - - const tip = int64(0x22228889) - tipHex := "0x22228889" - var tipSourceHits atomic.Int64 - - gock.New("http://rpc2.localhost"). - Post(""). - Filter(func(r *http.Request) bool { - body := util.SafeReadBody(r) - if !strings.Contains(body, "eth_getBlockByNumber") || !strings.Contains(body, tipHex) { - return false - } - tipSourceHits.Add(1) - return true - }). - Reply(200). - JSON([]byte(`{"result":{"number":"0x22228889","hash":"0xtipsrc","parentHash":"0xparent","timestamp":"0x6702a8f1"}}`)) - - cfg := &common.Config{ - Server: &common.ServerConfig{ - MaxTimeout: common.Duration(100 * time.Second).Ptr(), - }, - Projects: []*common.ProjectConfig{ - { - Id: "test_project", - Networks: []*common.NetworkConfig{ - { - Architecture: "evm", - Evm: &common.EvmNetworkConfig{ - ChainId: 123, - Integrity: &common.EvmIntegrityConfig{ - EnforceHighestBlock: util.BoolPtr(true), - }, - }, - Failsafe: []*common.FailsafeConfig{ - { - Retry: &common.RetryPolicyConfig{MaxAttempts: 3}, - }, - }, - }, - }, - Upstreams: []*common.UpstreamConfig{ - { - Id: "rpc1", - Endpoint: "http://rpc1.localhost", - Type: common.UpstreamTypeEvm, - Evm: &common.EvmUpstreamConfig{ - ChainId: 123, - StatePollerInterval: common.Duration(10 * time.Second), - }, - }, - { - Id: "rpc2", - Endpoint: "http://rpc2.localhost", - Type: common.UpstreamTypeEvm, - Evm: &common.EvmUpstreamConfig{ - ChainId: 123, - StatePollerInterval: common.Duration(10 * time.Second), - }, - }, - }, - }, - }, - } - - sendRequest, _, _, shutdown, erpcInstance := createServerTestFixtures(cfg, t) - defer shutdown() - - prj, err := erpcInstance.GetProject("test_project") - require.NoError(t, err) - policy.OverrideAllForTest(prj.policyEngine) - // Prefer lagging HTTP first so EnforceHighestBlock re-fetch pin is exercised - // even when partition would otherwise put the tip-source first. - policy.OverrideOrderForTest(prj.policyEngine, "evm:123", "rpc1", "rpc2") - - time.Sleep(500 * time.Millisecond) - - nw, err := prj.GetNetwork(context.Background(), "evm:123") - require.NoError(t, err) - - // Advance TipHW + tip-source id WITHOUT a cached header and WITHOUT - // bumping rpc2's poller. That forces EnforceHighestBlock to pin the - // concrete tip re-fetch to rpc2 (cache cannot short-circuit; partition - // cannot reorder rpc2 ahead). - nw.NoteObservedLatestHead(context.Background(), tip, nil, "rpc2") - - require.Equal(t, tip, nw.EvmHighestLatestBlockNumber(context.Background())) - cachedNum, cachedPayload, tipSrc := nw.LastObservedLatestHead() - require.Equal(t, tip, cachedNum) - require.Empty(t, cachedPayload) - require.Equal(t, "rpc2", tipSrc) - - statusCode, _, body := sendRequest(`{ - "jsonrpc": "2.0", - "id": 1, - "method": "eth_getBlockByNumber", - "params": ["latest", false] - }`, nil, nil) - - require.Equal(t, http.StatusOK, statusCode) - - var respObject map[string]interface{} - require.NoError(t, sonic.UnmarshalString(body, &respObject)) - result, ok := respObject["result"].(map[string]interface{}) - require.True(t, ok, "response should have a result object, got: %s", body) - assert.Equal(t, tipHex, result["number"]) - assert.Equal(t, "0xtipsrc", result["hash"]) - assert.GreaterOrEqual(t, tipSourceHits.Load(), int64(1), - "EnforceHighestBlock must pin the concrete tip re-fetch to the tip-source upstream") -} diff --git a/erpc/networks.go b/erpc/networks.go index 89c01c8c5..d1a569d32 100644 --- a/erpc/networks.go +++ b/erpc/networks.go @@ -79,9 +79,8 @@ type Network struct { // observedLatestHead is the last WS newHeads header we noted before fan-out. type observedLatestHead struct { - number int64 - payload json.RawMessage - upstreamId string // upstream that delivered this tip (e.g. WS ingress source) + number int64 + payload json.RawMessage } // NoteObservedLatestBlock records that this Network has observed head @@ -118,78 +117,53 @@ func (n *Network) NoteObservedLatestBlock(ctx context.Context, blockNumber int64 } // NoteObservedLatestHead records a WS newHeads tip together with its header -// payload and the upstream that delivered it. It advances TipHW via -// NoteObservedLatestBlock, caches the header for HTTP tip flooring, and -// records upstreamId so tip re-fetches can pin to that upstream. +// payload. It advances TipHW via NoteObservedLatestBlock and caches the +// header so eth_getBlockByNumber("latest", false) can return the same tip +// when HTTP upstreams are still behind. Tip ownership itself stays on the +// per-upstream state poller (SuggestLatestBlock → LatestBlock). // // Callers MUST invoke this before delivering the corresponding newHeads -// notification to any client. Empty payloads still advance TipHW. When -// upstreamId is set without a payload, a tip-source marker is stored for -// UseUpstream pin only if there is no newer cached head (and a prior -// header at a lower tip is left untouched). -func (n *Network) NoteObservedLatestHead(ctx context.Context, blockNumber int64, payload json.RawMessage, upstreamId string) { +// notification to any client. Empty payloads still advance TipHW but do +// not clobber a previously cached header. +func (n *Network) NoteObservedLatestHead(ctx context.Context, blockNumber int64, payload json.RawMessage) { n.NoteObservedLatestBlock(ctx, blockNumber) - if n == nil || blockNumber <= 0 { + if n == nil || blockNumber <= 0 || len(payload) == 0 { return } - - if len(payload) == 0 { - if upstreamId == "" { - return - } - for { - cur := n.lastObservedLatestHead.Load() - if cur != nil && blockNumber < cur.number { - return - } - if cur != nil && blockNumber > cur.number { - // Tip advanced without a header — do not clobber an older cached head. - return - } - stored := observedLatestHead{ - number: blockNumber, - upstreamId: upstreamId, - } - if cur != nil && len(cur.payload) > 0 { - stored.payload = append(json.RawMessage(nil), cur.payload...) - } - if n.lastObservedLatestHead.CompareAndSwap(cur, &stored) { - return - } - } - } - + // Copy so callers can reuse/recycle their buffer. stored := observedLatestHead{ - number: blockNumber, - payload: append(json.RawMessage(nil), payload...), - upstreamId: upstreamId, + number: blockNumber, + payload: append(json.RawMessage(nil), payload...), } for { cur := n.lastObservedLatestHead.Load() if cur != nil && blockNumber < cur.number { return } + // Same height with a new hash (reorg): replace. Lower heights: reject. + if cur != nil && blockNumber == cur.number { + if n.lastObservedLatestHead.CompareAndSwap(cur, &stored) { + return + } + continue + } if n.lastObservedLatestHead.CompareAndSwap(cur, &stored) { return } } } -// LastObservedLatestHead returns the cached newHeads header and tip-source -// upstream id for the highest tip we have noted on this pod, or zeros if none. -func (n *Network) LastObservedLatestHead() (blockNumber int64, payload []byte, upstreamId string) { +// LastObservedLatestHead returns the cached newHeads header for the highest +// tip we have noted on this pod, or (0, nil) if none. +func (n *Network) LastObservedLatestHead() (blockNumber int64, payload []byte) { if n == nil { - return 0, nil, "" + return 0, nil } cur := n.lastObservedLatestHead.Load() - if cur == nil || cur.number <= 0 { - return 0, nil, "" - } - var p []byte - if len(cur.payload) > 0 { - p = append([]byte(nil), cur.payload...) + if cur == nil || cur.number <= 0 || len(cur.payload) == 0 { + return 0, nil } - return cur.number, p, cur.upstreamId + return cur.number, append([]byte(nil), cur.payload...) } // Bootstrap registers this network with the policy engine. The engine kicks diff --git a/erpc/networks_ws_tip_test.go b/erpc/networks_ws_tip_test.go index ca45fd723..75151aa50 100644 --- a/erpc/networks_ws_tip_test.go +++ b/erpc/networks_ws_tip_test.go @@ -189,10 +189,9 @@ func TestNetworkHandle_SuggestLatestBlock_AdvancesNetworkTipBeforeFanOut(t *test "network tip must advance before any client would see the WS head") assert.GreaterOrEqual(t, network.lastReturnedLatestBlock.Load(), int64(90677359), "process-local high-water mark must cover the delivered WS tip") - cachedNum, cachedPayload, cachedUps := network.LastObservedLatestHead() + cachedNum, cachedPayload := network.LastObservedLatestHead() assert.Equal(t, int64(90677359), cachedNum) assert.Contains(t, string(cachedPayload), `"0x56789cf"`) - assert.Equal(t, "bor-1", cachedUps) } func TestNoteObservedLatestHead_CachesPayloadAndAdvancesTip(t *testing.T) { @@ -259,32 +258,23 @@ func TestNoteObservedLatestHead_CachesPayloadAndAdvancesTip(t *testing.T) { upsList[0].EvmStatePoller().SuggestLatestBlock(1000) time.Sleep(50 * time.Millisecond) - // Empty payload + tip-source id with no prior cache stores a pin-only marker. - network.NoteObservedLatestHead(ctx, 1000, nil, "rpc2") - gotNum, gotPayload, gotUps := network.LastObservedLatestHead() - assert.Equal(t, int64(1000), gotNum) - assert.Empty(t, gotPayload) - assert.Equal(t, "rpc2", gotUps) - payload := []byte(`{"number":"0x3e9","hash":"0xdead","parentHash":"0xbeef"}`) - network.NoteObservedLatestHead(ctx, 1001, payload, "rpc1") + network.NoteObservedLatestHead(ctx, 1001, payload) assert.Equal(t, int64(1001), network.EvmHighestLatestBlockNumber(ctx)) - gotNum, gotPayload, gotUps = network.LastObservedLatestHead() + gotNum, gotPayload := network.LastObservedLatestHead() assert.Equal(t, int64(1001), gotNum) assert.JSONEq(t, string(payload), string(gotPayload)) - assert.Equal(t, "rpc1", gotUps) // Empty payload still advances tip but does not clobber a good cache. - network.NoteObservedLatestHead(ctx, 1002, nil, "rpc1") + network.NoteObservedLatestHead(ctx, 1002, nil) assert.Equal(t, int64(1002), network.EvmHighestLatestBlockNumber(ctx)) - gotNum, gotPayload, gotUps = network.LastObservedLatestHead() + gotNum, gotPayload = network.LastObservedLatestHead() assert.Equal(t, int64(1001), gotNum, "empty payload must not replace cached head") assert.JSONEq(t, string(payload), string(gotPayload)) - assert.Equal(t, "rpc1", gotUps) // Lower tip must not regress the cache. - network.NoteObservedLatestHead(ctx, 999, []byte(`{"number":"0x3e7","hash":"0xold"}`), "rpc1") - gotNum, _, _ = network.LastObservedLatestHead() + network.NoteObservedLatestHead(ctx, 999, []byte(`{"number":"0x3e7","hash":"0xold"}`)) + gotNum, _ = network.LastObservedLatestHead() assert.Equal(t, int64(1001), gotNum) } diff --git a/erpc/subscription_manager.go b/erpc/subscription_manager.go index 7a03584fa..96b2208b2 100644 --- a/erpc/subscription_manager.go +++ b/erpc/subscription_manager.go @@ -594,7 +594,7 @@ func (h *networkHandle) SuggestLatestBlock(sourceId string, blockNumber int64, h } break } - h.nw.NoteObservedLatestHead(h.nw.appCtx, blockNumber, headPayload, upstreamID) + h.nw.NoteObservedLatestHead(h.nw.appCtx, blockNumber, headPayload) } // Interface checks: fail the build if either contract drifts. From d14244430c6fb3190745858a45ce81b173210022 Mon Sep 17 00:00:00 2001 From: Jonny Date: Wed, 22 Jul 2026 15:56:31 +0100 Subject: [PATCH 23/40] fix(ws): pin tip re-fetch to EvmLeaderUpstream Drop the newHeads header-cache / tip-source registry overbuild. Tip ownership already lives on per-upstream pollers (SuggestLatestBlock) and EvmLeaderUpstream; EnforceHighestBlock now UseUpstream-pins the concrete tip re-fetch to that leader when its LatestBlock covers TipHW. Co-authored-by: Cursor --- architecture/evm/eth_getBlockByNumber.go | 76 ++------- erpc/http_server_ws_tip_floor_test.go | 99 ------------ erpc/http_server_ws_tip_leader_test.go | 146 ++++++++++++++++++ erpc/networks.go | 119 +------------- erpc/networks_block_partition_test.go | 50 ------ erpc/networks_ws_tip_test.go | 90 +---------- erpc/subscription_manager.go | 10 +- .../wsupstream/adapter_reconnect_test.go | 2 +- indexer/indexer.go | 2 +- indexer/indexer_test.go | 2 +- indexer/ingress.go | 13 +- indexer/integration_test.go | 2 +- 12 files changed, 176 insertions(+), 435 deletions(-) delete mode 100644 erpc/http_server_ws_tip_floor_test.go create mode 100644 erpc/http_server_ws_tip_leader_test.go diff --git a/architecture/evm/eth_getBlockByNumber.go b/architecture/evm/eth_getBlockByNumber.go index 93c24a4ec..66916565b 100644 --- a/architecture/evm/eth_getBlockByNumber.go +++ b/architecture/evm/eth_getBlockByNumber.go @@ -108,39 +108,6 @@ func networkPostForward_eth_getBlockByNumber(ctx context.Context, network common return enforceNonNullBlock(nq, nr) } -// observedLatestHeadProvider is implemented by networks that cache the last -// WS newHeads header before fan-out (erpc.Network). Used to floor HTTP -// "latest" responses when upstream re-fetch of the tip would fail open. -type observedLatestHeadProvider interface { - LastObservedLatestHead() (blockNumber int64, payload []byte) -} - -// responseFromObservedLatestHead builds an eth_getBlockByNumber response from -// the cached WS newHeads header when it matches expectedTip. ok is false when -// the cache is missing, behind, or the network does not expose a tip cache. -func responseFromObservedLatestHead(network common.Network, nq *common.NormalizedRequest, expectedTip int64) (*common.NormalizedResponse, bool) { - provider, ok := network.(observedLatestHeadProvider) - if !ok || expectedTip <= 0 { - return nil, false - } - cachedNumber, payload := provider.LastObservedLatestHead() - if cachedNumber != expectedTip || len(payload) == 0 { - return nil, false - } - idBytes, err := common.SonicCfg.Marshal(nq.ID()) - if err != nil { - return nil, false - } - jrr, err := common.NewJsonRpcResponseFromBytes(idBytes, append([]byte(nil), payload...), nil) - if err != nil { - return nil, false - } - resp := common.NewNormalizedResponse(). - WithRequest(nq). - WithJsonRpcResponse(jrr) - return resp, true -} - func enforceHighestBlock(ctx context.Context, network common.Network, nq *common.NormalizedRequest, nr *common.NormalizedResponse, re error) (*common.NormalizedResponse, error) { if re != nil { return nr, re @@ -215,19 +182,6 @@ func enforceHighestBlock(ctx context.Context, network common.Network, nq *common if len(rqj.Params) > 1 { itx, _ = rqj.Params[1].(bool) } - - // Prefer a cached WS newHeads header when the HTTP tip lags. Re-fetching - // the concrete tip often fails when only the WS upstream has seen it yet; - // pickHighestBlock would then fail-open to the stale response. - if !itx { - if cached, ok := responseFromObservedLatestHead(network, nq, highestBlockNumber); ok { - if nr != nil { - nr.Release() - } - return cached, nil - } - } - request, err := BuildGetBlockByNumberRequest(highestBlockNumber, itx) if err != nil { return nil, err @@ -239,10 +193,18 @@ func enforceHighestBlock(ctx context.Context, network common.Network, nq *common newReq := common.NewNormalizedRequestFromJsonRpcRequest(request) dr := nq.Directives().Clone() dr.SkipCacheRead = "true" - // Exclude the stale responder. Upstreams that already have the tip - // (via SuggestLatestBlock → LatestBlock) are preferred by - // partitionUpstreamsByLatestBlock on the concrete re-fetch. - if respBlockNumber > 0 { + // Prefer the upstream whose poller already owns this tip + // (EvmLeaderUpstream — typically the WS ingress that called + // SuggestLatestBlock). Fall back to excluding the stale + // responder when no local poller has caught up yet. + if leader := network.EvmLeaderUpstream(ctx); leader != nil { + if eu, ok := leader.(common.EvmUpstream); ok { + if sp := eu.EvmStatePoller(); sp != nil && !sp.IsObjectNull() && sp.LatestBlock() >= highestBlockNumber { + dr.UseUpstream = leader.Id() + } + } + } + if dr.UseUpstream == "" && respBlockNumber > 0 { dr.UseUpstream = fmt.Sprintf("!%s", nr.UpstreamId()) } newReq.SetDirectives(dr) @@ -254,19 +216,7 @@ func enforceHighestBlock(ctx context.Context, network common.Network, nq *common nnr, err := network.Forward(ctx, newReq) // This is needed in case highest block number is corrupted somehow and for example // it is requesting a very high non-existent block number. - picked, pickErr := pickHighestBlock(ctx, nnr, nr, err) - // If re-fetch still lost to the stale tip, try the WS-cached header once more - // (TipHW may have advanced mid-flight after the first cache check). - if !itx && pickErr == nil && picked != nil { - _, pickedNumber, refErr := ExtractBlockReferenceFromResponse(ctx, picked) - if refErr == nil && pickedNumber < highestBlockNumber { - if cached, ok := responseFromObservedLatestHead(network, nq, highestBlockNumber); ok { - picked.Release() - return cached, nil - } - } - } - return picked, pickErr + return pickHighestBlock(ctx, nnr, nr, err) } else { return nr, re } diff --git a/erpc/http_server_ws_tip_floor_test.go b/erpc/http_server_ws_tip_floor_test.go deleted file mode 100644 index 29f538334..000000000 --- a/erpc/http_server_ws_tip_floor_test.go +++ /dev/null @@ -1,99 +0,0 @@ -package erpc - -import ( - "context" - "net/http" - "testing" - "time" - - "github.com/bytedance/sonic" - "github.com/erpc/erpc/common" - "github.com/erpc/erpc/internal/policy" - "github.com/erpc/erpc/util" - "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/require" -) - -func init() { - util.ConfigureTestLogger() -} - -// Reproduces the tip race that trips consumers when HTTP "latest" lags a -// WS newHeads tip already delivered on the same pod: the only HTTP upstream -// still serves N, TipHW/cache is N+1 from WS, and re-fetch of N+1 would fail. -// eth_getBlockByNumber("latest", false) must return the cached WS header. -func TestHttpServer_GetBlockByNumberLatest_UsesCachedWsTipWhenHttpLags(t *testing.T) { - util.ResetGock() - defer util.ResetGock() - util.SetupMocksForEvmStatePoller() - defer util.AssertNoPendingMocks(t, 0) - - cfg := &common.Config{ - Server: &common.ServerConfig{ - MaxTimeout: common.Duration(100 * time.Second).Ptr(), - }, - Projects: []*common.ProjectConfig{ - { - Id: "test_project", - Networks: []*common.NetworkConfig{ - { - Architecture: "evm", - Evm: &common.EvmNetworkConfig{ - ChainId: 123, - Integrity: &common.EvmIntegrityConfig{ - EnforceHighestBlock: util.BoolPtr(true), - }, - }, - }, - }, - Upstreams: []*common.UpstreamConfig{ - { - Id: "rpc1", - Endpoint: "http://rpc1.localhost", - Type: common.UpstreamTypeEvm, - Evm: &common.EvmUpstreamConfig{ - ChainId: 123, - StatePollerInterval: common.Duration(10 * time.Second), - }, - }, - }, - }, - }, - } - - sendRequest, _, _, shutdown, erpcInstance := createServerTestFixtures(cfg, t) - defer shutdown() - - prj, err := erpcInstance.GetProject("test_project") - require.NoError(t, err) - policy.OverrideAllForTest(prj.policyEngine) - - // Let state poller settle at 0x11118888 (SetupMocksForEvmStatePoller). - time.Sleep(500 * time.Millisecond) - - nw, err := prj.GetNetwork(context.Background(), "evm:123") - require.NoError(t, err) - - // WS path noted tip N+1 before fan-out; HTTP upstream still only has N. - const tip = int64(0x11118889) - wsHeader := []byte(`{"number":"0x11118889","hash":"0xwshead","parentHash":"0xwsparent","timestamp":"0x6702a8f1"}`) - nw.NoteObservedLatestHead(context.Background(), tip, wsHeader) - require.Equal(t, tip, nw.EvmHighestLatestBlockNumber(context.Background())) - - statusCode, _, body := sendRequest(`{ - "jsonrpc": "2.0", - "id": 1, - "method": "eth_getBlockByNumber", - "params": ["latest", false] - }`, nil, nil) - - require.Equal(t, http.StatusOK, statusCode) - - var respObject map[string]interface{} - require.NoError(t, sonic.UnmarshalString(body, &respObject)) - result, ok := respObject["result"].(map[string]interface{}) - require.True(t, ok, "response should have a result object, got: %s", body) - assert.Equal(t, "0x11118889", result["number"], - "HTTP latest must not regress below the WS tip already noted on this pod") - assert.Equal(t, "0xwshead", result["hash"]) -} diff --git a/erpc/http_server_ws_tip_leader_test.go b/erpc/http_server_ws_tip_leader_test.go new file mode 100644 index 000000000..1dbb56672 --- /dev/null +++ b/erpc/http_server_ws_tip_leader_test.go @@ -0,0 +1,146 @@ +package erpc + +import ( + "context" + "net/http" + "strings" + "sync/atomic" + "testing" + "time" + + "github.com/bytedance/sonic" + "github.com/erpc/erpc/common" + "github.com/erpc/erpc/internal/policy" + "github.com/erpc/erpc/upstream" + "github.com/erpc/erpc/util" + "github.com/h2non/gock" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +func init() { + util.ConfigureTestLogger() +} + +// EnforceHighestBlock must re-fetch the concrete tip via EvmLeaderUpstream +// (the poller advanced by SuggestLatestBlock), not fail-open to a lagging +// sibling that answered "latest" first. +func TestHttpServer_GetBlockByNumberLatest_RefetchPinsEvmLeaderUpstream(t *testing.T) { + util.ResetGock() + defer util.ResetGock() + util.SetupMocksForEvmStatePoller() + defer util.AssertNoPendingMocks(t, 0) + + const tip = int64(0x22228889) + tipHex := "0x22228889" + var leaderHits atomic.Int64 + + gock.New("http://rpc2.localhost"). + Post(""). + Filter(func(r *http.Request) bool { + body := util.SafeReadBody(r) + if !strings.Contains(body, "eth_getBlockByNumber") || !strings.Contains(body, tipHex) { + return false + } + leaderHits.Add(1) + return true + }). + Reply(200). + JSON([]byte(`{"result":{"number":"0x22228889","hash":"0xleader","parentHash":"0xparent","timestamp":"0x6702a8f1"}}`)) + + cfg := &common.Config{ + Server: &common.ServerConfig{ + MaxTimeout: common.Duration(100 * time.Second).Ptr(), + }, + Projects: []*common.ProjectConfig{ + { + Id: "test_project", + Networks: []*common.NetworkConfig{ + { + Architecture: "evm", + Evm: &common.EvmNetworkConfig{ + ChainId: 123, + Integrity: &common.EvmIntegrityConfig{ + EnforceHighestBlock: util.BoolPtr(true), + }, + }, + Failsafe: []*common.FailsafeConfig{ + { + Retry: &common.RetryPolicyConfig{MaxAttempts: 3}, + }, + }, + }, + }, + Upstreams: []*common.UpstreamConfig{ + { + Id: "rpc1", + Endpoint: "http://rpc1.localhost", + Type: common.UpstreamTypeEvm, + Evm: &common.EvmUpstreamConfig{ + ChainId: 123, + StatePollerInterval: common.Duration(10 * time.Second), + }, + }, + { + Id: "rpc2", + Endpoint: "http://rpc2.localhost", + Type: common.UpstreamTypeEvm, + Evm: &common.EvmUpstreamConfig{ + ChainId: 123, + StatePollerInterval: common.Duration(10 * time.Second), + }, + }, + }, + }, + }, + } + + sendRequest, _, _, shutdown, erpcInstance := createServerTestFixtures(cfg, t) + defer shutdown() + + prj, err := erpcInstance.GetProject("test_project") + require.NoError(t, err) + policy.OverrideAllForTest(prj.policyEngine) + // Prefer lagging rpc1 for the initial "latest" so EnforceHighestBlock re-fetch runs. + policy.OverrideOrderForTest(prj.policyEngine, "evm:123", "rpc1", "rpc2") + + time.Sleep(500 * time.Millisecond) + + nw, err := prj.GetNetwork(context.Background(), "evm:123") + require.NoError(t, err) + + var leader *upstream.Upstream + for _, u := range nw.upstreamsRegistry.GetNetworkUpstreams(context.Background(), "evm:123") { + if u.Id() == "rpc2" { + leader = u + break + } + } + require.NotNil(t, leader) + + // Mirror WS ingest: tip-source poller + TipHW before any client sees the head. + leader.EvmStatePoller().SuggestLatestBlock(tip) + nw.NoteObservedLatestBlock(context.Background(), tip) + + require.Equal(t, tip, leader.EvmStatePoller().LatestBlock()) + require.Equal(t, "rpc2", nw.EvmLeaderUpstream(context.Background()).Id()) + require.Equal(t, tip, nw.EvmHighestLatestBlockNumber(context.Background())) + + statusCode, _, body := sendRequest(`{ + "jsonrpc": "2.0", + "id": 1, + "method": "eth_getBlockByNumber", + "params": ["latest", false] + }`, nil, nil) + + require.Equal(t, http.StatusOK, statusCode) + + var respObject map[string]interface{} + require.NoError(t, sonic.UnmarshalString(body, &respObject)) + result, ok := respObject["result"].(map[string]interface{}) + require.True(t, ok, "response should have a result object, got: %s", body) + assert.Equal(t, tipHex, result["number"]) + assert.Equal(t, "0xleader", result["hash"]) + assert.GreaterOrEqual(t, leaderHits.Load(), int64(1), + "EnforceHighestBlock must pin the tip re-fetch to EvmLeaderUpstream") +} diff --git a/erpc/networks.go b/erpc/networks.go index d1a569d32..6a88d3241 100644 --- a/erpc/networks.go +++ b/erpc/networks.go @@ -2,7 +2,6 @@ package erpc import ( "context" - "encoding/json" "errors" "fmt" "math" @@ -69,18 +68,6 @@ type Network struct { // cannot regress below a head we have already delivered on the same pod. lastReturnedLatestBlock atomic.Int64 lastReturnedFinalizedBlock atomic.Int64 - - // lastObservedLatestHead caches the most recent newHeads header payload - // (eth_subscription result) whose tip we have already committed to TipHW. - // enforceHighestBlock serves this when HTTP upstreams lag the WS tip and - // a re-fetch of the concrete tip block would fail open to a stale response. - lastObservedLatestHead atomic.Pointer[observedLatestHead] -} - -// observedLatestHead is the last WS newHeads header we noted before fan-out. -type observedLatestHead struct { - number int64 - payload json.RawMessage } // NoteObservedLatestBlock records that this Network has observed head @@ -116,56 +103,6 @@ func (n *Network) NoteObservedLatestBlock(ctx context.Context, blockNumber int64 } } -// NoteObservedLatestHead records a WS newHeads tip together with its header -// payload. It advances TipHW via NoteObservedLatestBlock and caches the -// header so eth_getBlockByNumber("latest", false) can return the same tip -// when HTTP upstreams are still behind. Tip ownership itself stays on the -// per-upstream state poller (SuggestLatestBlock → LatestBlock). -// -// Callers MUST invoke this before delivering the corresponding newHeads -// notification to any client. Empty payloads still advance TipHW but do -// not clobber a previously cached header. -func (n *Network) NoteObservedLatestHead(ctx context.Context, blockNumber int64, payload json.RawMessage) { - n.NoteObservedLatestBlock(ctx, blockNumber) - if n == nil || blockNumber <= 0 || len(payload) == 0 { - return - } - // Copy so callers can reuse/recycle their buffer. - stored := observedLatestHead{ - number: blockNumber, - payload: append(json.RawMessage(nil), payload...), - } - for { - cur := n.lastObservedLatestHead.Load() - if cur != nil && blockNumber < cur.number { - return - } - // Same height with a new hash (reorg): replace. Lower heights: reject. - if cur != nil && blockNumber == cur.number { - if n.lastObservedLatestHead.CompareAndSwap(cur, &stored) { - return - } - continue - } - if n.lastObservedLatestHead.CompareAndSwap(cur, &stored) { - return - } - } -} - -// LastObservedLatestHead returns the cached newHeads header for the highest -// tip we have noted on this pod, or (0, nil) if none. -func (n *Network) LastObservedLatestHead() (blockNumber int64, payload []byte) { - if n == nil { - return 0, nil - } - cur := n.lastObservedLatestHead.Load() - if cur == nil || cur.number <= 0 || len(cur.payload) == 0 { - return 0, nil - } - return cur.number, append([]byte(nil), cur.payload...) -} - // Bootstrap registers this network with the policy engine. The engine kicks // off the slot's ticker and runs an initial synchronous eval so request-path // reads through `policyEngine.GetOrdered` always see a populated cache. @@ -737,8 +674,7 @@ func (n *Network) Forward(ctx context.Context, req *common.NormalizedRequest) (* } // Block-availability-aware routing: when the request targets a specific - // block (or tip-tagged "latest" with a TipHW already noted from WS), - // prefer upstreams whose state poller has already observed it. + // block, prefer upstreams whose state poller has already observed it. // Without this, requests for a block we just delivered to a client via // WS would still get routed to an HTTP-only sibling whose own polling // loop hasn't caught up — checkUpstreamBlockAvailability rejects with @@ -748,7 +684,7 @@ func (n *Network) Forward(ctx context.Context, req *common.NormalizedRequest) (* // is stable so it composes with the tier and score orderings layered // on top. if n.Architecture() == common.ArchitectureEvm { - if bn := n.routingBlockNumber(ctx, req); bn > 0 { + if bn := requestBlockNumber(ctx, req); bn > 0 { upsList = partitionUpstreamsByLatestBlock(upsList, bn) } } @@ -2237,54 +2173,3 @@ func requestBlockNumber(ctx context.Context, req *common.NormalizedRequest) int6 } return 0 } - -// routingBlockNumber is the block height used for tip-aware upstream -// partitioning. Concrete numeric targets use requestBlockNumber; tip-tagged -// "latest" reads (eth_getBlockByNumber / eth_blockNumber) use TipHW so WS -// upstreams that already observed the head are tried before lagging HTTP -// siblings. -func (n *Network) routingBlockNumber(ctx context.Context, req *common.NormalizedRequest) int64 { - if bn := requestBlockNumber(ctx, req); bn > 0 { - return bn - } - if n == nil || req == nil { - return 0 - } - method, err := req.Method() - if err != nil { - return 0 - } - switch method { - case "eth_getBlockByNumber", "eth_blockNumber": - if !requestTargetsLatestTip(ctx, req, method) { - return 0 - } - return n.lastReturnedLatestBlock.Load() - default: - return 0 - } -} - -// requestTargetsLatestTip reports whether the request is a tip-tagged -// "latest" read (as opposed to a concrete hex / finalized / safe tag). -func requestTargetsLatestTip(ctx context.Context, req *common.NormalizedRequest, method string) bool { - if method == "eth_blockNumber" { - return true - } - if ref := req.EvmBlockRef(); ref != nil { - if s, ok := ref.(string); ok { - return s == "latest" - } - } - jrq, err := req.JsonRpcRequest(ctx) - if err != nil || jrq == nil { - return false - } - jrq.RLock() - defer jrq.RUnlock() - if len(jrq.Params) == 0 { - return false - } - s, ok := jrq.Params[0].(string) - return ok && s == "latest" -} diff --git a/erpc/networks_block_partition_test.go b/erpc/networks_block_partition_test.go index d506eb0f0..34186179c 100644 --- a/erpc/networks_block_partition_test.go +++ b/erpc/networks_block_partition_test.go @@ -1,12 +1,10 @@ package erpc import ( - "context" "testing" "github.com/erpc/erpc/common" "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/require" ) // upstream helper: a FakeUpstream wired with a FakeEvmStatePoller at the @@ -101,51 +99,3 @@ func TestPartitionUpstreamsByLatestBlock_SingleUpstreamIsNoOp(t *testing.T) { got := partitionUpstreamsByLatestBlock(in, 100) assert.Equal(t, in, got, "no other upstream to prefer; partition is a no-op") } - -func TestRoutingBlockNumber_LatestUsesTipHW(t *testing.T) { - n := &Network{} - n.lastReturnedLatestBlock.Store(1001) - - latestReq := common.NewNormalizedRequest([]byte( - `{"jsonrpc":"2.0","id":1,"method":"eth_getBlockByNumber","params":["latest",false]}`, - )) - assert.Equal(t, int64(1001), n.routingBlockNumber(context.Background(), latestReq)) - - blockNumberReq := common.NewNormalizedRequest([]byte( - `{"jsonrpc":"2.0","id":1,"method":"eth_blockNumber","params":[]}`, - )) - assert.Equal(t, int64(1001), n.routingBlockNumber(context.Background(), blockNumberReq)) - - concreteReq := common.NewNormalizedRequest([]byte( - `{"jsonrpc":"2.0","id":1,"method":"eth_getBlockByNumber","params":["0x64",false]}`, - )) - // ExtractBlockReferenceFromRequest may set EvmBlockNumber during extract; - // either way concrete hex must win over TipHW. - bn := n.routingBlockNumber(context.Background(), concreteReq) - assert.Equal(t, int64(0x64), bn) - - finalizedReq := common.NewNormalizedRequest([]byte( - `{"jsonrpc":"2.0","id":1,"method":"eth_getBlockByNumber","params":["finalized",false]}`, - )) - assert.Equal(t, int64(0), n.routingBlockNumber(context.Background(), finalizedReq), - "finalized tag must not use TipHW partition") -} - -func TestRoutingBlockNumber_LatestPartitionsWsAheadOfHttp(t *testing.T) { - n := &Network{} - n.lastReturnedLatestBlock.Store(1001) - - latestReq := common.NewNormalizedRequest([]byte( - `{"jsonrpc":"2.0","id":1,"method":"eth_getBlockByNumber","params":["latest",false]}`, - )) - bn := n.routingBlockNumber(context.Background(), latestReq) - require.Equal(t, int64(1001), bn) - - in := []common.Upstream{ - upWithLatest("http-lagging", 1000), - upWithLatest("ws-tip", 1001), - } - got := partitionUpstreamsByLatestBlock(in, bn) - assert.Equal(t, []string{"ws-tip", "http-lagging"}, ids(got), - "WS upstream that already observed TipHW must be tried before lagging HTTP") -} diff --git a/erpc/networks_ws_tip_test.go b/erpc/networks_ws_tip_test.go index 75151aa50..d4cf18009 100644 --- a/erpc/networks_ws_tip_test.go +++ b/erpc/networks_ws_tip_test.go @@ -181,7 +181,7 @@ func TestNetworkHandle_SuggestLatestBlock_AdvancesNetworkTipBeforeFanOut(t *test handle := &networkHandle{nw: network} // Mirrors indexer.Ingest ordering: SuggestLatestBlock then fan-out. - handle.SuggestLatestBlock("ws:bor-1", 90677359, []byte(`{"number":"0x56789cf","hash":"0xabc"}`)) + handle.SuggestLatestBlock("ws:bor-1", 90677359) assert.Equal(t, int64(90677359), upsList[0].EvmStatePoller().LatestBlock(), "per-upstream poller must advance") @@ -189,92 +189,4 @@ func TestNetworkHandle_SuggestLatestBlock_AdvancesNetworkTipBeforeFanOut(t *test "network tip must advance before any client would see the WS head") assert.GreaterOrEqual(t, network.lastReturnedLatestBlock.Load(), int64(90677359), "process-local high-water mark must cover the delivered WS tip") - cachedNum, cachedPayload := network.LastObservedLatestHead() - assert.Equal(t, int64(90677359), cachedNum) - assert.Contains(t, string(cachedPayload), `"0x56789cf"`) -} - -func TestNoteObservedLatestHead_CachesPayloadAndAdvancesTip(t *testing.T) { - util.ResetGock() - defer util.ResetGock() - util.SetupMocksForEvmStatePoller() - - ctx, cancel := context.WithCancel(context.Background()) - defer cancel() - - up := &common.UpstreamConfig{ - Type: common.UpstreamTypeEvm, - Id: "rpc1", - Endpoint: "http://rpc1.localhost", - Evm: &common.EvmUpstreamConfig{ChainId: 123}, - } - - gock.New("http://rpc1.localhost"). - Post(""). - Persist(). - Filter(func(r *http.Request) bool { - return strings.Contains(util.SafeReadBody(r), `eth_chainId`) - }). - Reply(200). - JSON([]byte(`{"result":"0x7b"}`)) - - rateLimitersRegistry, _ := upstream.NewRateLimitersRegistry(context.Background(), &common.RateLimiterConfig{}, &log.Logger) - metricsTracker := health.NewTracker(&log.Logger, "test", time.Minute) - - vr := thirdparty.NewVendorsRegistry() - pr, err := thirdparty.NewProvidersRegistry(&log.Logger, vr, []*common.ProviderConfig{}, nil) - require.NoError(t, err) - - ssr, err := data.NewSharedStateRegistry(ctx, &log.Logger, &common.SharedStateConfig{ - Connector: &common.ConnectorConfig{ - Driver: "memory", - Memory: &common.MemoryConnectorConfig{MaxItems: 100_000, MaxTotalSize: "1GB"}, - }, - }) - require.NoError(t, err) - - upstreamsRegistry := upstream.NewUpstreamsRegistry( - ctx, &log.Logger, "test", - []*common.UpstreamConfig{up}, ssr, rateLimitersRegistry, vr, pr, nil, - metricsTracker, nil, - ) - - networkConfig := &common.NetworkConfig{ - Architecture: common.ArchitectureEvm, - Evm: &common.EvmNetworkConfig{ChainId: 123}, - } - network, err := NewNetwork(ctx, &log.Logger, "test", networkConfig, - rateLimitersRegistry, upstreamsRegistry, metricsTracker, nil) - require.NoError(t, err) - - upstreamsRegistry.Bootstrap(ctx) - time.Sleep(200 * time.Millisecond) - require.NoError(t, upstreamsRegistry.GetInitializer().WaitForTasks(ctx)) - require.NoError(t, network.Bootstrap(ctx)) - time.Sleep(250 * time.Millisecond) - - upsList := upstreamsRegistry.GetNetworkUpstreams(ctx, util.EvmNetworkId(123)) - require.Len(t, upsList, 1) - upsList[0].EvmStatePoller().SuggestLatestBlock(1000) - time.Sleep(50 * time.Millisecond) - - payload := []byte(`{"number":"0x3e9","hash":"0xdead","parentHash":"0xbeef"}`) - network.NoteObservedLatestHead(ctx, 1001, payload) - - assert.Equal(t, int64(1001), network.EvmHighestLatestBlockNumber(ctx)) - gotNum, gotPayload := network.LastObservedLatestHead() - assert.Equal(t, int64(1001), gotNum) - assert.JSONEq(t, string(payload), string(gotPayload)) - - // Empty payload still advances tip but does not clobber a good cache. - network.NoteObservedLatestHead(ctx, 1002, nil) - assert.Equal(t, int64(1002), network.EvmHighestLatestBlockNumber(ctx)) - gotNum, gotPayload = network.LastObservedLatestHead() - assert.Equal(t, int64(1001), gotNum, "empty payload must not replace cached head") - assert.JSONEq(t, string(payload), string(gotPayload)) - - // Lower tip must not regress the cache. - network.NoteObservedLatestHead(ctx, 999, []byte(`{"number":"0x3e7","hash":"0xold"}`)) - gotNum, _ = network.LastObservedLatestHead() - assert.Equal(t, int64(1001), gotNum) } diff --git a/erpc/subscription_manager.go b/erpc/subscription_manager.go index 96b2208b2..1c9823a04 100644 --- a/erpc/subscription_manager.go +++ b/erpc/subscription_manager.go @@ -569,16 +569,14 @@ func (h *networkHandle) FinalityDepth() int64 { } // SuggestLatestBlock routes a per-source block observation to the -// upstream's state poller, then advances the network-level latest tip -// and caches the newHeads header payload for HTTP tip enforcement. +// upstream's state poller, then advances the network-level latest tip. // sourceId is the ingress adapter's Name(), which for wsupstream.Adapter // is "ws:". // // Ordering matters: Indexer.Ingest calls this BEFORE fan-out, so by the // time any client sees head N on WS, EvmHighestLatestBlockNumber on this -// pod is already ≥ N and LastObservedLatestHead can serve that header -// (see Network.NoteObservedLatestHead). -func (h *networkHandle) SuggestLatestBlock(sourceId string, blockNumber int64, headPayload []byte) { +// pod is already ≥ N (see Network.NoteObservedLatestBlock). +func (h *networkHandle) SuggestLatestBlock(sourceId string, blockNumber int64) { const prefix = "ws:" if !strings.HasPrefix(sourceId, prefix) { return @@ -594,7 +592,7 @@ func (h *networkHandle) SuggestLatestBlock(sourceId string, blockNumber int64, h } break } - h.nw.NoteObservedLatestHead(h.nw.appCtx, blockNumber, headPayload) + h.nw.NoteObservedLatestBlock(h.nw.appCtx, blockNumber) } // Interface checks: fail the build if either contract drifts. diff --git a/indexer/adapters/wsupstream/adapter_reconnect_test.go b/indexer/adapters/wsupstream/adapter_reconnect_test.go index d56ae4f39..f07475b0b 100644 --- a/indexer/adapters/wsupstream/adapter_reconnect_test.go +++ b/indexer/adapters/wsupstream/adapter_reconnect_test.go @@ -28,7 +28,7 @@ type fakeNetworkHandle struct{} func (fakeNetworkHandle) Id() string { return "evm:324" } func (fakeNetworkHandle) FinalityDepth() int64 { return 0 } -func (fakeNetworkHandle) SuggestLatestBlock(string, int64, []byte) {} +func (fakeNetworkHandle) SuggestLatestBlock(string, int64) {} type fakeSink struct { events chan indexer.StreamEvent diff --git a/indexer/indexer.go b/indexer/indexer.go index f9d060e4c..a686fecad 100644 --- a/indexer/indexer.go +++ b/indexer/indexer.go @@ -361,7 +361,7 @@ func (i *Indexer) Ingest(ev StreamEvent) { // the indexer level — otherwise a lagging source's state poller // stalls on the first dup. if ev.Kind == KindNewHead && !ev.Block.Zero() && ev.SourceId != "" { - ns.handle.SuggestLatestBlock(ev.SourceId, ev.Block.Number, ev.Payload) + ns.handle.SuggestLatestBlock(ev.SourceId, ev.Block.Number) } // Dedup. diff --git a/indexer/indexer_test.go b/indexer/indexer_test.go index d965d9590..b586fb935 100644 --- a/indexer/indexer_test.go +++ b/indexer/indexer_test.go @@ -31,7 +31,7 @@ func newFakeNetwork(id string, depth int64) *fakeNetwork { func (n *fakeNetwork) Id() string { return n.id } func (n *fakeNetwork) FinalityDepth() int64 { return n.finalityDepth } -func (n *fakeNetwork) SuggestLatestBlock(sourceId string, block int64, _ []byte) { +func (n *fakeNetwork) SuggestLatestBlock(sourceId string, block int64) { n.mu.Lock() n.suggestedBy[sourceId] = append(n.suggestedBy[sourceId], block) n.mu.Unlock() diff --git a/indexer/ingress.go b/indexer/ingress.go index 80bd54657..d33f57705 100644 --- a/indexer/ingress.go +++ b/indexer/ingress.go @@ -26,13 +26,12 @@ type NetworkHandle interface { FinalityDepth() int64 // SuggestLatestBlock advances the per-source latest-block tracker // (and the network-level latest tip) before the indexer dedupes / - // fans out. headPayload is the newHeads eth_subscription result - // (block header JSON); it may be nil when the caller only has a - // number. Preserving "update-before-dedup" and "tip-before-fanout" - // ordering is critical — the state poller needs to see every - // observation (even ones we drop), and HTTP "latest" must not be - // allowed to regress below a head we are about to deliver on WS. - SuggestLatestBlock(sourceId string, blockNumber int64, headPayload []byte) + // fans out. Preserving "update-before-dedup" and + // "tip-before-fanout" ordering is critical — the state poller needs + // to see every observation (even ones we drop), and HTTP "latest" + // must not be allowed to regress below a head we are about to + // deliver on WS. + SuggestLatestBlock(sourceId string, blockNumber int64) } // EventIngress is an adapter that converts some transport-specific diff --git a/indexer/integration_test.go b/indexer/integration_test.go index 9be2c01fb..3eec40b0a 100644 --- a/indexer/integration_test.go +++ b/indexer/integration_test.go @@ -107,7 +107,7 @@ type stubNetwork struct { func (s *stubNetwork) Id() string { return s.id } func (s *stubNetwork) FinalityDepth() int64 { return s.finality } -func (s *stubNetwork) SuggestLatestBlock(sourceID string, block int64, _ []byte) { +func (s *stubNetwork) SuggestLatestBlock(sourceID string, block int64) { s.mu.Lock() if s.suggestions == nil { s.suggestions = make(map[string][]int64) From 0b81d3e48453c2037fa14415726af48bcab3e03d Mon Sep 17 00:00:00 2001 From: Jonny Date: Thu, 23 Jul 2026 10:22:06 +0100 Subject: [PATCH 24/40] fix(ws): bypass poller debounce when tip gate needs a fresh head Availability checks and EnforceHighestBlock were calling PollLatestBlockNumber, which can reuse a debounced tip behind network TipHW (WS/Redis), falsely rejecting eth_call and fail-opening stale latest. Add PollLatestBlockNumberNow and use it on those paths; force-poll the leader before pinning the tip re-fetch. Co-authored-by: Cursor --- architecture/evm/eth_getBlockByNumber.go | 15 +++++++++++---- architecture/evm/evm_state_poller.go | 15 +++++++++++++-- common/architecture_evm.go | 4 ++++ common/upstream_fake.go | 4 ++++ erpc/networks.go | 2 +- upstream/upstream.go | 8 ++++++-- upstream/upstream_block_availability_test.go | 6 ++++++ upstream/upstream_test.go | 6 ++++++ 8 files changed, 51 insertions(+), 9 deletions(-) diff --git a/architecture/evm/eth_getBlockByNumber.go b/architecture/evm/eth_getBlockByNumber.go index 66916565b..9ff883138 100644 --- a/architecture/evm/eth_getBlockByNumber.go +++ b/architecture/evm/eth_getBlockByNumber.go @@ -195,12 +195,19 @@ func enforceHighestBlock(ctx context.Context, network common.Network, nq *common dr.SkipCacheRead = "true" // Prefer the upstream whose poller already owns this tip // (EvmLeaderUpstream — typically the WS ingress that called - // SuggestLatestBlock). Fall back to excluding the stale - // responder when no local poller has caught up yet. + // SuggestLatestBlock). If TipHW advanced via Redis/WS while + // local pollers lag inside their debounce window, force-poll + // the leader once before deciding. Fall back to excluding the + // stale responder when no local poller has caught up yet. if leader := network.EvmLeaderUpstream(ctx); leader != nil { if eu, ok := leader.(common.EvmUpstream); ok { - if sp := eu.EvmStatePoller(); sp != nil && !sp.IsObjectNull() && sp.LatestBlock() >= highestBlockNumber { - dr.UseUpstream = leader.Id() + if sp := eu.EvmStatePoller(); sp != nil && !sp.IsObjectNull() { + if sp.LatestBlock() < highestBlockNumber { + _, _ = sp.PollLatestBlockNumberNow(ctx) + } + if sp.LatestBlock() >= highestBlockNumber { + dr.UseUpstream = leader.Id() + } } } } diff --git a/architecture/evm/evm_state_poller.go b/architecture/evm/evm_state_poller.go index a026cdcf9..9f6746319 100644 --- a/architecture/evm/evm_state_poller.go +++ b/architecture/evm/evm_state_poller.go @@ -380,21 +380,32 @@ func (e *EvmStatePoller) resolveDebounce(cfg *common.EvmNetworkConfig) time.Dura // PollLatestBlockNumber fetches the latest block number in a blocking manner. // Respects the debounce interval if configured (if the last poll happened too recently, it reuses the cached value). func (e *EvmStatePoller) PollLatestBlockNumber(ctx context.Context) (int64, error) { + e.stateMu.RLock() + cfg := e.cfg + e.stateMu.RUnlock() + return e.pollLatestBlockNumber(ctx, e.resolveDebounce(cfg)) +} + +// PollLatestBlockNumberNow fetches the latest block number, bypassing debounce. +func (e *EvmStatePoller) PollLatestBlockNumberNow(ctx context.Context) (int64, error) { + return e.pollLatestBlockNumber(ctx, 0) +} + +func (e *EvmStatePoller) pollLatestBlockNumber(ctx context.Context, dbi time.Duration) (int64, error) { if e.shouldSkipLatestBlockCheck() { e.logger.Trace().Msg("skipping latest block number poll as it is not supported by the upstream") return 0, nil } e.stateMu.RLock() - cfg := e.cfg networkLabel := e.networkLabel e.stateMu.RUnlock() - dbi := e.resolveDebounce(cfg) e.logger.Trace().Int64("debounceMs", dbi.Milliseconds()).Msg("attempt to poll latest block number") ctx, span := common.StartDetailSpan(ctx, "EvmStatePoller.PollLatestBlockNumber", trace.WithAttributes( attribute.String("upstream.id", e.upstream.Id()), attribute.String("network.id", e.upstream.NetworkId()), + attribute.Int64("debounce_ms", dbi.Milliseconds()), ), ) defer span.End() diff --git a/common/architecture_evm.go b/common/architecture_evm.go index c9d1e7baf..89002aece 100644 --- a/common/architecture_evm.go +++ b/common/architecture_evm.go @@ -105,6 +105,10 @@ type EvmStatePoller interface { Bootstrap(ctx context.Context) error Poll(ctx context.Context) error PollLatestBlockNumber(ctx context.Context) (int64, error) + // PollLatestBlockNumberNow fetches latest ignoring the poller debounce. + // Used when a request needs a block ahead of the cached tip — debounce + // would otherwise reuse a stale value and falsely reject the upstream. + PollLatestBlockNumberNow(ctx context.Context) (int64, error) PollFinalizedBlockNumber(ctx context.Context) (int64, error) PollEarliestBlockNumber(ctx context.Context, probe EvmAvailabilityProbeType, staleness time.Duration) (int64, error) SyncingState() EvmSyncingState diff --git a/common/upstream_fake.go b/common/upstream_fake.go index 26b0f41c9..7a21b2695 100644 --- a/common/upstream_fake.go +++ b/common/upstream_fake.go @@ -251,6 +251,10 @@ func (p *FakeEvmStatePoller) PollLatestBlockNumber(ctx context.Context) (int64, return p.latestBlockNumber, nil } +func (p *FakeEvmStatePoller) PollLatestBlockNumberNow(ctx context.Context) (int64, error) { + return p.PollLatestBlockNumber(ctx) +} + func (p *FakeEvmStatePoller) SetNetworkConfig(config *NetworkConfig) { // No-op for testing } diff --git a/erpc/networks.go b/erpc/networks.go index 6a88d3241..a19a3c283 100644 --- a/erpc/networks.go +++ b/erpc/networks.go @@ -1595,7 +1595,7 @@ func (n *Network) handleBlockSkip( go func() { pollCtx, cancel := context.WithTimeout(context.Background(), 10*time.Second) defer cancel() - _, _ = sp.PollLatestBlockNumber(pollCtx) + _, _ = sp.PollLatestBlockNumberNow(pollCtx) }() } } diff --git a/upstream/upstream.go b/upstream/upstream.go index 112547afd..d7b294636 100644 --- a/upstream/upstream.go +++ b/upstream/upstream.go @@ -1081,10 +1081,14 @@ func (u *Upstream) EvmAssertBlockAvailability(ctx context.Context, forMethod str // UPPER BOUND: Check if block is before the latest block // latestBlock := statePoller.LatestBlock() - // If the requested block is beyond the current latest block, try force-polling once + // If the requested block is beyond the current latest block, force-poll + // once with debounce bypassed. A debounced PollLatestBlockNumber can + // reuse a tip that is still behind the request (common when network + // TipHW advanced via WS/Redis while this upstream's poller has not + // refreshed yet) and falsely trip the upper-bound gate. if blockNumber > latestBlock && forceFreshIfStale { var err error - latestBlock, err = statePoller.PollLatestBlockNumber(ctx) + latestBlock, err = statePoller.PollLatestBlockNumberNow(ctx) if err != nil { return false, fmt.Errorf("failed to poll latest block number: %w", err) } diff --git a/upstream/upstream_block_availability_test.go b/upstream/upstream_block_availability_test.go index dcd51f010..74957f839 100644 --- a/upstream/upstream_block_availability_test.go +++ b/upstream/upstream_block_availability_test.go @@ -37,6 +37,9 @@ func (m *mockEvmStatePollerEnhanced) PollLatestBlockNumber(ctx context.Context) } return m.latestBlock, nil } +func (m *mockEvmStatePollerEnhanced) PollLatestBlockNumberNow(ctx context.Context) (int64, error) { + return m.PollLatestBlockNumber(ctx) +} func (m *mockEvmStatePollerEnhanced) PollFinalizedBlockNumber(ctx context.Context) (int64, error) { if m.pollError != nil { return 0, m.pollError @@ -603,6 +606,9 @@ func (m *mockEvmStatePollerWithCustomBehavior) Poll(ctx context.Context) error func (m *mockEvmStatePollerWithCustomBehavior) PollLatestBlockNumber(ctx context.Context) (int64, error) { return m.pollLatestBlockNumber(ctx) } +func (m *mockEvmStatePollerWithCustomBehavior) PollLatestBlockNumberNow(ctx context.Context) (int64, error) { + return m.PollLatestBlockNumber(ctx) +} func (m *mockEvmStatePollerWithCustomBehavior) PollFinalizedBlockNumber(ctx context.Context) (int64, error) { return m.finalizedBlock, nil } diff --git a/upstream/upstream_test.go b/upstream/upstream_test.go index fd1451e6b..e02c40b4c 100644 --- a/upstream/upstream_test.go +++ b/upstream/upstream_test.go @@ -283,6 +283,9 @@ func (m *mockEvmStatePoller) Poll(ctx context.Context) error { return nil } func (m *mockEvmStatePoller) PollLatestBlockNumber(ctx context.Context) (int64, error) { return m.latestBlock, nil } +func (m *mockEvmStatePoller) PollLatestBlockNumberNow(ctx context.Context) (int64, error) { + return m.PollLatestBlockNumber(ctx) +} func (m *mockEvmStatePoller) PollFinalizedBlockNumber(ctx context.Context) (int64, error) { return m.finalizedBlock, nil } @@ -597,6 +600,9 @@ func (m *mockEvmStatePollerWithUpdate) PollLatestBlockNumber(ctx context.Context m.hasPolled = true return m.polledLatest, nil } +func (m *mockEvmStatePollerWithUpdate) PollLatestBlockNumberNow(ctx context.Context) (int64, error) { + return m.PollLatestBlockNumber(ctx) +} func (m *mockEvmStatePollerWithUpdate) PollFinalizedBlockNumber(ctx context.Context) (int64, error) { return m.polledLatest - 10, nil } From 8d21290b603a8cfe91115e19549bd1359226382f Mon Sep 17 00:00:00 2001 From: Jonny Date: Thu, 23 Jul 2026 10:57:37 +0100 Subject: [PATCH 25/40] fix(ws): refuse stale latest when tip re-fetch misses TipHW EnforceHighestBlock used pickHighestBlock, which fail-opened to a lagging "latest" when the concrete TipHW fetch returned null/error. That is the MultiNode FOOS trigger after WS newHeads already delivered the higher head. Re-fetch tip (leader pin, then unconstrained), accept only responses that meet the tip floor, and return an error instead of stale. Also fail-open the per-upstream availability gate when poller lags TipHW, and only skip enforcement for cached latest that already meets the tip. Co-authored-by: Cursor --- architecture/evm/eth_getBlockByNumber.go | 308 +++++++++++++---------- erpc/http_server_test.go | 22 +- erpc/http_server_ws_tip_leader_test.go | 129 +++++++++- erpc/networks.go | 17 ++ 4 files changed, 342 insertions(+), 134 deletions(-) diff --git a/architecture/evm/eth_getBlockByNumber.go b/architecture/evm/eth_getBlockByNumber.go index 9ff883138..a3f6cffae 100644 --- a/architecture/evm/eth_getBlockByNumber.go +++ b/architecture/evm/eth_getBlockByNumber.go @@ -121,15 +121,22 @@ func enforceHighestBlock(ctx context.Context, network common.Network, nq *common logger := network.Logger().With().Str("method", "eth_getBlockByNumber").Logger() - // If response is from cache, skip enforcement otherwise there's no point in caching. - // As we'll definetely have higher latest block number vs what we have in cache. - // The correct way to deal with this situation is to set proper TTL for "realtime" cache policy. + // Cached "latest" can lag TipHW across pods / tip races. Only skip + // enforcement when the cached payload already meets the tip floor. if nr.FromCache() { - logger.Trace(). - Object("request", nq). - Object("response", nr). - Msg("skipping enforcement of highest block number as response is from cache") - return nr, re + highestBlockNumber := network.EvmHighestLatestBlockNumber(ctx) + _, cachedBN, cerr := ExtractBlockReferenceFromResponse(ctx, nr) + if cerr == nil && cachedBN >= highestBlockNumber { + logger.Trace(). + Object("request", nq). + Object("response", nr). + Msg("skipping enforcement of highest block number as cached response meets tip") + return nr, re + } + logger.Debug(). + Int64("highestBlockNumber", highestBlockNumber). + Int64("cachedBlockNumber", cachedBN). + Msg("cached latest lags tip; enforcing highest block") } rqj, err := nq.JsonRpcRequest(ctx) @@ -137,15 +144,21 @@ func enforceHighestBlock(ctx context.Context, network common.Network, nq *common return nil, err } rqj.RLock() - defer rqj.RUnlock() - if len(rqj.Params) < 1 { + rqj.RUnlock() return nr, re } bnp, ok := rqj.Params[0].(string) if !ok { + rqj.RUnlock() return nr, re } + var itx bool + if len(rqj.Params) > 1 { + itx, _ = rqj.Params[1].(bool) + } + rqj.RUnlock() + if bnp != "latest" && bnp != "finalized" { return nr, re } @@ -157,135 +170,133 @@ func enforceHighestBlock(ctx context.Context, network common.Network, nq *common if err != nil { return nil, err } - if highestBlockNumber > respBlockNumber { - logger.Debug(). - Str("blockTag", bnp). - Object("request", nq). - Object("response", nr). - Interface("highestBlockNumber", highestBlockNumber). - Interface("respBlockNumber", respBlockNumber). - Interface("err", err). - Msg("enforcing highest latest block") - if respBlockNumber > 0 { - // When extracted block number is 0, it mostly means response is actually a json-rpc error - // therefore we better fetch the highest block number again. - ups := nr.Upstream() - telemetry.MetricUpstreamStaleLatestBlock.WithLabelValues( - network.ProjectId(), - ups.VendorName(), - network.Label(), - ups.Id(), - "eth_getBlockByNumber", - ).Inc() - } - var itx bool - if len(rqj.Params) > 1 { - itx, _ = rqj.Params[1].(bool) - } - request, err := BuildGetBlockByNumberRequest(highestBlockNumber, itx) - if err != nil { - return nil, err - } - err = request.SetID(nq.ID()) - if err != nil { - return nil, err - } - newReq := common.NewNormalizedRequestFromJsonRpcRequest(request) - dr := nq.Directives().Clone() - dr.SkipCacheRead = "true" - // Prefer the upstream whose poller already owns this tip - // (EvmLeaderUpstream — typically the WS ingress that called - // SuggestLatestBlock). If TipHW advanced via Redis/WS while - // local pollers lag inside their debounce window, force-poll - // the leader once before deciding. Fall back to excluding the - // stale responder when no local poller has caught up yet. - if leader := network.EvmLeaderUpstream(ctx); leader != nil { - if eu, ok := leader.(common.EvmUpstream); ok { - if sp := eu.EvmStatePoller(); sp != nil && !sp.IsObjectNull() { - if sp.LatestBlock() < highestBlockNumber { - _, _ = sp.PollLatestBlockNumberNow(ctx) - } - if sp.LatestBlock() >= highestBlockNumber { - dr.UseUpstream = leader.Id() - } + if highestBlockNumber <= respBlockNumber { + return nr, re + } + logger.Debug(). + Str("blockTag", bnp). + Object("request", nq). + Object("response", nr). + Interface("highestBlockNumber", highestBlockNumber). + Interface("respBlockNumber", respBlockNumber). + Msg("enforcing highest latest block") + if respBlockNumber > 0 { + ups := nr.Upstream() + telemetry.MetricUpstreamStaleLatestBlock.WithLabelValues( + network.ProjectId(), + ups.VendorName(), + network.Label(), + ups.Id(), + "eth_getBlockByNumber", + ).Inc() + } + + // Prefer the upstream whose poller already owns this tip + // (EvmLeaderUpstream — typically the WS ingress that called + // SuggestLatestBlock). If TipHW advanced via Redis/WS while + // local pollers lag inside their debounce window, force-poll + // the leader once before deciding. Fall back to excluding the + // stale responder when no local poller has caught up yet. + useUpstream := "" + if leader := network.EvmLeaderUpstream(ctx); leader != nil { + if eu, ok := leader.(common.EvmUpstream); ok { + if sp := eu.EvmStatePoller(); sp != nil && !sp.IsObjectNull() { + if sp.LatestBlock() < highestBlockNumber { + _, _ = sp.PollLatestBlockNumberNow(ctx) + } + if sp.LatestBlock() >= highestBlockNumber { + useUpstream = leader.Id() } } } - if dr.UseUpstream == "" && respBlockNumber > 0 { - dr.UseUpstream = fmt.Sprintf("!%s", nr.UpstreamId()) - } - newReq.SetDirectives(dr) - newReq.SetNetwork(network) - - // Copy HTTP context (headers, query parameters, user) for proper metrics tracking - newReq.CopyHttpContextFrom(nq) + } + if useUpstream == "" && respBlockNumber > 0 { + useUpstream = fmt.Sprintf("!%s", nr.UpstreamId()) + } - nnr, err := network.Forward(ctx, newReq) - // This is needed in case highest block number is corrupted somehow and for example - // it is requesting a very high non-existent block number. - return pickHighestBlock(ctx, nnr, nr, err) - } else { - return nr, re + // Do not use pickHighestBlock against the stale "latest" response — + // that helper fail-opens to stale when the tip re-fetch misses, which + // is exactly the MultiNode FOOS / EnforceRepeatableRead trigger. + nnr, ferr := forwardGetBlockByNumber(ctx, network, nq, highestBlockNumber, itx, useUpstream) + if meetsTipFloor(ctx, nnr, highestBlockNumber) { + if nr != nil { + nr.Release() + } + return nnr, nil + } + if nnr != nil { + nnr.Release() } + + // Pinned / excluded re-fetch missed the tip (sibling fullnode + // lag, WS JSON-RPC miss, etc.). Retry with no UseUpstream pin + // so every upstream can serve the concrete TipHW block. + nnr2, ferr2 := forwardGetBlockByNumber(ctx, network, nq, highestBlockNumber, itx, "") + if meetsTipFloor(ctx, nnr2, highestBlockNumber) { + if nr != nil { + nr.Release() + } + return nnr2, nil + } + if nnr2 != nil { + nnr2.Release() + } + + // NEVER fail-open to a tip below TipHW. Prefer an error over stale. + logger.Warn(). + Int64("highestBlockNumber", highestBlockNumber). + Int64("staleBlockNumber", respBlockNumber). + Err(ferr2). + Msg("tip re-fetch could not reach TipHW; refusing stale latest") + if nr != nil { + nr.Release() + } + if ferr2 != nil { + return nil, ferr2 + } + if ferr != nil { + return nil, ferr + } + details := map[string]interface{}{"blockNumber": highestBlockNumber} + return nil, common.NewErrEndpointMissingData( + common.NewErrJsonRpcExceptionInternal( + 0, + common.JsonRpcErrorMissingData, + fmt.Sprintf("block not found with number %d", highestBlockNumber), + nil, + details, + ), + nil, + ) case "finalized": highestBlockNumber := network.EvmHighestFinalizedBlockNumber(ctx) _, respBlockNumber, err := ExtractBlockReferenceFromResponse(ctx, nr) if err != nil { return nil, err } - if highestBlockNumber > respBlockNumber { - logger.Debug(). - Str("blockTag", bnp). - Interface("highestBlockNumber", highestBlockNumber). - Interface("respBlockNumber", respBlockNumber). - Interface("err", err). - Msg("enforcing highest finalized block") - if respBlockNumber > 0 { - // When extracted block number is 0, it mostly means response is actually a json-rpc error - // therefore we better fetch the highest block number again. - ups := nr.Upstream() - telemetry.MetricUpstreamStaleFinalizedBlock.WithLabelValues( - network.ProjectId(), - ups.VendorName(), - network.Label(), - ups.Id(), - ).Inc() - } - var itx bool - if len(rqj.Params) > 1 { - itx, _ = rqj.Params[1].(bool) - } - request, err := BuildGetBlockByNumberRequest(highestBlockNumber, itx) - if err != nil { - return nil, err - } - err = request.SetID(nq.ID()) - if err != nil { - return nil, err - } - newReq2 := common.NewNormalizedRequestFromJsonRpcRequest(request) - dr := nq.Directives().Clone() - dr.SkipCacheRead = "true" - if respBlockNumber > 0 { - // In case a block number is extracted, it means the node actually has an older latest block. - // Therefore we exclude the current upstream from the request (as high likely it doesn't have this block). - // Otherwise we still allow the current upstream to be used in case json-rpc error was an intermittent issue. - // Also, if response from cache we don't need to exclude the current upstream. - dr.UseUpstream = fmt.Sprintf("!%s", nr.UpstreamId()) - } - newReq2.SetDirectives(dr) - newReq2.SetNetwork(network) - - // Copy HTTP context (headers, query parameters, user) for proper metrics tracking - newReq2.CopyHttpContextFrom(nq) - - nnr, err := network.Forward(ctx, newReq2) - // This is needed in case highest block number is corrupted somehow and for example - // it is requesting a very high non-existent block number. - return pickHighestBlock(ctx, nnr, nr, err) - } else { + if highestBlockNumber <= respBlockNumber { return nr, re } + logger.Debug(). + Str("blockTag", bnp). + Interface("highestBlockNumber", highestBlockNumber). + Interface("respBlockNumber", respBlockNumber). + Msg("enforcing highest finalized block") + if respBlockNumber > 0 { + ups := nr.Upstream() + telemetry.MetricUpstreamStaleFinalizedBlock.WithLabelValues( + network.ProjectId(), + ups.VendorName(), + network.Label(), + ups.Id(), + ).Inc() + } + useUpstream := "" + if respBlockNumber > 0 { + useUpstream = fmt.Sprintf("!%s", nr.UpstreamId()) + } + nnr, err := forwardGetBlockByNumber(ctx, network, nq, highestBlockNumber, itx, useUpstream) + return pickHighestBlock(ctx, nnr, nr, err) default: return nr, re } @@ -339,6 +350,50 @@ func enforceNonNullBlock(nq *common.NormalizedRequest, nr *common.NormalizedResp ) } +func forwardGetBlockByNumber( + ctx context.Context, + network common.Network, + original *common.NormalizedRequest, + blockNumber int64, + includeTx bool, + useUpstream string, +) (*common.NormalizedResponse, error) { + request, err := BuildGetBlockByNumberRequest(blockNumber, includeTx) + if err != nil { + return nil, err + } + if err := request.SetID(original.ID()); err != nil { + return nil, err + } + newReq := common.NewNormalizedRequestFromJsonRpcRequest(request) + dr := original.Directives().Clone() + dr.SkipCacheRead = "true" + dr.UseUpstream = useUpstream + newReq.SetDirectives(dr) + newReq.SetNetwork(network) + newReq.CopyHttpContextFrom(original) + return network.Forward(ctx, newReq) +} + +// meetsTipFloor reports whether resp carries a block number >= minBlock. +func meetsTipFloor(ctx context.Context, resp *common.NormalizedResponse, minBlock int64) bool { + if resp == nil || resp.IsObjectNull() || resp.IsResultEmptyish() || minBlock <= 0 { + return false + } + // Peek number directly — ExtractBlockReferenceFromResponse can fail on + // incomplete header fields (hash/parentHash) even when number is present. + jrr, err := resp.JsonRpcResponse(ctx) + if err != nil || jrr == nil { + return false + } + numStr, err := jrr.PeekStringByPath(ctx, "number") + if err != nil || numStr == "" { + return false + } + bn, err := common.HexToInt64(numStr) + return err == nil && bn >= minBlock +} + func pickHighestBlock(ctx context.Context, x *common.NormalizedResponse, y *common.NormalizedResponse, err error) (*common.NormalizedResponse, error) { ctx, span := common.StartDetailSpan(ctx, "Evm.PickHighestBlock") defer span.End() @@ -761,3 +816,4 @@ func validateBlockTransactions(u common.Upstream, dirs *common.RequestDirectives return nil } + diff --git a/erpc/http_server_test.go b/erpc/http_server_test.go index a61e5ca61..3b5aa7912 100644 --- a/erpc/http_server_test.go +++ b/erpc/http_server_test.go @@ -6235,13 +6235,18 @@ func TestHttpServer_EvmGetBlockByNumber(t *testing.T) { statusCode, _, body := sendRequest(requestBody, nil, nil) - assert.Equal(t, http.StatusOK, statusCode) - var respObject map[string]interface{} err = sonic.UnmarshalString(body, &respObject) assert.NoError(t, err, "should parse response body successfully") - assert.Contains(t, body, "0x123") + // Tip re-fetch of TipHW (0x777) misses; must not fail-open to stale 0x123. + if result, ok := respObject["result"].(map[string]interface{}); ok { + assert.NotEqual(t, "0x123", result["number"], + "must not fail-open to stale latest below TipHW; body=%s", body) + } + _, hasErr := respObject["error"] + assert.True(t, hasErr || statusCode >= 400, + "expected error when tip re-fetch cannot reach TipHW, got status=%d body=%s", statusCode, body) }) @@ -6427,13 +6432,18 @@ func TestHttpServer_EvmGetBlockByNumber(t *testing.T) { statusCode, _, body := sendRequest(requestBody, nil, nil) - assert.Equal(t, http.StatusOK, statusCode) - var respObject map[string]interface{} err = sonic.UnmarshalString(body, &respObject) assert.NoError(t, err, "should parse response body successfully") - assert.Contains(t, body, "0x123") + // Tip re-fetch of TipHW (0x777) misses; must not fail-open to stale 0x123. + if result, ok := respObject["result"].(map[string]interface{}); ok { + assert.NotEqual(t, "0x123", result["number"], + "must not fail-open to stale latest below TipHW; body=%s", body) + } + _, hasErr := respObject["error"] + assert.True(t, hasErr || statusCode >= 400, + "expected error when tip re-fetch cannot reach TipHW, got status=%d body=%s", statusCode, body) }) diff --git a/erpc/http_server_ws_tip_leader_test.go b/erpc/http_server_ws_tip_leader_test.go index 1dbb56672..5dda808b3 100644 --- a/erpc/http_server_ws_tip_leader_test.go +++ b/erpc/http_server_ws_tip_leader_test.go @@ -46,7 +46,7 @@ func TestHttpServer_GetBlockByNumberLatest_RefetchPinsEvmLeaderUpstream(t *testi return true }). Reply(200). - JSON([]byte(`{"result":{"number":"0x22228889","hash":"0xleader","parentHash":"0xparent","timestamp":"0x6702a8f1"}}`)) + JSON([]byte(`{"result":{"number":"0x22228889","hash":"0xaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa","parentHash":"0xbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb","timestamp":"0x6702a8f1"}}`)) cfg := &common.Config{ Server: &common.ServerConfig{ @@ -140,7 +140,132 @@ func TestHttpServer_GetBlockByNumberLatest_RefetchPinsEvmLeaderUpstream(t *testi result, ok := respObject["result"].(map[string]interface{}) require.True(t, ok, "response should have a result object, got: %s", body) assert.Equal(t, tipHex, result["number"]) - assert.Equal(t, "0xleader", result["hash"]) + assert.Equal(t, "0xaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", result["hash"]) assert.GreaterOrEqual(t, leaderHits.Load(), int64(1), "EnforceHighestBlock must pin the tip re-fetch to EvmLeaderUpstream") } + +// When TipHW is ahead of every upstream's concrete block response, +// EnforceHighestBlock must NOT fail-open to the stale "latest" — that is +// the MultiNode FOOS trigger once WS has already delivered the higher head. +func TestHttpServer_GetBlockByNumberLatest_RefusesStaleFailOpen(t *testing.T) { + util.ResetGock() + defer util.ResetGock() + util.SetupMocksForEvmStatePoller() + // Two Persist tip-null mocks remain pending by design. + defer util.AssertNoPendingMocks(t, 2) + + const tip = int64(0x22228889) + tipHex := "0x22228889" + staleHex := "0x22228888" + + // Tip re-fetch always misses (null) — pinned and unconstrained paths. + gock.New("http://rpc1.localhost"). + Post(""). + Persist(). + Filter(func(r *http.Request) bool { + body := util.SafeReadBody(r) + return strings.Contains(body, "eth_getBlockByNumber") && strings.Contains(body, tipHex) + }). + Reply(200). + JSON([]byte(`{"result":null}`)) + gock.New("http://rpc2.localhost"). + Post(""). + Persist(). + Filter(func(r *http.Request) bool { + body := util.SafeReadBody(r) + return strings.Contains(body, "eth_getBlockByNumber") && strings.Contains(body, tipHex) + }). + Reply(200). + JSON([]byte(`{"result":null}`)) + + cfg := &common.Config{ + Server: &common.ServerConfig{ + MaxTimeout: common.Duration(100 * time.Second).Ptr(), + }, + Projects: []*common.ProjectConfig{ + { + Id: "test_project", + Networks: []*common.NetworkConfig{ + { + Architecture: "evm", + Evm: &common.EvmNetworkConfig{ + ChainId: 123, + Integrity: &common.EvmIntegrityConfig{ + EnforceHighestBlock: util.BoolPtr(true), + }, + }, + Failsafe: []*common.FailsafeConfig{ + { + Retry: &common.RetryPolicyConfig{MaxAttempts: 2}, + }, + }, + }, + }, + Upstreams: []*common.UpstreamConfig{ + { + Id: "rpc1", + Endpoint: "http://rpc1.localhost", + Type: common.UpstreamTypeEvm, + Evm: &common.EvmUpstreamConfig{ + ChainId: 123, + StatePollerInterval: common.Duration(10 * time.Second), + }, + }, + { + Id: "rpc2", + Endpoint: "http://rpc2.localhost", + Type: common.UpstreamTypeEvm, + Evm: &common.EvmUpstreamConfig{ + ChainId: 123, + StatePollerInterval: common.Duration(10 * time.Second), + }, + }, + }, + }, + }, + } + + sendRequest, _, _, shutdown, erpcInstance := createServerTestFixtures(cfg, t) + defer shutdown() + + prj, err := erpcInstance.GetProject("test_project") + require.NoError(t, err) + policy.OverrideAllForTest(prj.policyEngine) + policy.OverrideOrderForTest(prj.policyEngine, "evm:123", "rpc1", "rpc2") + + time.Sleep(500 * time.Millisecond) + + nw, err := prj.GetNetwork(context.Background(), "evm:123") + require.NoError(t, err) + + var leader *upstream.Upstream + for _, u := range nw.upstreamsRegistry.GetNetworkUpstreams(context.Background(), "evm:123") { + if u.Id() == "rpc2" { + leader = u + break + } + } + require.NotNil(t, leader) + leader.EvmStatePoller().SuggestLatestBlock(tip) + nw.NoteObservedLatestBlock(context.Background(), tip) + + statusCode, _, body := sendRequest(`{ + "jsonrpc": "2.0", + "id": 1, + "method": "eth_getBlockByNumber", + "params": ["latest", false] + }`, nil, nil) + + var respObject map[string]interface{} + require.NoError(t, sonic.UnmarshalString(body, &respObject)) + if result, ok := respObject["result"].(map[string]interface{}); ok { + require.NotEqual(t, staleHex, result["number"], + "must not fail-open to stale tip below TipHW; status=%d body=%s", statusCode, body) + require.NotEqual(t, tipHex, result["number"], + "tip was mocked as null; unexpected tip success: %s", body) + } + _, hasErr := respObject["error"] + require.True(t, hasErr || statusCode >= 400, + "expected error when tip re-fetch cannot reach TipHW, got status=%d body=%s", statusCode, body) +} diff --git a/erpc/networks.go b/erpc/networks.go index a19a3c283..b6631905e 100644 --- a/erpc/networks.go +++ b/erpc/networks.go @@ -1687,6 +1687,23 @@ func (n *Network) checkUpstreamBlockAvailability(ctx context.Context, u common.U finalizedBlock = sp.FinalizedBlock() } + // Poller lag behind network TipHW (WS/Redis) is not evidence the + // upstream node lacks the block — TipHW means some ingress on this + // network already observed it. Fail-open so tip eth_call / reads + // reach the node instead of cascading ErrUpstreamBlockUnavailable. + if bn > latestBlock && latestBlock > 0 { + if tip := n.EvmHighestLatestBlockNumber(ctx); tip >= bn { + n.logger.Debug(). + Str("upstreamId", u.Id()). + Int64("blockNumber", bn). + Int64("pollerLatest", latestBlock). + Int64("networkTip", tip). + Str("method", method). + Msg("poller lags TipHW; failing open block availability gate") + return nil, false + } + } + blockErr := common.NewErrUpstreamBlockUnavailable(u.Id(), bn, latestBlock, finalizedBlock) // Determine if this is retryable based on distance From 0387ea25ce1b1beef75c970fd0b57d7c176204b7 Mon Sep 17 00:00:00 2001 From: Jonny Date: Thu, 23 Jul 2026 14:03:57 +0100 Subject: [PATCH 26/40] fix(ws): sync TipHW publish and refresh on HTTP false-negative Cross-pod TipHW Redis push was async, so sibling pods could still serve a lower HTTP latest after WS newHeads advanced highestUserObservations, silently demoting MultiNode via FOOS. Publish TipHW before fan-out and refresh from Redis when local TipHW would skip EnforceHighestBlock. Co-authored-by: Cursor --- architecture/evm/eth_blockNumber.go | 6 + architecture/evm/eth_getBlockByNumber.go | 50 ++++++-- architecture/evm/eth_getBlockByNumber_test.go | 28 +++++ data/shared_state_variable.go | 107 ++++++++++++++++-- data/shared_state_variable_test.go | 101 +++++++++++++++++ erpc/networks.go | 22 +++- erpc/networks_ws_tip_test.go | 72 ++++++++++++ 7 files changed, 368 insertions(+), 18 deletions(-) diff --git a/architecture/evm/eth_blockNumber.go b/architecture/evm/eth_blockNumber.go index e0604fad8..e9dee8b95 100644 --- a/architecture/evm/eth_blockNumber.go +++ b/architecture/evm/eth_blockNumber.go @@ -56,6 +56,12 @@ func projectPreForward_eth_blockNumber(ctx context.Context, network common.Netwo // Step 3: collect the highest block from all EVM upstream pollers for this network highestBlock := network.EvmHighestLatestBlockNumber(ctx) + if highestBlock <= blockNumber { + // Same cross-pod TipHW lag guard as enforceHighestBlock("latest"). + if refreshed := refreshHighestLatestBlockNumber(ctx, network); refreshed > highestBlock { + highestBlock = refreshed + } + } if common.IsTracingDetailed { blockNumberLag := highestBlock - blockNumber if blockNumberLag < 0 { diff --git a/architecture/evm/eth_getBlockByNumber.go b/architecture/evm/eth_getBlockByNumber.go index a3f6cffae..1d2c0148c 100644 --- a/architecture/evm/eth_getBlockByNumber.go +++ b/architecture/evm/eth_getBlockByNumber.go @@ -13,6 +13,19 @@ import ( "go.opentelemetry.io/otel/trace" ) +// tipRefresher is implemented by *erpc.Network. Optional so test doubles that +// only stub EvmHighestLatestBlockNumber keep compiling. +type tipRefresher interface { + EvmRefreshHighestLatestBlockNumber(ctx context.Context) int64 +} + +func refreshHighestLatestBlockNumber(ctx context.Context, network common.Network) int64 { + if r, ok := network.(tipRefresher); ok { + return r.EvmRefreshHighestLatestBlockNumber(ctx) + } + return network.EvmHighestLatestBlockNumber(ctx) +} + func BuildGetBlockByNumberRequest(blockNumberOrTag interface{}, includeTransactions bool) (*common.JsonRpcRequest, error) { var bkt string var err error @@ -126,6 +139,11 @@ func enforceHighestBlock(ctx context.Context, network common.Network, nq *common if nr.FromCache() { highestBlockNumber := network.EvmHighestLatestBlockNumber(ctx) _, cachedBN, cerr := ExtractBlockReferenceFromResponse(ctx, nr) + if cerr == nil && cachedBN >= highestBlockNumber { + if refreshed := refreshHighestLatestBlockNumber(ctx, network); refreshed > highestBlockNumber { + highestBlockNumber = refreshed + } + } if cerr == nil && cachedBN >= highestBlockNumber { logger.Trace(). Object("request", nq). @@ -171,15 +189,31 @@ func enforceHighestBlock(ctx context.Context, network common.Network, nq *common return nil, err } if highestBlockNumber <= respBlockNumber { - return nr, re + // Local TipHW appears caught up — but sibling pods may have + // published a higher tip to Redis that this process has not yet + // adopted via async pubsub. Refresh once before skipping enforce; + // this is the cross-pod race that silently demotes MultiNode FOOS. + if refreshed := refreshHighestLatestBlockNumber(ctx, network); refreshed > highestBlockNumber { + highestBlockNumber = refreshed + } + if highestBlockNumber <= respBlockNumber { + return nr, re + } + logger.Debug(). + Str("blockTag", bnp). + Int64("highestBlockNumber", highestBlockNumber). + Int64("respBlockNumber", respBlockNumber). + Msg("tip refresh from remote raised TipHW; enforcing highest latest block") + } else { + logger.Debug(). + Str("blockTag", bnp). + Object("request", nq). + Object("response", nr). + Interface("highestBlockNumber", highestBlockNumber). + Interface("respBlockNumber", respBlockNumber). + Msg("enforcing highest latest block") } - logger.Debug(). - Str("blockTag", bnp). - Object("request", nq). - Object("response", nr). - Interface("highestBlockNumber", highestBlockNumber). - Interface("respBlockNumber", respBlockNumber). - Msg("enforcing highest latest block") + // fall through to tip re-fetch / refuse-stale (logger already emitted) if respBlockNumber > 0 { ups := nr.Upstream() telemetry.MetricUpstreamStaleLatestBlock.WithLabelValues( diff --git a/architecture/evm/eth_getBlockByNumber_test.go b/architecture/evm/eth_getBlockByNumber_test.go index 54c34d90d..1c77d6a01 100644 --- a/architecture/evm/eth_getBlockByNumber_test.go +++ b/architecture/evm/eth_getBlockByNumber_test.go @@ -70,6 +70,34 @@ func (t *testNetwork) GetFinality(ctx context.Context, req *common.NormalizedReq return common.DataFinalityStateFinalized } +// tipRefreshNetwork stubs local TipHW separately from a remote-refreshed TipHW +// so we can exercise the cross-pod false-negative refresh path. +type tipRefreshNetwork struct { + testNetwork + localTip int64 + remoteTip int64 +} + +func (n *tipRefreshNetwork) EvmHighestLatestBlockNumber(ctx context.Context) int64 { + return n.localTip +} + +func (n *tipRefreshNetwork) EvmRefreshHighestLatestBlockNumber(ctx context.Context) int64 { + return n.remoteTip +} + +func TestRefreshHighestLatestBlockNumber_UsesTipRefresher(t *testing.T) { + n := &tipRefreshNetwork{localTip: 1000, remoteTip: 1001} + got := refreshHighestLatestBlockNumber(context.Background(), n) + assert.Equal(t, int64(1001), got, "must prefer remote TipHW from tipRefresher") +} + +func TestRefreshHighestLatestBlockNumber_FallsBackWithoutRefresher(t *testing.T) { + n := &testNetwork{} + got := refreshHighestLatestBlockNumber(context.Background(), n) + assert.Equal(t, int64(0), got, "plain Network stubs use EvmHighestLatestBlockNumber") +} + func TestAllPhantomTransactions(t *testing.T) { t.Run("EmptySlice", func(t *testing.T) { assert.True(t, allPhantomTransactions(nil)) diff --git a/data/shared_state_variable.go b/data/shared_state_variable.go index 2d2ef1def..55c63173e 100644 --- a/data/shared_state_variable.go +++ b/data/shared_state_variable.go @@ -22,6 +22,16 @@ type CounterInt64SharedVariable interface { GetValue() int64 TryUpdateIfStale(ctx context.Context, staleness time.Duration, getNewValue func(ctx context.Context) (int64, error)) (int64, error) TryUpdate(ctx context.Context, newValue int64) int64 + // TryUpdateAndPublish advances the local counter then synchronously + // publishes SET+pubsub to remote so sibling pods can observe the tip + // before the caller fans out a WS newHeads notification. On publish + // failure it falls back to the async background push path. + TryUpdateAndPublish(ctx context.Context, newValue int64) int64 + // RefreshFromRemote performs a synchronous Redis GET and adopts the + // remote value when it is ahead of the local cache. Used on the HTTP + // tip-floor false-negative path (local TipHW appears caught up but a + // sibling pod already published a higher tip). + RefreshFromRemote(ctx context.Context) int64 OnValue(callback func(int64)) OnLargeRollback(callback func(currentVal, newVal int64)) } @@ -387,6 +397,78 @@ func (c *counterInt64) TryUpdate(ctx context.Context, newValue int64) int64 { return c.value.Load() } +// tipPublishTimeout bounds the synchronous TipHW SET+pubsub so WS fan-out +// never stalls hard on a slow Redis. Callers may pass a tighter deadline via ctx. +const tipPublishTimeout = 100 * time.Millisecond + +func (c *counterInt64) TryUpdateAndPublish(ctx context.Context, newValue int64) int64 { + ctx, span := common.StartSpan(ctx, "CounterInt64.TryUpdateAndPublish", + trace.WithAttributes( + attribute.String("key", c.key), + ), + ) + defer span.End() + if common.IsTracingDetailed { + span.SetAttributes( + attribute.Int64("new_value", newValue), + ) + } + + updated := c.processNewValue(UpdateSourceTryUpdate, newValue) + local := c.localState() + if local.UpdatedAt <= 0 { + return c.value.Load() + } + + pubCtx := ctx + if _, hasDeadline := ctx.Deadline(); !hasDeadline { + var cancel context.CancelFunc + pubCtx, cancel = context.WithTimeout(ctx, tipPublishTimeout) + defer cancel() + } + + if err := c.publishRemoteState(pubCtx, local); err != nil { + c.registry.logger.Debug().Err(err). + Str("key", c.key). + Int64("value", local.Value). + Msg("sync tip publish failed; falling back to background push") + c.scheduleBackgroundPushCurrent() + } else if updated { + // Sync publish succeeded for the new tip; still schedule background + // reconcile under the distributed lock so remote SET stays consistent. + c.scheduleBackgroundPushCurrent() + } + return c.value.Load() +} + +func (c *counterInt64) RefreshFromRemote(ctx context.Context) int64 { + ctx, span := common.StartSpan(ctx, "CounterInt64.RefreshFromRemote", + trace.WithAttributes( + attribute.String("key", c.key), + ), + ) + defer span.End() + + getCtx := ctx + if _, hasDeadline := ctx.Deadline(); !hasDeadline { + var cancel context.CancelFunc + getCtx, cancel = context.WithTimeout(ctx, tipPublishTimeout) + defer cancel() + } + + remote, ok := c.tryGetRemoteState(getCtx) + if !ok { + return c.value.Load() + } + if c.processNewState(UpdateSourceRemoteCheck, remote) { + c.registry.logger.Debug(). + Str("key", c.key). + Int64("value", remote.Value). + Msg("adopted higher tip from remote refresh") + } + return c.value.Load() +} + func (c *counterInt64) TryUpdateIfStale(ctx context.Context, staleness time.Duration, executeNewValueFn func(ctx context.Context) (int64, error)) (int64, error) { ctx, span := common.StartSpan(ctx, "CounterInt64.TryUpdateIfStale", trace.WithAttributes( @@ -610,6 +692,12 @@ func (c *counterInt64) tryGetRemoteState(ctx context.Context) (CounterInt64State } func (c *counterInt64) updateRemoteState(ctx context.Context, st CounterInt64State) { + _ = c.publishRemoteState(ctx, st) +} + +// publishRemoteState writes SET + pubsub for the counter. Returns an error +// when either step fails so sync tip publishers can fall back to the async path. +func (c *counterInt64) publishRemoteState(ctx context.Context, st CounterInt64State) error { if st.UpdatedBy == "" { st.UpdatedBy = c.registry.instanceId } @@ -619,7 +707,7 @@ func (c *counterInt64) updateRemoteState(ctx context.Context, st CounterInt64Sta Str("key", c.key). Int64("value", st.Value). Msg("failed to marshal counter state for remote update") - return + return err } err = c.registry.connector.Set(ctx, c.key, "value", payload, nil) @@ -631,14 +719,15 @@ func (c *counterInt64) updateRemoteState(ctx context.Context, st CounterInt64Sta Str("key", c.key). Int64("value", st.Value). Msg("failed to update remote counter value") - } else { - c.registry.logger.Debug(). - Str("key", c.key). - Int64("value", st.Value). - Int64("updatedAt", st.UpdatedAt). - Str("updatedBy", st.UpdatedBy). - Msg("published counter value to remote") - } + return err + } + c.registry.logger.Debug(). + Str("key", c.key). + Int64("value", st.Value). + Int64("updatedAt", st.UpdatedAt). + Str("updatedBy", st.UpdatedBy). + Msg("published counter value to remote") + return nil } // scheduleBackgroundPushCurrent dedupes and pushes the current local value to the diff --git a/data/shared_state_variable_test.go b/data/shared_state_variable_test.go index 4a706cb19..c179edb3b 100644 --- a/data/shared_state_variable_test.go +++ b/data/shared_state_variable_test.go @@ -1835,3 +1835,104 @@ func TestCounterInt64_FresherLocalPushesToStaleRemote(t *testing.T) { setCallMu.Unlock() }) } + +func TestCounterInt64_TryUpdateAndPublish_Sync(t *testing.T) { + registry, connector, ctx := setupTest("sync-pub") + + published := make(chan struct{}, 1) + connector.On("Set", mock.Anything, "test", "value", mock.Anything, mock.Anything). + Run(func(args mock.Arguments) { + select { + case published <- struct{}{}: + default: + } + }). + Return(nil) + connector.On("PublishCounterInt64", mock.Anything, "test", mock.Anything).Return(nil) + // Background reconcile may also Lock/Get after sync publish. + lock := &MockLock{} + lock.On("Unlock", mock.Anything).Return(nil).Maybe() + connector.On("Lock", mock.Anything, "test", mock.Anything).Return(lock, nil).Maybe() + connector.On("Get", mock.Anything, ConnectorMainIndex, "test", "value", nil). + Return([]byte(`{"v":10,"t":1,"b":"test"}`), nil).Maybe() + + counter := &counterInt64{ + registry: registry, + key: "test", + ignoreRollbackOf: 1024, + } + + got := counter.TryUpdateAndPublish(ctx, 10) + assert.Equal(t, int64(10), got) + + select { + case <-published: + // sync SET happened before return + case <-time.After(50 * time.Millisecond): + t.Fatal("expected synchronous Set before TryUpdateAndPublish returned") + } +} + +func TestCounterInt64_TryUpdateAndPublish_FallsBackOnPublishError(t *testing.T) { + registry, connector, ctx := setupTest("sync-pub-fail") + + connector.On("Set", mock.Anything, "test", "value", mock.Anything, mock.Anything). + Return(errors.New("redis down")) + // Async fallback still attempts publish/lock. + lock := &MockLock{} + lock.On("Unlock", mock.Anything).Return(nil).Maybe() + connector.On("Lock", mock.Anything, "test", mock.Anything).Return(lock, nil).Maybe() + connector.On("Get", mock.Anything, ConnectorMainIndex, "test", "value", nil). + Return([]byte(""), errors.New("get failed")).Maybe() + connector.On("PublishCounterInt64", mock.Anything, "test", mock.Anything).Return(nil).Maybe() + connector.On("Set", mock.Anything, "test", "value", mock.Anything, mock.Anything).Return(errors.New("redis down")).Maybe() + + counter := &counterInt64{ + registry: registry, + key: "test", + ignoreRollbackOf: 1024, + } + + got := counter.TryUpdateAndPublish(ctx, 42) + assert.Equal(t, int64(42), got, "local tip must advance even when sync publish fails") + time.Sleep(50 * time.Millisecond) // allow bg fallback to run +} + +func TestCounterInt64_RefreshFromRemote_AdoptsHigherTip(t *testing.T) { + registry, connector, ctx := setupTest("refresh-tip") + + remoteTs := time.Now().UnixMilli() + 60_000 + connector.On("Get", mock.Anything, ConnectorMainIndex, "test", "value", nil). + Return([]byte(fmt.Sprintf(`{"v":1001,"t":%d,"b":"pod-a"}`, remoteTs)), nil) + + counter := &counterInt64{ + registry: registry, + key: "test", + ignoreRollbackOf: 1024, + } + counter.value.Store(1000) + counter.updatedAtUnixMs.Store(time.Now().UnixMilli()) + + got := counter.RefreshFromRemote(ctx) + assert.Equal(t, int64(1001), got) + assert.Equal(t, int64(1001), counter.GetValue()) + connector.AssertExpectations(t) +} + +func TestCounterInt64_RefreshFromRemote_KeepsLocalWhenRemoteMissing(t *testing.T) { + registry, connector, ctx := setupTest("refresh-missing") + + connector.On("Get", mock.Anything, ConnectorMainIndex, "test", "value", nil). + Return([]byte(""), common.NewErrRecordNotFound("test", "value", "mock")) + + counter := &counterInt64{ + registry: registry, + key: "test", + ignoreRollbackOf: 1024, + } + counter.value.Store(777) + counter.updatedAtUnixMs.Store(time.Now().UnixMilli()) + + got := counter.RefreshFromRemote(ctx) + assert.Equal(t, int64(777), got) +} diff --git a/erpc/networks.go b/erpc/networks.go index b6631905e..ac6a71087 100644 --- a/erpc/networks.go +++ b/erpc/networks.go @@ -79,6 +79,10 @@ type Network struct { // notification to any client. Otherwise a concurrent HTTP // eth_getBlockByNumber("latest") / eth_blockNumber can race and return a // lower tip than a head already (or about to be) served on WS. +// +// TipHW is published to Redis synchronously (bounded timeout) so sibling +// pods can refresh TipHW before serving HTTP latest — closing the +// cross-pod race that silently demotes MultiNode via FOOS. func (n *Network) NoteObservedLatestBlock(ctx context.Context, blockNumber int64) { if n == nil || blockNumber <= 0 { return @@ -90,7 +94,7 @@ func (n *Network) NoteObservedLatestBlock(ctx context.Context, blockNumber int64 ctx = context.Background() } if n.latestBlockShared != nil { - n.latestBlockShared.TryUpdate(ctx, blockNumber) + n.latestBlockShared.TryUpdateAndPublish(ctx, blockNumber) } for { cur := n.lastReturnedLatestBlock.Load() @@ -103,6 +107,22 @@ func (n *Network) NoteObservedLatestBlock(ctx context.Context, blockNumber int64 } } +// EvmRefreshHighestLatestBlockNumber pulls TipHW from Redis once and returns +// the network tip after adopting any higher remote value. Used when the local +// TipHW cache would otherwise skip EnforceHighestBlock (false-negative under +// async TipHW pubsub lag). +func (n *Network) EvmRefreshHighestLatestBlockNumber(ctx context.Context) int64 { + ctx, span := common.StartDetailSpan(ctx, "Network.EvmRefreshHighestLatestBlockNumber") + defer span.End() + + if n.latestBlockShared != nil { + n.latestBlockShared.RefreshFromRemote(ctx) + } + result := n.EvmHighestLatestBlockNumber(ctx) + span.SetAttributes(attribute.Int64("highest_latest_block", result)) + return result +} + // Bootstrap registers this network with the policy engine. The engine kicks // off the slot's ticker and runs an initial synchronous eval so request-path // reads through `policyEngine.GetOrdered` always see a populated cache. diff --git a/erpc/networks_ws_tip_test.go b/erpc/networks_ws_tip_test.go index d4cf18009..fb9ea3f8b 100644 --- a/erpc/networks_ws_tip_test.go +++ b/erpc/networks_ws_tip_test.go @@ -190,3 +190,75 @@ func TestNetworkHandle_SuggestLatestBlock_AdvancesNetworkTipBeforeFanOut(t *test assert.GreaterOrEqual(t, network.lastReturnedLatestBlock.Load(), int64(90677359), "process-local high-water mark must cover the delivered WS tip") } + +// EvmRefreshHighestLatestBlockNumber must not regress the tip after +// NoteObservedLatestBlock (sync publish path) has advanced TipHW. +func TestEvmRefreshHighestLatestBlockNumber_PreservesObservedTip(t *testing.T) { + util.ResetGock() + defer util.ResetGock() + util.SetupMocksForEvmStatePoller() + + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + + up := &common.UpstreamConfig{ + Type: common.UpstreamTypeEvm, + Id: "rpc1", + Endpoint: "http://rpc1.localhost", + Evm: &common.EvmUpstreamConfig{ChainId: 123}, + } + + gock.New("http://rpc1.localhost"). + Post(""). + Persist(). + Filter(func(r *http.Request) bool { + return strings.Contains(util.SafeReadBody(r), `eth_chainId`) + }). + Reply(200). + JSON([]byte(`{"result":"0x7b"}`)) + + rateLimitersRegistry, _ := upstream.NewRateLimitersRegistry(context.Background(), &common.RateLimiterConfig{}, &log.Logger) + metricsTracker := health.NewTracker(&log.Logger, "test", time.Minute) + + vr := thirdparty.NewVendorsRegistry() + pr, err := thirdparty.NewProvidersRegistry(&log.Logger, vr, []*common.ProviderConfig{}, nil) + require.NoError(t, err) + + ssr, err := data.NewSharedStateRegistry(ctx, &log.Logger, &common.SharedStateConfig{ + Connector: &common.ConnectorConfig{ + Driver: "memory", + Memory: &common.MemoryConnectorConfig{MaxItems: 100_000, MaxTotalSize: "1GB"}, + }, + }) + require.NoError(t, err) + + upstreamsRegistry := upstream.NewUpstreamsRegistry( + ctx, &log.Logger, "test", + []*common.UpstreamConfig{up}, ssr, rateLimitersRegistry, vr, pr, nil, + metricsTracker, nil, + ) + + networkConfig := &common.NetworkConfig{ + Architecture: common.ArchitectureEvm, + Evm: &common.EvmNetworkConfig{ChainId: 123}, + } + network, err := NewNetwork(ctx, &log.Logger, "test", networkConfig, + rateLimitersRegistry, upstreamsRegistry, metricsTracker, nil) + require.NoError(t, err) + + upstreamsRegistry.Bootstrap(ctx) + time.Sleep(200 * time.Millisecond) + require.NoError(t, upstreamsRegistry.GetInitializer().WaitForTasks(ctx)) + require.NoError(t, network.Bootstrap(ctx)) + time.Sleep(250 * time.Millisecond) + + upsList := upstreamsRegistry.GetNetworkUpstreams(ctx, util.EvmNetworkId(123)) + require.Len(t, upsList, 1) + upsList[0].EvmStatePoller().SuggestLatestBlock(1000) + time.Sleep(50 * time.Millisecond) + + network.NoteObservedLatestBlock(ctx, 1001) + assert.Equal(t, int64(1001), network.EvmRefreshHighestLatestBlockNumber(ctx), + "refresh after sync TipHW publish must keep the observed tip") + assert.Equal(t, int64(1001), network.EvmHighestLatestBlockNumber(ctx)) +} From 28cd7f9c2e9ef274cd69379f84bf4e3159d930ae Mon Sep 17 00:00:00 2001 From: Jonny Date: Thu, 23 Jul 2026 14:31:12 +0100 Subject: [PATCH 27/40] fix(ws): serve cached newHeads header when tip re-fetch misses TipHW If TipHW advanced from a WS newHeads observation on this pod, HTTP eth_getBlockByNumber("latest", false) must return that header instead of hard-failing when concrete tip re-fetch cannot reach TipHW yet. Co-authored-by: Cursor --- architecture/evm/eth_getBlockByNumber.go | 57 ++++++++++- erpc/http_server_ws_tip_floor_test.go | 96 +++++++++++++++++++ erpc/networks.go | 58 +++++++++++ erpc/networks_ws_tip_test.go | 6 +- erpc/subscription_manager.go | 11 ++- .../wsupstream/adapter_reconnect_test.go | 7 +- indexer/indexer.go | 2 +- indexer/indexer_test.go | 10 +- indexer/ingress.go | 16 ++-- indexer/integration_test.go | 2 +- 10 files changed, 242 insertions(+), 23 deletions(-) create mode 100644 erpc/http_server_ws_tip_floor_test.go diff --git a/architecture/evm/eth_getBlockByNumber.go b/architecture/evm/eth_getBlockByNumber.go index 1d2c0148c..b06e18e78 100644 --- a/architecture/evm/eth_getBlockByNumber.go +++ b/architecture/evm/eth_getBlockByNumber.go @@ -26,6 +26,39 @@ func refreshHighestLatestBlockNumber(ctx context.Context, network common.Network return network.EvmHighestLatestBlockNumber(ctx) } +// observedLatestHeadProvider is implemented by networks that cache the last +// WS newHeads header before fan-out (erpc.Network). Used to serve HTTP +// "latest" when concrete tip re-fetch cannot reach TipHW. +type observedLatestHeadProvider interface { + LastObservedLatestHead() (blockNumber int64, payload []byte) +} + +// responseFromObservedLatestHead builds an eth_getBlockByNumber response from +// the cached WS newHeads header when it matches expectedTip. ok is false when +// the cache is missing, behind, or the network does not expose a tip cache. +func responseFromObservedLatestHead(network common.Network, nq *common.NormalizedRequest, expectedTip int64) (*common.NormalizedResponse, bool) { + provider, ok := network.(observedLatestHeadProvider) + if !ok || expectedTip <= 0 { + return nil, false + } + cachedNumber, payload := provider.LastObservedLatestHead() + if cachedNumber != expectedTip || len(payload) == 0 { + return nil, false + } + idBytes, err := common.SonicCfg.Marshal(nq.ID()) + if err != nil { + return nil, false + } + jrr, err := common.NewJsonRpcResponseFromBytes(idBytes, append([]byte(nil), payload...), nil) + if err != nil { + return nil, false + } + resp := common.NewNormalizedResponse(). + WithRequest(nq). + WithJsonRpcResponse(jrr) + return resp, true +} + func BuildGetBlockByNumberRequest(blockNumberOrTag interface{}, includeTransactions bool) (*common.JsonRpcRequest, error) { var bkt string var err error @@ -225,6 +258,18 @@ func enforceHighestBlock(ctx context.Context, network common.Network, nq *common ).Inc() } + // If we already have the TipHW header from WS newHeads on this pod, + // serve it for header-only latest. TipHW means we observed the block; + // failing the client because HTTP tip re-fetch lags is a bug. + if !itx { + if cached, ok := responseFromObservedLatestHead(network, nq, highestBlockNumber); ok { + if nr != nil { + nr.Release() + } + return cached, nil + } + } + // Prefer the upstream whose poller already owns this tip // (EvmLeaderUpstream — typically the WS ingress that called // SuggestLatestBlock). If TipHW advanced via Redis/WS while @@ -276,6 +321,17 @@ func enforceHighestBlock(ctx context.Context, network common.Network, nq *common nnr2.Release() } + // Tip re-fetch still missed — try the WS-cached header once more + // (TipHW may have been noted mid-flight after the early cache check). + if !itx { + if cached, ok := responseFromObservedLatestHead(network, nq, highestBlockNumber); ok { + if nr != nil { + nr.Release() + } + return cached, nil + } + } + // NEVER fail-open to a tip below TipHW. Prefer an error over stale. logger.Warn(). Int64("highestBlockNumber", highestBlockNumber). @@ -850,4 +906,3 @@ func validateBlockTransactions(u common.Upstream, dirs *common.RequestDirectives return nil } - diff --git a/erpc/http_server_ws_tip_floor_test.go b/erpc/http_server_ws_tip_floor_test.go new file mode 100644 index 000000000..479742a22 --- /dev/null +++ b/erpc/http_server_ws_tip_floor_test.go @@ -0,0 +1,96 @@ +package erpc + +import ( + "context" + "net/http" + "testing" + "time" + + "github.com/bytedance/sonic" + "github.com/erpc/erpc/common" + "github.com/erpc/erpc/internal/policy" + "github.com/erpc/erpc/util" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +func init() { + util.ConfigureTestLogger() +} + +// TipHW advanced via WS newHeads must be serveable on HTTP "latest" even when +// every HTTP upstream still lags TipHW and concrete tip re-fetch would miss. +func TestHttpServer_GetBlockByNumberLatest_UsesCachedWsTipWhenHttpLags(t *testing.T) { + util.ResetGock() + defer util.ResetGock() + util.SetupMocksForEvmStatePoller() + defer util.AssertNoPendingMocks(t, 0) + + cfg := &common.Config{ + Server: &common.ServerConfig{ + MaxTimeout: common.Duration(100 * time.Second).Ptr(), + }, + Projects: []*common.ProjectConfig{ + { + Id: "test_project", + Networks: []*common.NetworkConfig{ + { + Architecture: "evm", + Evm: &common.EvmNetworkConfig{ + ChainId: 123, + Integrity: &common.EvmIntegrityConfig{ + EnforceHighestBlock: util.BoolPtr(true), + }, + }, + }, + }, + Upstreams: []*common.UpstreamConfig{ + { + Id: "rpc1", + Endpoint: "http://rpc1.localhost", + Type: common.UpstreamTypeEvm, + Evm: &common.EvmUpstreamConfig{ + ChainId: 123, + StatePollerInterval: common.Duration(10 * time.Second), + }, + }, + }, + }, + }, + } + + sendRequest, _, _, shutdown, erpcInstance := createServerTestFixtures(cfg, t) + defer shutdown() + + prj, err := erpcInstance.GetProject("test_project") + require.NoError(t, err) + policy.OverrideAllForTest(prj.policyEngine) + + // Let state poller settle at 0x11118888 (SetupMocksForEvmStatePoller). + time.Sleep(500 * time.Millisecond) + + nw, err := prj.GetNetwork(context.Background(), "evm:123") + require.NoError(t, err) + + const tip = int64(0x11118889) + wsHeader := []byte(`{"number":"0x11118889","hash":"0xwshead","parentHash":"0xwsparent","timestamp":"0x6702a8f1"}`) + nw.NoteObservedLatestHead(context.Background(), tip, wsHeader) + require.Equal(t, tip, nw.EvmHighestLatestBlockNumber(context.Background())) + + statusCode, _, body := sendRequest(`{ + "jsonrpc": "2.0", + "id": 1, + "method": "eth_getBlockByNumber", + "params": ["latest", false] + }`, nil, nil) + + require.Equal(t, http.StatusOK, statusCode) + + var respObject map[string]interface{} + require.NoError(t, sonic.UnmarshalString(body, &respObject)) + result, ok := respObject["result"].(map[string]interface{}) + require.True(t, ok, "response should have a result object, got: %s", body) + assert.Equal(t, "0x11118889", result["number"], + "HTTP latest must serve the WS tip already observed on this pod") + assert.Equal(t, "0xwshead", result["hash"]) +} diff --git a/erpc/networks.go b/erpc/networks.go index ac6a71087..332e7a53a 100644 --- a/erpc/networks.go +++ b/erpc/networks.go @@ -2,6 +2,7 @@ package erpc import ( "context" + "encoding/json" "errors" "fmt" "math" @@ -68,6 +69,17 @@ type Network struct { // cannot regress below a head we have already delivered on the same pod. lastReturnedLatestBlock atomic.Int64 lastReturnedFinalizedBlock atomic.Int64 + + // lastObservedLatestHead caches the most recent newHeads header payload + // noted before fan-out, so eth_getBlockByNumber("latest", false) can + // return the same tip when HTTP tip re-fetch cannot reach TipHW. + lastObservedLatestHead atomic.Pointer[observedLatestHead] +} + +// observedLatestHead is the last WS newHeads header we noted before fan-out. +type observedLatestHead struct { + number int64 + payload json.RawMessage } // NoteObservedLatestBlock records that this Network has observed head @@ -107,6 +119,52 @@ func (n *Network) NoteObservedLatestBlock(ctx context.Context, blockNumber int64 } } +// NoteObservedLatestHead records a WS newHeads tip together with its header +// payload. It advances TipHW via NoteObservedLatestBlock and caches the +// header so eth_getBlockByNumber("latest", false) can return the same tip +// when HTTP upstreams are still behind TipHW. +// +// Callers MUST invoke this before delivering the corresponding newHeads +// notification to any client. Empty payloads still advance TipHW. +func (n *Network) NoteObservedLatestHead(ctx context.Context, blockNumber int64, payload json.RawMessage) { + n.NoteObservedLatestBlock(ctx, blockNumber) + if n == nil || blockNumber <= 0 || len(payload) == 0 { + return + } + stored := observedLatestHead{ + number: blockNumber, + payload: append(json.RawMessage(nil), payload...), + } + for { + cur := n.lastObservedLatestHead.Load() + if cur != nil && blockNumber < cur.number { + return + } + if cur != nil && blockNumber == cur.number { + if n.lastObservedLatestHead.CompareAndSwap(cur, &stored) { + return + } + continue + } + if n.lastObservedLatestHead.CompareAndSwap(cur, &stored) { + return + } + } +} + +// LastObservedLatestHead returns the cached newHeads header for the highest +// tip we have noted on this pod, or (0, nil) if none. +func (n *Network) LastObservedLatestHead() (blockNumber int64, payload []byte) { + if n == nil { + return 0, nil + } + cur := n.lastObservedLatestHead.Load() + if cur == nil || cur.number <= 0 || len(cur.payload) == 0 { + return 0, nil + } + return cur.number, append([]byte(nil), cur.payload...) +} + // EvmRefreshHighestLatestBlockNumber pulls TipHW from Redis once and returns // the network tip after adopting any higher remote value. Used when the local // TipHW cache would otherwise skip EnforceHighestBlock (false-negative under diff --git a/erpc/networks_ws_tip_test.go b/erpc/networks_ws_tip_test.go index fb9ea3f8b..aee79ef7d 100644 --- a/erpc/networks_ws_tip_test.go +++ b/erpc/networks_ws_tip_test.go @@ -181,7 +181,8 @@ func TestNetworkHandle_SuggestLatestBlock_AdvancesNetworkTipBeforeFanOut(t *test handle := &networkHandle{nw: network} // Mirrors indexer.Ingest ordering: SuggestLatestBlock then fan-out. - handle.SuggestLatestBlock("ws:bor-1", 90677359) + wsHeader := []byte(`{"number":"0x56789cf","hash":"0xabc","parentHash":"0xdef"}`) + handle.SuggestLatestBlock("ws:bor-1", 90677359, wsHeader) assert.Equal(t, int64(90677359), upsList[0].EvmStatePoller().LatestBlock(), "per-upstream poller must advance") @@ -189,6 +190,9 @@ func TestNetworkHandle_SuggestLatestBlock_AdvancesNetworkTipBeforeFanOut(t *test "network tip must advance before any client would see the WS head") assert.GreaterOrEqual(t, network.lastReturnedLatestBlock.Load(), int64(90677359), "process-local high-water mark must cover the delivered WS tip") + cachedNum, cachedPayload := network.LastObservedLatestHead() + assert.Equal(t, int64(90677359), cachedNum) + assert.Contains(t, string(cachedPayload), `"0x56789cf"`) } // EvmRefreshHighestLatestBlockNumber must not regress the tip after diff --git a/erpc/subscription_manager.go b/erpc/subscription_manager.go index 1c9823a04..d22e682d3 100644 --- a/erpc/subscription_manager.go +++ b/erpc/subscription_manager.go @@ -2,6 +2,7 @@ package erpc import ( "context" + "encoding/json" "fmt" "net/url" "strings" @@ -569,14 +570,16 @@ func (h *networkHandle) FinalityDepth() int64 { } // SuggestLatestBlock routes a per-source block observation to the -// upstream's state poller, then advances the network-level latest tip. +// upstream's state poller, then advances the network-level latest tip +// and caches the newHeads header payload when present. // sourceId is the ingress adapter's Name(), which for wsupstream.Adapter // is "ws:". // // Ordering matters: Indexer.Ingest calls this BEFORE fan-out, so by the // time any client sees head N on WS, EvmHighestLatestBlockNumber on this -// pod is already ≥ N (see Network.NoteObservedLatestBlock). -func (h *networkHandle) SuggestLatestBlock(sourceId string, blockNumber int64) { +// pod is already ≥ N and eth_getBlockByNumber("latest", false) can serve +// the same header (see Network.NoteObservedLatestHead). +func (h *networkHandle) SuggestLatestBlock(sourceId string, blockNumber int64, payload json.RawMessage) { const prefix = "ws:" if !strings.HasPrefix(sourceId, prefix) { return @@ -592,7 +595,7 @@ func (h *networkHandle) SuggestLatestBlock(sourceId string, blockNumber int64) { } break } - h.nw.NoteObservedLatestBlock(h.nw.appCtx, blockNumber) + h.nw.NoteObservedLatestHead(h.nw.appCtx, blockNumber, payload) } // Interface checks: fail the build if either contract drifts. diff --git a/indexer/adapters/wsupstream/adapter_reconnect_test.go b/indexer/adapters/wsupstream/adapter_reconnect_test.go index f07475b0b..66affe948 100644 --- a/indexer/adapters/wsupstream/adapter_reconnect_test.go +++ b/indexer/adapters/wsupstream/adapter_reconnect_test.go @@ -2,6 +2,7 @@ package wsupstream import ( "context" + "encoding/json" "errors" "fmt" "io" @@ -26,9 +27,9 @@ import ( type fakeNetworkHandle struct{} -func (fakeNetworkHandle) Id() string { return "evm:324" } -func (fakeNetworkHandle) FinalityDepth() int64 { return 0 } -func (fakeNetworkHandle) SuggestLatestBlock(string, int64) {} +func (fakeNetworkHandle) Id() string { return "evm:324" } +func (fakeNetworkHandle) FinalityDepth() int64 { return 0 } +func (fakeNetworkHandle) SuggestLatestBlock(string, int64, json.RawMessage) {} type fakeSink struct { events chan indexer.StreamEvent diff --git a/indexer/indexer.go b/indexer/indexer.go index a686fecad..f9d060e4c 100644 --- a/indexer/indexer.go +++ b/indexer/indexer.go @@ -361,7 +361,7 @@ func (i *Indexer) Ingest(ev StreamEvent) { // the indexer level — otherwise a lagging source's state poller // stalls on the first dup. if ev.Kind == KindNewHead && !ev.Block.Zero() && ev.SourceId != "" { - ns.handle.SuggestLatestBlock(ev.SourceId, ev.Block.Number) + ns.handle.SuggestLatestBlock(ev.SourceId, ev.Block.Number, ev.Payload) } // Dedup. diff --git a/indexer/indexer_test.go b/indexer/indexer_test.go index b586fb935..dd57698f1 100644 --- a/indexer/indexer_test.go +++ b/indexer/indexer_test.go @@ -17,8 +17,8 @@ type fakeNetwork struct { id string finalityDepth int64 - mu sync.Mutex - suggestedBy map[string][]int64 // sourceId -> block nums seen + mu sync.Mutex + suggestedBy map[string][]int64 // sourceId -> block nums seen } func newFakeNetwork(id string, depth int64) *fakeNetwork { @@ -29,9 +29,9 @@ func newFakeNetwork(id string, depth int64) *fakeNetwork { } } -func (n *fakeNetwork) Id() string { return n.id } -func (n *fakeNetwork) FinalityDepth() int64 { return n.finalityDepth } -func (n *fakeNetwork) SuggestLatestBlock(sourceId string, block int64) { +func (n *fakeNetwork) Id() string { return n.id } +func (n *fakeNetwork) FinalityDepth() int64 { return n.finalityDepth } +func (n *fakeNetwork) SuggestLatestBlock(sourceId string, block int64, _ json.RawMessage) { n.mu.Lock() n.suggestedBy[sourceId] = append(n.suggestedBy[sourceId], block) n.mu.Unlock() diff --git a/indexer/ingress.go b/indexer/ingress.go index d33f57705..01c37894b 100644 --- a/indexer/ingress.go +++ b/indexer/ingress.go @@ -1,6 +1,9 @@ package indexer -import "context" +import ( + "context" + "encoding/json" +) // Sink is the interface an ingress uses to push StreamEvents into the // indexer pipeline. The indexer itself implements Sink — a pointer to the @@ -26,12 +29,11 @@ type NetworkHandle interface { FinalityDepth() int64 // SuggestLatestBlock advances the per-source latest-block tracker // (and the network-level latest tip) before the indexer dedupes / - // fans out. Preserving "update-before-dedup" and - // "tip-before-fanout" ordering is critical — the state poller needs - // to see every observation (even ones we drop), and HTTP "latest" - // must not be allowed to regress below a head we are about to - // deliver on WS. - SuggestLatestBlock(sourceId string, blockNumber int64) + // fans out. payload is the verbatim newHeads header JSON when known + // (empty is allowed) so HTTP "latest" can serve the same tip when + // concrete tip re-fetch misses. Preserving "update-before-dedup" and + // "tip-before-fanout" ordering is critical. + SuggestLatestBlock(sourceId string, blockNumber int64, payload json.RawMessage) } // EventIngress is an adapter that converts some transport-specific diff --git a/indexer/integration_test.go b/indexer/integration_test.go index 3eec40b0a..ce00ff713 100644 --- a/indexer/integration_test.go +++ b/indexer/integration_test.go @@ -107,7 +107,7 @@ type stubNetwork struct { func (s *stubNetwork) Id() string { return s.id } func (s *stubNetwork) FinalityDepth() int64 { return s.finality } -func (s *stubNetwork) SuggestLatestBlock(sourceID string, block int64) { +func (s *stubNetwork) SuggestLatestBlock(sourceID string, block int64, _ json.RawMessage) { s.mu.Lock() if s.suggestions == nil { s.suggestions = make(map[string][]int64) From 53456da86a3adc2135b46307d3125c595e20affb Mon Sep 17 00:00:00 2001 From: Jonny Date: Thu, 23 Jul 2026 14:43:54 +0100 Subject: [PATCH 28/40] fix(ws): stop TipHW inflation from fallback WS; escape on emptyish miss Fallback newHeads must not advance TipHW while primaries are up (same rule as poller aggregation). Also let the per-request fallback escape hatch fire when primaries return null/emptyish for a concrete tip block, so tip re-fetch can use public fallbacks instead of hard-failing. Co-authored-by: Cursor --- erpc/networks.go | 83 ++++++++++-- erpc/networks_failover_escape_test.go | 67 ++++++++++ erpc/networks_ws_tip_test.go | 178 ++++++++++++++++++++++++++ erpc/subscription_manager.go | 14 ++ 4 files changed, 330 insertions(+), 12 deletions(-) diff --git a/erpc/networks.go b/erpc/networks.go index 332e7a53a..a8f7e3977 100644 --- a/erpc/networks.go +++ b/erpc/networks.go @@ -590,20 +590,61 @@ func (n *Network) EvmLowestFinalizedBlockNumber(ctx context.Context) int64 { return minBlock } +// anyPrimaryUpstreamUp reports whether at least one non-fallback upstream +// on this network has a closed circuit breaker. Same "up" definition as +// evmHighestBlockNumber. +func (n *Network) anyPrimaryUpstreamUp(ctx context.Context) bool { + if n == nil || n.upstreamsRegistry == nil { + return false + } + if ctx == nil { + ctx = context.Background() + } + for _, u := range n.upstreamsRegistry.GetNetworkUpstreams(ctx, n.networkId) { + if u.Config() != nil && u.Config().HasTag(common.TagTierFallback) { + continue + } + if !u.IsDown() { + return true + } + } + return false +} + +// EvmLeaderUpstream returns the upstream whose state poller has the highest +// latest tip. Fallback-tier upstreams are ignored while any primary is up, +// matching TipHW aggregation — otherwise tip re-fetch pins UseUpstream to a +// cordoned fallback that is not in the ordered primary list. func (n *Network) EvmLeaderUpstream(ctx context.Context) common.Upstream { - var leader common.Upstream - var leaderLastBlock int64 = 0 + var leader, fallbackLeader common.Upstream + var leaderLastBlock, fallbackLastBlock int64 + anyPrimaryUp := false upsList := n.upstreamsRegistry.GetNetworkUpstreams(ctx, n.networkId) for _, u := range upsList { - if statePoller := u.EvmStatePoller(); statePoller != nil { - lastBlock := statePoller.LatestBlock() - if lastBlock > leaderLastBlock { - leader = u - leaderLastBlock = lastBlock + statePoller := u.EvmStatePoller() + if statePoller == nil { + continue + } + lastBlock := statePoller.LatestBlock() + if u.Config() != nil && u.Config().HasTag(common.TagTierFallback) { + if lastBlock > fallbackLastBlock { + fallbackLeader = u + fallbackLastBlock = lastBlock } + continue + } + if !u.IsDown() { + anyPrimaryUp = true + } + if lastBlock > leaderLastBlock { + leader = u + leaderLastBlock = lastBlock } } - return leader + if anyPrimaryUp || fallbackLeader == nil { + return leader + } + return fallbackLeader } func (n *Network) getFailsafeExecutor(ctx context.Context, req *common.NormalizedRequest) *networkExecutor { @@ -1075,7 +1116,17 @@ func (n *Network) Forward(ctx context.Context, req *common.NormalizedRequest) (* common.SetTraceSpanError(loopSpan, err) } else if r != nil { bestResp = r - loopSpan.SetStatus(codes.Ok, "") + if r.IsResultEmptyish() { + // Soft miss (e.g. eth_getBlockByNumber null): seed + // lastErr so the fallback escape hatch can fire after + // primaries are exhausted. Without this, emptyish + // responses leave lastErr nil and block escalation. + lastErr = common.NewErrEndpointMissingData( + fmt.Errorf("upstream responded emptyish"), u, + ) + } else { + loopSpan.SetStatus(codes.Ok, "") + } } loopSpan.End() } @@ -1101,14 +1152,18 @@ func (n *Network) Forward(ctx context.Context, req *common.NormalizedRequest) (* // // Bounds: // - At most one escalation per request (MarkEscalatedToFallbacks). - // - bestResp != nil means we have an emptyish response that - // failsafe's emptyResultDelay should evaluate — don't pre-empt it. + // - Emptyish bestResp still escapes: methods like + // eth_getBlockByNumber return null for missing blocks without + // setting err, and that soft miss must escalate to fallbacks + // the same way a gate-reject does. Non-empty bestResp means we + // already have a usable candidate — leave it for failsafe. // - Deterministic client errors return from the inner loop // immediately, so any non-nil lastErr here means "this upstream // couldn't serve; try a different one" — exactly the escape's job. // - Consensus requires strict per-upstream semantics; don't modify // the candidate set mid-execution. - if bestResp == nil && + bestRespEmptyish := bestResp != nil && bestResp.IsResultEmptyish() + if (bestResp == nil || bestRespEmptyish) && !effectiveReq.HasEscalatedToFallbacks() && lastErr != nil && n.cfg.Failover != nil && n.cfg.Failover.Enabled() && @@ -1123,6 +1178,10 @@ func (n *Network) Forward(ctx context.Context, req *common.NormalizedRequest) (* // re-selects them; primaries are removed from upsList so they // won't be picked again, but their state in attempted / // ErrorsByUpstream is preserved for eventual error reporting. + if bestRespEmptyish { + bestResp.Release() + bestResp = nil + } fbCommon := make([]common.Upstream, 0, len(fallbacks)) for _, fb := range fallbacks { fbCommon = append(fbCommon, fb) diff --git a/erpc/networks_failover_escape_test.go b/erpc/networks_failover_escape_test.go index 91a4204f5..6aae47a6c 100644 --- a/erpc/networks_failover_escape_test.go +++ b/erpc/networks_failover_escape_test.go @@ -626,4 +626,71 @@ func TestFailover_EscapeHatch(t *testing.T) { assert.Equal(t, before+1, after, "escape hatch must fire exactly once for the non-retryable gate-skip case") }) + + t.Run("EscapesOnEmptyishGetBlockByNumber", func(t *testing.T) { + defer util.ResetGock() + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + + // Primaries and fallbacks both report tip 1002 via poller, so the + // availability gate fails open / passes. Primaries return null for + // the concrete tip block (missing data); fallbacks return the header. + // Before the emptyish-escape fix, bestResp=null blocked escalation. + network, _, _ := setupFailoverFixture(t, ctx, failoverFixtureOpts{ + primaryLatest: "0x3ea", // 1002 + fallbackLatest: "0x3ea", // 1002 + enableFailover: true, + }) + + nullBlock := `{"jsonrpc":"2.0","id":1,"result":null}` + okBlock := `{"jsonrpc":"2.0","id":1,"result":{"number":"0x3ea","hash":"0xabc","parentHash":"0xdef","timestamp":"0x6702a8f0"}}` + for _, host := range []string{"rpc1.localhost", "rpc2.localhost"} { + gock.New("http://" + host). + Post(""). + Persist(). + Filter(func(r *http.Request) bool { + b := util.SafeReadBody(r) + return strings.Contains(b, "eth_getBlockByNumber") && strings.Contains(b, `"0x3ea"`) + }). + Reply(200). + JSON([]byte(nullBlock)) + } + for _, host := range []string{"rpc3.localhost", "rpc4.localhost"} { + gock.New("http://" + host). + Post(""). + Persist(). + Filter(func(r *http.Request) bool { + b := util.SafeReadBody(r) + return strings.Contains(b, "eth_getBlockByNumber") && strings.Contains(b, `"0x3ea"`) + }). + Reply(200). + JSON([]byte(okBlock)) + } + + counter := telemetry.MetricNetworkFallbackEscapeTotal.WithLabelValues("main", "evm:999", "eth_getBlockByNumber") + before := promUtil.ToFloat64(counter) + + req := common.NewNormalizedRequest([]byte( + `{"jsonrpc":"2.0","id":1,"method":"eth_getBlockByNumber","params":["0x3ea",false]}`, + )) + req.SetNetwork(network) + resp, err := network.Forward(ctx, req) + require.NoError(t, err, "null from primaries must escalate to fallbacks on the same request") + require.NotNil(t, resp) + defer resp.Release() + + jrr, err := resp.JsonRpcResponse() + require.NoError(t, err) + require.False(t, jrr.IsResultEmptyish(), "fallback must return a non-null block header") + num, err := jrr.PeekStringByPath(ctx, "number") + require.NoError(t, err) + assert.Equal(t, "0x3ea", num) + + assert.Contains(t, []string{"fallback-1", "fallback-2"}, resp.UpstreamId(), + "emptyish primary miss must be served by a fallback") + + after := promUtil.ToFloat64(counter) + assert.Equal(t, before+1, after, + "escape hatch must fire for emptyish eth_getBlockByNumber primary misses") + }) } diff --git a/erpc/networks_ws_tip_test.go b/erpc/networks_ws_tip_test.go index aee79ef7d..116723c8e 100644 --- a/erpc/networks_ws_tip_test.go +++ b/erpc/networks_ws_tip_test.go @@ -195,6 +195,184 @@ func TestNetworkHandle_SuggestLatestBlock_AdvancesNetworkTipBeforeFanOut(t *test assert.Contains(t, string(cachedPayload), `"0x56789cf"`) } +// Fallback WS tips must not inflate TipHW while any primary is up. +// Poller still advances so failover/partition can prefer the fallback. +func TestNetworkHandle_SuggestLatestBlock_FallbackDoesNotAdvanceTipHWWhenPrimaryUp(t *testing.T) { + util.ResetGock() + defer util.ResetGock() + util.SetupMocksForEvmStatePoller() + + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + + primary := &common.UpstreamConfig{ + Type: common.UpstreamTypeEvm, + Id: "primary-ws", + Endpoint: "http://primary.localhost", + Evm: &common.EvmUpstreamConfig{ChainId: 123}, + } + fallback := &common.UpstreamConfig{ + Type: common.UpstreamTypeEvm, + Id: "fallback-ws", + Endpoint: "http://fallback.localhost", + Tags: []string{common.TagTierFallback}, + Evm: &common.EvmUpstreamConfig{ChainId: 123}, + } + + for _, host := range []string{"primary.localhost", "fallback.localhost"} { + gock.New("http://" + host). + Post(""). + Persist(). + Filter(func(r *http.Request) bool { + return strings.Contains(util.SafeReadBody(r), `eth_chainId`) + }). + Reply(200). + JSON([]byte(`{"result":"0x7b"}`)) + } + + rateLimitersRegistry, _ := upstream.NewRateLimitersRegistry(context.Background(), &common.RateLimiterConfig{}, &log.Logger) + metricsTracker := health.NewTracker(&log.Logger, "test", time.Minute) + + vr := thirdparty.NewVendorsRegistry() + pr, err := thirdparty.NewProvidersRegistry(&log.Logger, vr, []*common.ProviderConfig{}, nil) + require.NoError(t, err) + + ssr, err := data.NewSharedStateRegistry(ctx, &log.Logger, &common.SharedStateConfig{ + Connector: &common.ConnectorConfig{ + Driver: "memory", + Memory: &common.MemoryConnectorConfig{MaxItems: 100_000, MaxTotalSize: "1GB"}, + }, + }) + require.NoError(t, err) + + upstreamsRegistry := upstream.NewUpstreamsRegistry( + ctx, &log.Logger, "test", + []*common.UpstreamConfig{primary, fallback}, ssr, rateLimitersRegistry, vr, pr, nil, + metricsTracker, nil, + ) + + networkConfig := &common.NetworkConfig{ + Architecture: common.ArchitectureEvm, + Evm: &common.EvmNetworkConfig{ChainId: 123}, + } + network, err := NewNetwork(ctx, &log.Logger, "test", networkConfig, + rateLimitersRegistry, upstreamsRegistry, metricsTracker, nil) + require.NoError(t, err) + + upstreamsRegistry.Bootstrap(ctx) + time.Sleep(200 * time.Millisecond) + require.NoError(t, upstreamsRegistry.GetInitializer().WaitForTasks(ctx)) + require.NoError(t, network.Bootstrap(ctx)) + time.Sleep(250 * time.Millisecond) + + upsList := upstreamsRegistry.GetNetworkUpstreams(ctx, util.EvmNetworkId(123)) + require.Len(t, upsList, 2) + var primaryUp, fallbackUp *upstream.Upstream + for _, u := range upsList { + switch u.Id() { + case "primary-ws": + primaryUp = u + case "fallback-ws": + fallbackUp = u + } + } + require.NotNil(t, primaryUp) + require.NotNil(t, fallbackUp) + + primaryUp.EvmStatePoller().SuggestLatestBlock(1000) + time.Sleep(50 * time.Millisecond) + require.Equal(t, int64(1000), network.EvmHighestLatestBlockNumber(ctx)) + + handle := &networkHandle{nw: network} + wsHeader := []byte(`{"number":"0x3ea","hash":"0xabc","parentHash":"0xdef"}`) + handle.SuggestLatestBlock("ws:fallback-ws", 1002, wsHeader) + + assert.Equal(t, int64(1002), fallbackUp.EvmStatePoller().LatestBlock(), + "fallback poller must still advance for selection/escape") + assert.Equal(t, int64(1000), network.EvmHighestLatestBlockNumber(ctx), + "TipHW must stay on primary tip while primaries are up") + cachedNum, _ := network.LastObservedLatestHead() + assert.Equal(t, int64(0), cachedNum, + "must not cache fallback header into TipHW header cache while primaries are up") + + // Primary WS tip still advances TipHW. + handle.SuggestLatestBlock("ws:primary-ws", 1001, []byte(`{"number":"0x3e9","hash":"0x1","parentHash":"0x2"}`)) + assert.Equal(t, int64(1001), network.EvmHighestLatestBlockNumber(ctx), + "primary WS tip must advance TipHW") +} + +// When every primary is down, fallback WS may advance TipHW (same rule as +// evmHighestBlockNumber). +func TestNetworkHandle_SuggestLatestBlock_FallbackAdvancesTipHWWhenNoPrimaryUp(t *testing.T) { + util.ResetGock() + defer util.ResetGock() + util.SetupMocksForEvmStatePoller() + + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + + fallback := &common.UpstreamConfig{ + Type: common.UpstreamTypeEvm, + Id: "fallback-only", + Endpoint: "http://fallback.localhost", + Tags: []string{common.TagTierFallback}, + Evm: &common.EvmUpstreamConfig{ChainId: 123}, + } + + gock.New("http://fallback.localhost"). + Post(""). + Persist(). + Filter(func(r *http.Request) bool { + return strings.Contains(util.SafeReadBody(r), `eth_chainId`) + }). + Reply(200). + JSON([]byte(`{"result":"0x7b"}`)) + + rateLimitersRegistry, _ := upstream.NewRateLimitersRegistry(context.Background(), &common.RateLimiterConfig{}, &log.Logger) + metricsTracker := health.NewTracker(&log.Logger, "test", time.Minute) + + vr := thirdparty.NewVendorsRegistry() + pr, err := thirdparty.NewProvidersRegistry(&log.Logger, vr, []*common.ProviderConfig{}, nil) + require.NoError(t, err) + + ssr, err := data.NewSharedStateRegistry(ctx, &log.Logger, &common.SharedStateConfig{ + Connector: &common.ConnectorConfig{ + Driver: "memory", + Memory: &common.MemoryConnectorConfig{MaxItems: 100_000, MaxTotalSize: "1GB"}, + }, + }) + require.NoError(t, err) + + upstreamsRegistry := upstream.NewUpstreamsRegistry( + ctx, &log.Logger, "test", + []*common.UpstreamConfig{fallback}, ssr, rateLimitersRegistry, vr, pr, nil, + metricsTracker, nil, + ) + + networkConfig := &common.NetworkConfig{ + Architecture: common.ArchitectureEvm, + Evm: &common.EvmNetworkConfig{ChainId: 123}, + } + network, err := NewNetwork(ctx, &log.Logger, "test", networkConfig, + rateLimitersRegistry, upstreamsRegistry, metricsTracker, nil) + require.NoError(t, err) + + upstreamsRegistry.Bootstrap(ctx) + time.Sleep(200 * time.Millisecond) + require.NoError(t, upstreamsRegistry.GetInitializer().WaitForTasks(ctx)) + require.NoError(t, network.Bootstrap(ctx)) + time.Sleep(250 * time.Millisecond) + + handle := &networkHandle{nw: network} + handle.SuggestLatestBlock("ws:fallback-only", 2005, []byte(`{"number":"0x7d5","hash":"0xabc","parentHash":"0xdef"}`)) + + assert.Equal(t, int64(2005), network.EvmHighestLatestBlockNumber(ctx), + "fallback WS may advance TipHW when no primary is up") + cachedNum, cachedPayload := network.LastObservedLatestHead() + assert.Equal(t, int64(2005), cachedNum) + assert.Contains(t, string(cachedPayload), `"0x7d5"`) +} + // EvmRefreshHighestLatestBlockNumber must not regress the tip after // NoteObservedLatestBlock (sync publish path) has advanced TipHW. func TestEvmRefreshHighestLatestBlockNumber_PreservesObservedTip(t *testing.T) { diff --git a/erpc/subscription_manager.go b/erpc/subscription_manager.go index d22e682d3..02459ea42 100644 --- a/erpc/subscription_manager.go +++ b/erpc/subscription_manager.go @@ -579,22 +579,36 @@ func (h *networkHandle) FinalityDepth() int64 { // time any client sees head N on WS, EvmHighestLatestBlockNumber on this // pod is already ≥ N and eth_getBlockByNumber("latest", false) can serve // the same header (see Network.NoteObservedLatestHead). +// +// TipHW publish mirrors evmHighestBlockNumber: a fallback-tier WS tip must +// not advance TipHW while any primary is up. Otherwise selection asks +// healthy-but-lagging primaries for a block only the fallback has, and +// tip-floor hard-fails before failover can help. The fallback's own poller +// still advances so partition/escape can prefer it when primaries miss. func (h *networkHandle) SuggestLatestBlock(sourceId string, blockNumber int64, payload json.RawMessage) { const prefix = "ws:" if !strings.HasPrefix(sourceId, prefix) { return } upstreamID := sourceId[len(prefix):] + var sourceUp *upstream.Upstream for _, u := range h.nw.upstreamsRegistry.GetNetworkUpstreams(context.Background(), h.nw.networkId) { if u.Id() != upstreamID { continue } + sourceUp = u poller := u.EvmStatePoller() if poller != nil && !poller.IsObjectNull() { poller.SuggestLatestBlock(blockNumber) } break } + if sourceUp != nil && + sourceUp.Config() != nil && + sourceUp.Config().HasTag(common.TagTierFallback) && + h.nw.anyPrimaryUpstreamUp(context.Background()) { + return + } h.nw.NoteObservedLatestHead(h.nw.appCtx, blockNumber, payload) } From 9ba72dbb411c67cb081f6711064123ab8d085f5f Mon Sep 17 00:00:00 2001 From: Jonny Date: Thu, 23 Jul 2026 14:45:26 +0100 Subject: [PATCH 29/40] fix(ws): drop cached newHeads header serve for HTTP latest That path masked TipHW inflation / missing failover. Keep TipHW floor from primary WS only, refuse-stale when tip re-fetch misses, and escape to fallbacks on emptyish primary misses. Co-authored-by: Cursor --- architecture/evm/eth_getBlockByNumber.go | 59 +----------------------- erpc/http_server_ws_tip_floor_test.go | 28 ++++++----- erpc/networks.go | 58 ----------------------- erpc/networks_ws_tip_test.go | 9 ---- erpc/subscription_manager.go | 11 ++--- 5 files changed, 23 insertions(+), 142 deletions(-) diff --git a/architecture/evm/eth_getBlockByNumber.go b/architecture/evm/eth_getBlockByNumber.go index b06e18e78..451a979fc 100644 --- a/architecture/evm/eth_getBlockByNumber.go +++ b/architecture/evm/eth_getBlockByNumber.go @@ -26,39 +26,6 @@ func refreshHighestLatestBlockNumber(ctx context.Context, network common.Network return network.EvmHighestLatestBlockNumber(ctx) } -// observedLatestHeadProvider is implemented by networks that cache the last -// WS newHeads header before fan-out (erpc.Network). Used to serve HTTP -// "latest" when concrete tip re-fetch cannot reach TipHW. -type observedLatestHeadProvider interface { - LastObservedLatestHead() (blockNumber int64, payload []byte) -} - -// responseFromObservedLatestHead builds an eth_getBlockByNumber response from -// the cached WS newHeads header when it matches expectedTip. ok is false when -// the cache is missing, behind, or the network does not expose a tip cache. -func responseFromObservedLatestHead(network common.Network, nq *common.NormalizedRequest, expectedTip int64) (*common.NormalizedResponse, bool) { - provider, ok := network.(observedLatestHeadProvider) - if !ok || expectedTip <= 0 { - return nil, false - } - cachedNumber, payload := provider.LastObservedLatestHead() - if cachedNumber != expectedTip || len(payload) == 0 { - return nil, false - } - idBytes, err := common.SonicCfg.Marshal(nq.ID()) - if err != nil { - return nil, false - } - jrr, err := common.NewJsonRpcResponseFromBytes(idBytes, append([]byte(nil), payload...), nil) - if err != nil { - return nil, false - } - resp := common.NewNormalizedResponse(). - WithRequest(nq). - WithJsonRpcResponse(jrr) - return resp, true -} - func BuildGetBlockByNumberRequest(blockNumberOrTag interface{}, includeTransactions bool) (*common.JsonRpcRequest, error) { var bkt string var err error @@ -258,18 +225,6 @@ func enforceHighestBlock(ctx context.Context, network common.Network, nq *common ).Inc() } - // If we already have the TipHW header from WS newHeads on this pod, - // serve it for header-only latest. TipHW means we observed the block; - // failing the client because HTTP tip re-fetch lags is a bug. - if !itx { - if cached, ok := responseFromObservedLatestHead(network, nq, highestBlockNumber); ok { - if nr != nil { - nr.Release() - } - return cached, nil - } - } - // Prefer the upstream whose poller already owns this tip // (EvmLeaderUpstream — typically the WS ingress that called // SuggestLatestBlock). If TipHW advanced via Redis/WS while @@ -309,7 +264,8 @@ func enforceHighestBlock(ctx context.Context, network common.Network, nq *common // Pinned / excluded re-fetch missed the tip (sibling fullnode // lag, WS JSON-RPC miss, etc.). Retry with no UseUpstream pin - // so every upstream can serve the concrete TipHW block. + // so every upstream (including fallbacks via escape) can serve + // the concrete TipHW block. nnr2, ferr2 := forwardGetBlockByNumber(ctx, network, nq, highestBlockNumber, itx, "") if meetsTipFloor(ctx, nnr2, highestBlockNumber) { if nr != nil { @@ -321,17 +277,6 @@ func enforceHighestBlock(ctx context.Context, network common.Network, nq *common nnr2.Release() } - // Tip re-fetch still missed — try the WS-cached header once more - // (TipHW may have been noted mid-flight after the early cache check). - if !itx { - if cached, ok := responseFromObservedLatestHead(network, nq, highestBlockNumber); ok { - if nr != nil { - nr.Release() - } - return cached, nil - } - } - // NEVER fail-open to a tip below TipHW. Prefer an error over stale. logger.Warn(). Int64("highestBlockNumber", highestBlockNumber). diff --git a/erpc/http_server_ws_tip_floor_test.go b/erpc/http_server_ws_tip_floor_test.go index 479742a22..3e72b0e6f 100644 --- a/erpc/http_server_ws_tip_floor_test.go +++ b/erpc/http_server_ws_tip_floor_test.go @@ -18,9 +18,9 @@ func init() { util.ConfigureTestLogger() } -// TipHW advanced via WS newHeads must be serveable on HTTP "latest" even when -// every HTTP upstream still lags TipHW and concrete tip re-fetch would miss. -func TestHttpServer_GetBlockByNumberLatest_UsesCachedWsTipWhenHttpLags(t *testing.T) { +// TipHW advanced via WS must not fail-open to a stale HTTP "latest" when tip +// re-fetch cannot reach TipHW. Prefer an error over demoting MultiNode FOOS. +func TestHttpServer_GetBlockByNumberLatest_RefusesStaleWhenTipRefetchMisses(t *testing.T) { util.ResetGock() defer util.ResetGock() util.SetupMocksForEvmStatePoller() @@ -73,8 +73,7 @@ func TestHttpServer_GetBlockByNumberLatest_UsesCachedWsTipWhenHttpLags(t *testin require.NoError(t, err) const tip = int64(0x11118889) - wsHeader := []byte(`{"number":"0x11118889","hash":"0xwshead","parentHash":"0xwsparent","timestamp":"0x6702a8f1"}`) - nw.NoteObservedLatestHead(context.Background(), tip, wsHeader) + nw.NoteObservedLatestBlock(context.Background(), tip) require.Equal(t, tip, nw.EvmHighestLatestBlockNumber(context.Background())) statusCode, _, body := sendRequest(`{ @@ -84,13 +83,18 @@ func TestHttpServer_GetBlockByNumberLatest_UsesCachedWsTipWhenHttpLags(t *testin "params": ["latest", false] }`, nil, nil) - require.Equal(t, http.StatusOK, statusCode) - var respObject map[string]interface{} require.NoError(t, sonic.UnmarshalString(body, &respObject)) - result, ok := respObject["result"].(map[string]interface{}) - require.True(t, ok, "response should have a result object, got: %s", body) - assert.Equal(t, "0x11118889", result["number"], - "HTTP latest must serve the WS tip already observed on this pod") - assert.Equal(t, "0xwshead", result["hash"]) + + // Upstream only has 0x11118888; TipHW is one ahead. Must not return the + // stale header as success. + if statusCode == http.StatusOK { + if result, ok := respObject["result"].(map[string]interface{}); ok { + assert.NotEqual(t, "0x11118888", result["number"], + "must not fail-open to stale latest below TipHW; body=%s", body) + } + } else { + _, hasErr := respObject["error"] + assert.True(t, hasErr, "non-OK response should carry a JSON-RPC error; body=%s", body) + } } diff --git a/erpc/networks.go b/erpc/networks.go index a8f7e3977..8d60a0bd1 100644 --- a/erpc/networks.go +++ b/erpc/networks.go @@ -2,7 +2,6 @@ package erpc import ( "context" - "encoding/json" "errors" "fmt" "math" @@ -69,17 +68,6 @@ type Network struct { // cannot regress below a head we have already delivered on the same pod. lastReturnedLatestBlock atomic.Int64 lastReturnedFinalizedBlock atomic.Int64 - - // lastObservedLatestHead caches the most recent newHeads header payload - // noted before fan-out, so eth_getBlockByNumber("latest", false) can - // return the same tip when HTTP tip re-fetch cannot reach TipHW. - lastObservedLatestHead atomic.Pointer[observedLatestHead] -} - -// observedLatestHead is the last WS newHeads header we noted before fan-out. -type observedLatestHead struct { - number int64 - payload json.RawMessage } // NoteObservedLatestBlock records that this Network has observed head @@ -119,52 +107,6 @@ func (n *Network) NoteObservedLatestBlock(ctx context.Context, blockNumber int64 } } -// NoteObservedLatestHead records a WS newHeads tip together with its header -// payload. It advances TipHW via NoteObservedLatestBlock and caches the -// header so eth_getBlockByNumber("latest", false) can return the same tip -// when HTTP upstreams are still behind TipHW. -// -// Callers MUST invoke this before delivering the corresponding newHeads -// notification to any client. Empty payloads still advance TipHW. -func (n *Network) NoteObservedLatestHead(ctx context.Context, blockNumber int64, payload json.RawMessage) { - n.NoteObservedLatestBlock(ctx, blockNumber) - if n == nil || blockNumber <= 0 || len(payload) == 0 { - return - } - stored := observedLatestHead{ - number: blockNumber, - payload: append(json.RawMessage(nil), payload...), - } - for { - cur := n.lastObservedLatestHead.Load() - if cur != nil && blockNumber < cur.number { - return - } - if cur != nil && blockNumber == cur.number { - if n.lastObservedLatestHead.CompareAndSwap(cur, &stored) { - return - } - continue - } - if n.lastObservedLatestHead.CompareAndSwap(cur, &stored) { - return - } - } -} - -// LastObservedLatestHead returns the cached newHeads header for the highest -// tip we have noted on this pod, or (0, nil) if none. -func (n *Network) LastObservedLatestHead() (blockNumber int64, payload []byte) { - if n == nil { - return 0, nil - } - cur := n.lastObservedLatestHead.Load() - if cur == nil || cur.number <= 0 || len(cur.payload) == 0 { - return 0, nil - } - return cur.number, append([]byte(nil), cur.payload...) -} - // EvmRefreshHighestLatestBlockNumber pulls TipHW from Redis once and returns // the network tip after adopting any higher remote value. Used when the local // TipHW cache would otherwise skip EnforceHighestBlock (false-negative under diff --git a/erpc/networks_ws_tip_test.go b/erpc/networks_ws_tip_test.go index 116723c8e..f2c8157dc 100644 --- a/erpc/networks_ws_tip_test.go +++ b/erpc/networks_ws_tip_test.go @@ -190,9 +190,6 @@ func TestNetworkHandle_SuggestLatestBlock_AdvancesNetworkTipBeforeFanOut(t *test "network tip must advance before any client would see the WS head") assert.GreaterOrEqual(t, network.lastReturnedLatestBlock.Load(), int64(90677359), "process-local high-water mark must cover the delivered WS tip") - cachedNum, cachedPayload := network.LastObservedLatestHead() - assert.Equal(t, int64(90677359), cachedNum) - assert.Contains(t, string(cachedPayload), `"0x56789cf"`) } // Fallback WS tips must not inflate TipHW while any primary is up. @@ -291,9 +288,6 @@ func TestNetworkHandle_SuggestLatestBlock_FallbackDoesNotAdvanceTipHWWhenPrimary "fallback poller must still advance for selection/escape") assert.Equal(t, int64(1000), network.EvmHighestLatestBlockNumber(ctx), "TipHW must stay on primary tip while primaries are up") - cachedNum, _ := network.LastObservedLatestHead() - assert.Equal(t, int64(0), cachedNum, - "must not cache fallback header into TipHW header cache while primaries are up") // Primary WS tip still advances TipHW. handle.SuggestLatestBlock("ws:primary-ws", 1001, []byte(`{"number":"0x3e9","hash":"0x1","parentHash":"0x2"}`)) @@ -368,9 +362,6 @@ func TestNetworkHandle_SuggestLatestBlock_FallbackAdvancesTipHWWhenNoPrimaryUp(t assert.Equal(t, int64(2005), network.EvmHighestLatestBlockNumber(ctx), "fallback WS may advance TipHW when no primary is up") - cachedNum, cachedPayload := network.LastObservedLatestHead() - assert.Equal(t, int64(2005), cachedNum) - assert.Contains(t, string(cachedPayload), `"0x7d5"`) } // EvmRefreshHighestLatestBlockNumber must not regress the tip after diff --git a/erpc/subscription_manager.go b/erpc/subscription_manager.go index 02459ea42..a18d977e0 100644 --- a/erpc/subscription_manager.go +++ b/erpc/subscription_manager.go @@ -570,15 +570,13 @@ func (h *networkHandle) FinalityDepth() int64 { } // SuggestLatestBlock routes a per-source block observation to the -// upstream's state poller, then advances the network-level latest tip -// and caches the newHeads header payload when present. +// upstream's state poller, then advances the network-level latest tip. // sourceId is the ingress adapter's Name(), which for wsupstream.Adapter -// is "ws:". +// is "ws:". payload is unused (kept for indexer.NetworkHandle). // // Ordering matters: Indexer.Ingest calls this BEFORE fan-out, so by the // time any client sees head N on WS, EvmHighestLatestBlockNumber on this -// pod is already ≥ N and eth_getBlockByNumber("latest", false) can serve -// the same header (see Network.NoteObservedLatestHead). +// pod is already ≥ N. // // TipHW publish mirrors evmHighestBlockNumber: a fallback-tier WS tip must // not advance TipHW while any primary is up. Otherwise selection asks @@ -586,6 +584,7 @@ func (h *networkHandle) FinalityDepth() int64 { // tip-floor hard-fails before failover can help. The fallback's own poller // still advances so partition/escape can prefer it when primaries miss. func (h *networkHandle) SuggestLatestBlock(sourceId string, blockNumber int64, payload json.RawMessage) { + _ = payload const prefix = "ws:" if !strings.HasPrefix(sourceId, prefix) { return @@ -609,7 +608,7 @@ func (h *networkHandle) SuggestLatestBlock(sourceId string, blockNumber int64, p h.nw.anyPrimaryUpstreamUp(context.Background()) { return } - h.nw.NoteObservedLatestHead(h.nw.appCtx, blockNumber, payload) + h.nw.NoteObservedLatestBlock(h.nw.appCtx, blockNumber) } // Interface checks: fail the build if either contract drifts. From f78796d13d0c761aa1c38e0884a6938cb27a0082 Mon Sep 17 00:00:00 2001 From: Jonny Date: Thu, 23 Jul 2026 15:13:03 +0100 Subject: [PATCH 30/40] fix(ws): always advance TipHW for fan-out heads, keep emptyish escape Skipping TipHW for fallback WS while Ingest still fans those heads out caused MultiNode FOOS on matic. TipHW must cover every delivered head; tip re-fetch of a fallback-advanced TipHW relies on the emptyish escape hatch to reach public upstreams when primaries miss. Co-authored-by: Cursor --- erpc/networks.go | 29 +++++------------------------ erpc/networks_ws_tip_test.go | 16 ++++++---------- erpc/subscription_manager.go | 21 ++++++--------------- 3 files changed, 17 insertions(+), 49 deletions(-) diff --git a/erpc/networks.go b/erpc/networks.go index 8d60a0bd1..6ee3e0821 100644 --- a/erpc/networks.go +++ b/erpc/networks.go @@ -532,31 +532,12 @@ func (n *Network) EvmLowestFinalizedBlockNumber(ctx context.Context) int64 { return minBlock } -// anyPrimaryUpstreamUp reports whether at least one non-fallback upstream -// on this network has a closed circuit breaker. Same "up" definition as -// evmHighestBlockNumber. -func (n *Network) anyPrimaryUpstreamUp(ctx context.Context) bool { - if n == nil || n.upstreamsRegistry == nil { - return false - } - if ctx == nil { - ctx = context.Background() - } - for _, u := range n.upstreamsRegistry.GetNetworkUpstreams(ctx, n.networkId) { - if u.Config() != nil && u.Config().HasTag(common.TagTierFallback) { - continue - } - if !u.IsDown() { - return true - } - } - return false -} - // EvmLeaderUpstream returns the upstream whose state poller has the highest -// latest tip. Fallback-tier upstreams are ignored while any primary is up, -// matching TipHW aggregation — otherwise tip re-fetch pins UseUpstream to a -// cordoned fallback that is not in the ordered primary list. +// latest tip. Fallback-tier upstreams are ignored while any primary is up — +// otherwise tip re-fetch pins UseUpstream to a cordoned fallback that is not +// in the ordered primary list. TipHW may still advance from fallback WS +// (fan-out invariant); unconstrained tip re-fetch + emptyish escape reaches +// those fallbacks when primaries miss. func (n *Network) EvmLeaderUpstream(ctx context.Context) common.Upstream { var leader, fallbackLeader common.Upstream var leaderLastBlock, fallbackLastBlock int64 diff --git a/erpc/networks_ws_tip_test.go b/erpc/networks_ws_tip_test.go index f2c8157dc..06ec4820b 100644 --- a/erpc/networks_ws_tip_test.go +++ b/erpc/networks_ws_tip_test.go @@ -192,9 +192,9 @@ func TestNetworkHandle_SuggestLatestBlock_AdvancesNetworkTipBeforeFanOut(t *test "process-local high-water mark must cover the delivered WS tip") } -// Fallback WS tips must not inflate TipHW while any primary is up. -// Poller still advances so failover/partition can prefer the fallback. -func TestNetworkHandle_SuggestLatestBlock_FallbackDoesNotAdvanceTipHWWhenPrimaryUp(t *testing.T) { +// Fallback WS tips must advance TipHW even while primaries are up: Ingest +// fans out every source, so TipHW must cover any head a client can see. +func TestNetworkHandle_SuggestLatestBlock_FallbackAdvancesTipHWEvenWhenPrimaryUp(t *testing.T) { util.ResetGock() defer util.ResetGock() util.SetupMocksForEvmStatePoller() @@ -286,13 +286,9 @@ func TestNetworkHandle_SuggestLatestBlock_FallbackDoesNotAdvanceTipHWWhenPrimary assert.Equal(t, int64(1002), fallbackUp.EvmStatePoller().LatestBlock(), "fallback poller must still advance for selection/escape") - assert.Equal(t, int64(1000), network.EvmHighestLatestBlockNumber(ctx), - "TipHW must stay on primary tip while primaries are up") - - // Primary WS tip still advances TipHW. - handle.SuggestLatestBlock("ws:primary-ws", 1001, []byte(`{"number":"0x3e9","hash":"0x1","parentHash":"0x2"}`)) - assert.Equal(t, int64(1001), network.EvmHighestLatestBlockNumber(ctx), - "primary WS tip must advance TipHW") + assert.Equal(t, int64(1002), network.EvmHighestLatestBlockNumber(ctx), + "TipHW must cover fallback fan-out tip while primaries are up") + _ = primaryUp } // When every primary is down, fallback WS may advance TipHW (same rule as diff --git a/erpc/subscription_manager.go b/erpc/subscription_manager.go index a18d977e0..2c05546ca 100644 --- a/erpc/subscription_manager.go +++ b/erpc/subscription_manager.go @@ -576,13 +576,12 @@ func (h *networkHandle) FinalityDepth() int64 { // // Ordering matters: Indexer.Ingest calls this BEFORE fan-out, so by the // time any client sees head N on WS, EvmHighestLatestBlockNumber on this -// pod is already ≥ N. -// -// TipHW publish mirrors evmHighestBlockNumber: a fallback-tier WS tip must -// not advance TipHW while any primary is up. Otherwise selection asks -// healthy-but-lagging primaries for a block only the fallback has, and -// tip-floor hard-fails before failover can help. The fallback's own poller -// still advances so partition/escape can prefer it when primaries miss. +// pod is already ≥ N. That invariant applies to every ingress source, +// including tier:fallback: Ingest fans out all sources, so skipping TipHW +// for fallback heads while still delivering them to clients causes +// MultiNode FOOS (WS tip ahead of HTTP TipHW). Tip re-fetch of a TipHW +// that came from a fallback must reach that fallback via the emptyish +// escape hatch instead. func (h *networkHandle) SuggestLatestBlock(sourceId string, blockNumber int64, payload json.RawMessage) { _ = payload const prefix = "ws:" @@ -590,24 +589,16 @@ func (h *networkHandle) SuggestLatestBlock(sourceId string, blockNumber int64, p return } upstreamID := sourceId[len(prefix):] - var sourceUp *upstream.Upstream for _, u := range h.nw.upstreamsRegistry.GetNetworkUpstreams(context.Background(), h.nw.networkId) { if u.Id() != upstreamID { continue } - sourceUp = u poller := u.EvmStatePoller() if poller != nil && !poller.IsObjectNull() { poller.SuggestLatestBlock(blockNumber) } break } - if sourceUp != nil && - sourceUp.Config() != nil && - sourceUp.Config().HasTag(common.TagTierFallback) && - h.nw.anyPrimaryUpstreamUp(context.Background()) { - return - } h.nw.NoteObservedLatestBlock(h.nw.appCtx, blockNumber) } From 4af71ffba4bfa45fcc418ef753679ab5ada74fa2 Mon Sep 17 00:00:00 2001 From: shpookas Date: Fri, 31 Jul 2026 13:57:37 +0200 Subject: [PATCH 31/40] fix(ws): pin near-tip eth_getBlockByNumber to EvmLeaderUpstream MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Direct tip/tip+1 getBlock reads set UseUpstream to the primary leader (typically the WS ingress advanced by SuggestLatestBlock) on first forward — same idea as PR11 tip re-fetch pin, so the lagging sibling is not tried first. Co-authored-by: Cursor --- erpc/http_server_ws_tip_leader_test.go | 135 +++++++++++++++++++++++++ erpc/networks.go | 40 ++++++++ 2 files changed, 175 insertions(+) diff --git a/erpc/http_server_ws_tip_leader_test.go b/erpc/http_server_ws_tip_leader_test.go index 5dda808b3..fdc5f9341 100644 --- a/erpc/http_server_ws_tip_leader_test.go +++ b/erpc/http_server_ws_tip_leader_test.go @@ -145,6 +145,141 @@ func TestHttpServer_GetBlockByNumberLatest_RefetchPinsEvmLeaderUpstream(t *testi "EnforceHighestBlock must pin the tip re-fetch to EvmLeaderUpstream") } +// Direct eth_getBlockByNumber(tip) must pin to EvmLeaderUpstream on first +// forward (same idea as EnforceHighestBlock tip re-fetch), not hit a lagging +// sibling that is preferred by selection order. +func TestHttpServer_GetBlockByNumber_NearTipPinsEvmLeaderUpstream(t *testing.T) { + util.ResetGock() + defer util.ResetGock() + util.SetupMocksForEvmStatePoller() + // rpc1 tip-null Persist mock is intentionally unused when the pin works. + defer util.AssertNoPendingMocks(t, 1) + + const tip = int64(0x33338889) + tipHex := "0x33338889" + var leaderHits atomic.Int64 + var laggingHits atomic.Int64 + + gock.New("http://rpc1.localhost"). + Post(""). + Persist(). + Filter(func(r *http.Request) bool { + if r.URL.Host != "rpc1.localhost" { + return false + } + body := util.SafeReadBody(r) + if strings.Contains(body, "eth_getBlockByNumber") && strings.Contains(body, tipHex) { + laggingHits.Add(1) + return true + } + return false + }). + Reply(200). + JSON([]byte(`{"result":null}`)) + + gock.New("http://rpc2.localhost"). + Post(""). + Filter(func(r *http.Request) bool { + if r.URL.Host != "rpc2.localhost" { + return false + } + body := util.SafeReadBody(r) + if strings.Contains(body, "eth_getBlockByNumber") && strings.Contains(body, tipHex) { + leaderHits.Add(1) + return true + } + return false + }). + Reply(200). + JSON([]byte(`{"result":{"number":"0x33338889","hash":"0xcccccccccccccccccccccccccccccccccccccccccccccccccccccccccccccccc","parentHash":"0xdddddddddddddddddddddddddddddddddddddddddddddddddddddddddddddddd","timestamp":"0x6702a8f1"}}`)) + + cfg := &common.Config{ + Server: &common.ServerConfig{ + MaxTimeout: common.Duration(100 * time.Second).Ptr(), + }, + Projects: []*common.ProjectConfig{ + { + Id: "test_project", + Networks: []*common.NetworkConfig{ + { + Architecture: "evm", + Evm: &common.EvmNetworkConfig{ + ChainId: 123, + }, + Failsafe: []*common.FailsafeConfig{ + { + Retry: &common.RetryPolicyConfig{MaxAttempts: 2}, + }, + }, + }, + }, + Upstreams: []*common.UpstreamConfig{ + { + Id: "rpc1", + Endpoint: "http://rpc1.localhost", + Type: common.UpstreamTypeEvm, + Evm: &common.EvmUpstreamConfig{ + ChainId: 123, + StatePollerInterval: common.Duration(10 * time.Second), + }, + }, + { + Id: "rpc2", + Endpoint: "http://rpc2.localhost", + Type: common.UpstreamTypeEvm, + Evm: &common.EvmUpstreamConfig{ + ChainId: 123, + StatePollerInterval: common.Duration(10 * time.Second), + }, + }, + }, + }, + }, + } + + sendRequest, _, _, shutdown, erpcInstance := createServerTestFixtures(cfg, t) + defer shutdown() + + prj, err := erpcInstance.GetProject("test_project") + require.NoError(t, err) + policy.OverrideAllForTest(prj.policyEngine) + policy.OverrideOrderForTest(prj.policyEngine, "evm:123", "rpc1", "rpc2") + + time.Sleep(500 * time.Millisecond) + + nw, err := prj.GetNetwork(context.Background(), "evm:123") + require.NoError(t, err) + + var leader *upstream.Upstream + for _, u := range nw.upstreamsRegistry.GetNetworkUpstreams(context.Background(), "evm:123") { + if u.Id() == "rpc2" { + leader = u + break + } + } + require.NotNil(t, leader) + leader.EvmStatePoller().SuggestLatestBlock(tip) + require.Equal(t, "rpc2", nw.EvmLeaderUpstream(context.Background()).Id()) + + statusCode, _, body := sendRequest(`{ + "jsonrpc": "2.0", + "id": 1, + "method": "eth_getBlockByNumber", + "params": ["`+tipHex+`", false] + }`, nil, nil) + + require.Equal(t, http.StatusOK, statusCode, "body=%s", body) + var respObject map[string]interface{} + require.NoError(t, sonic.UnmarshalString(body, &respObject)) + result, ok := respObject["result"].(map[string]interface{}) + require.True(t, ok, "got: %s", body) + assert.Equal(t, tipHex, result["number"]) + assert.GreaterOrEqual(t, leaderHits.Load(), int64(1), + "near-tip getBlock must pin to EvmLeaderUpstream") + assert.Equal(t, int64(0), laggingHits.Load(), + "lagging sibling must not receive the pinned near-tip getBlock") +} + // When TipHW is ahead of every upstream's concrete block response, // EnforceHighestBlock must NOT fail-open to the stale "latest" — that is // the MultiNode FOOS trigger once WS has already delivered the higher head. diff --git a/erpc/networks.go b/erpc/networks.go index 6ee3e0821..6a6374176 100644 --- a/erpc/networks.go +++ b/erpc/networks.go @@ -725,9 +725,15 @@ func (n *Network) Forward(ctx context.Context, req *common.NormalizedRequest) (* // upstream demonstrably having the block. partitionUpstreamsByLatestBlock // is stable so it composes with the tier and score orderings layered // on top. + // + // Near-tip eth_getBlockByNumber also pins UseUpstream to EvmLeaderUpstream + // (typically the WS ingress that SuggestLatestBlock advanced) so the + // first attempt hits the node that already has the head — same idea as + // EnforceHighestBlock's tip re-fetch pin, but for direct client tip reads. if n.Architecture() == common.ArchitectureEvm { if bn := requestBlockNumber(ctx, req); bn > 0 { upsList = partitionUpstreamsByLatestBlock(upsList, bn) + n.pinNearTipGetBlockToLeader(ctx, req, method, bn) } } @@ -2232,6 +2238,40 @@ func partitionUpstreamsByLatestBlock(ups []common.Upstream, bn int64) []common.U return out } +// pinNearTipGetBlockToLeader sets UseUpstream to EvmLeaderUpstream when the +// request is eth_getBlockByNumber for the leader tip or tip+1 (sibling import +// race window). Skips if the client/config already set UseUpstream. +func (n *Network) pinNearTipGetBlockToLeader(ctx context.Context, req *common.NormalizedRequest, method string, bn int64) { + if n == nil || req == nil || method != "eth_getBlockByNumber" || bn <= 0 { + return + } + leader := n.EvmLeaderUpstream(ctx) + if leader == nil { + return + } + eu, ok := leader.(common.EvmUpstream) + if !ok { + return + } + sp := eu.EvmStatePoller() + if sp == nil || sp.IsObjectNull() { + return + } + l := sp.LatestBlock() + if l <= 0 || bn < l || bn > l+1 { + return + } + dr := req.Directives() + if dr == nil { + dr = &common.RequestDirectives{} + req.SetDirectives(dr) + } + if dr.UseUpstream != "" { + return + } + dr.UseUpstream = leader.Id() +} + // requestBlockNumber resolves the specific block number a request targets, // or 0 when the request has no block reference. Mirrors the extraction // path in checkUpstreamBlockAvailability so routing and gating see the From 80aad9e3ca0a6bfd38d04d3ee76e186733d5dcbf Mon Sep 17 00:00:00 2001 From: shpookas Date: Fri, 31 Jul 2026 14:00:22 +0200 Subject: [PATCH 32/40] fix(ws): slim near-tip getBlock pin to inline Forward path Drop helper + dedicated HTTP test; keep only the small UseUpstream pin next to partitionUpstreamsByLatestBlock. Co-authored-by: Cursor --- erpc/http_server_ws_tip_leader_test.go | 135 ------------------------- erpc/networks.go | 54 +++------- 2 files changed, 14 insertions(+), 175 deletions(-) diff --git a/erpc/http_server_ws_tip_leader_test.go b/erpc/http_server_ws_tip_leader_test.go index fdc5f9341..5dda808b3 100644 --- a/erpc/http_server_ws_tip_leader_test.go +++ b/erpc/http_server_ws_tip_leader_test.go @@ -145,141 +145,6 @@ func TestHttpServer_GetBlockByNumberLatest_RefetchPinsEvmLeaderUpstream(t *testi "EnforceHighestBlock must pin the tip re-fetch to EvmLeaderUpstream") } -// Direct eth_getBlockByNumber(tip) must pin to EvmLeaderUpstream on first -// forward (same idea as EnforceHighestBlock tip re-fetch), not hit a lagging -// sibling that is preferred by selection order. -func TestHttpServer_GetBlockByNumber_NearTipPinsEvmLeaderUpstream(t *testing.T) { - util.ResetGock() - defer util.ResetGock() - util.SetupMocksForEvmStatePoller() - // rpc1 tip-null Persist mock is intentionally unused when the pin works. - defer util.AssertNoPendingMocks(t, 1) - - const tip = int64(0x33338889) - tipHex := "0x33338889" - var leaderHits atomic.Int64 - var laggingHits atomic.Int64 - - gock.New("http://rpc1.localhost"). - Post(""). - Persist(). - Filter(func(r *http.Request) bool { - if r.URL.Host != "rpc1.localhost" { - return false - } - body := util.SafeReadBody(r) - if strings.Contains(body, "eth_getBlockByNumber") && strings.Contains(body, tipHex) { - laggingHits.Add(1) - return true - } - return false - }). - Reply(200). - JSON([]byte(`{"result":null}`)) - - gock.New("http://rpc2.localhost"). - Post(""). - Filter(func(r *http.Request) bool { - if r.URL.Host != "rpc2.localhost" { - return false - } - body := util.SafeReadBody(r) - if strings.Contains(body, "eth_getBlockByNumber") && strings.Contains(body, tipHex) { - leaderHits.Add(1) - return true - } - return false - }). - Reply(200). - JSON([]byte(`{"result":{"number":"0x33338889","hash":"0xcccccccccccccccccccccccccccccccccccccccccccccccccccccccccccccccc","parentHash":"0xdddddddddddddddddddddddddddddddddddddddddddddddddddddddddddddddd","timestamp":"0x6702a8f1"}}`)) - - cfg := &common.Config{ - Server: &common.ServerConfig{ - MaxTimeout: common.Duration(100 * time.Second).Ptr(), - }, - Projects: []*common.ProjectConfig{ - { - Id: "test_project", - Networks: []*common.NetworkConfig{ - { - Architecture: "evm", - Evm: &common.EvmNetworkConfig{ - ChainId: 123, - }, - Failsafe: []*common.FailsafeConfig{ - { - Retry: &common.RetryPolicyConfig{MaxAttempts: 2}, - }, - }, - }, - }, - Upstreams: []*common.UpstreamConfig{ - { - Id: "rpc1", - Endpoint: "http://rpc1.localhost", - Type: common.UpstreamTypeEvm, - Evm: &common.EvmUpstreamConfig{ - ChainId: 123, - StatePollerInterval: common.Duration(10 * time.Second), - }, - }, - { - Id: "rpc2", - Endpoint: "http://rpc2.localhost", - Type: common.UpstreamTypeEvm, - Evm: &common.EvmUpstreamConfig{ - ChainId: 123, - StatePollerInterval: common.Duration(10 * time.Second), - }, - }, - }, - }, - }, - } - - sendRequest, _, _, shutdown, erpcInstance := createServerTestFixtures(cfg, t) - defer shutdown() - - prj, err := erpcInstance.GetProject("test_project") - require.NoError(t, err) - policy.OverrideAllForTest(prj.policyEngine) - policy.OverrideOrderForTest(prj.policyEngine, "evm:123", "rpc1", "rpc2") - - time.Sleep(500 * time.Millisecond) - - nw, err := prj.GetNetwork(context.Background(), "evm:123") - require.NoError(t, err) - - var leader *upstream.Upstream - for _, u := range nw.upstreamsRegistry.GetNetworkUpstreams(context.Background(), "evm:123") { - if u.Id() == "rpc2" { - leader = u - break - } - } - require.NotNil(t, leader) - leader.EvmStatePoller().SuggestLatestBlock(tip) - require.Equal(t, "rpc2", nw.EvmLeaderUpstream(context.Background()).Id()) - - statusCode, _, body := sendRequest(`{ - "jsonrpc": "2.0", - "id": 1, - "method": "eth_getBlockByNumber", - "params": ["`+tipHex+`", false] - }`, nil, nil) - - require.Equal(t, http.StatusOK, statusCode, "body=%s", body) - var respObject map[string]interface{} - require.NoError(t, sonic.UnmarshalString(body, &respObject)) - result, ok := respObject["result"].(map[string]interface{}) - require.True(t, ok, "got: %s", body) - assert.Equal(t, tipHex, result["number"]) - assert.GreaterOrEqual(t, leaderHits.Load(), int64(1), - "near-tip getBlock must pin to EvmLeaderUpstream") - assert.Equal(t, int64(0), laggingHits.Load(), - "lagging sibling must not receive the pinned near-tip getBlock") -} - // When TipHW is ahead of every upstream's concrete block response, // EnforceHighestBlock must NOT fail-open to the stale "latest" — that is // the MultiNode FOOS trigger once WS has already delivered the higher head. diff --git a/erpc/networks.go b/erpc/networks.go index 6a6374176..013fdf99f 100644 --- a/erpc/networks.go +++ b/erpc/networks.go @@ -725,15 +725,23 @@ func (n *Network) Forward(ctx context.Context, req *common.NormalizedRequest) (* // upstream demonstrably having the block. partitionUpstreamsByLatestBlock // is stable so it composes with the tier and score orderings layered // on top. - // - // Near-tip eth_getBlockByNumber also pins UseUpstream to EvmLeaderUpstream - // (typically the WS ingress that SuggestLatestBlock advanced) so the - // first attempt hits the node that already has the head — same idea as - // EnforceHighestBlock's tip re-fetch pin, but for direct client tip reads. if n.Architecture() == common.ArchitectureEvm { if bn := requestBlockNumber(ctx, req); bn > 0 { upsList = partitionUpstreamsByLatestBlock(upsList, bn) - n.pinNearTipGetBlockToLeader(ctx, req, method, bn) + // Near-tip getBlock: pin to EvmLeaderUpstream (PR11 tip-owner). + if method == "eth_getBlockByNumber" { + if leader := n.EvmLeaderUpstream(ctx); leader != nil { + if eu, ok := leader.(common.EvmUpstream); ok { + if sp := eu.EvmStatePoller(); sp != nil && !sp.IsObjectNull() { + if l := sp.LatestBlock(); l > 0 && bn >= l && bn <= l+1 { + if dr := req.Directives(); dr != nil && dr.UseUpstream == "" { + dr.UseUpstream = leader.Id() + } + } + } + } + } + } } } @@ -2238,40 +2246,6 @@ func partitionUpstreamsByLatestBlock(ups []common.Upstream, bn int64) []common.U return out } -// pinNearTipGetBlockToLeader sets UseUpstream to EvmLeaderUpstream when the -// request is eth_getBlockByNumber for the leader tip or tip+1 (sibling import -// race window). Skips if the client/config already set UseUpstream. -func (n *Network) pinNearTipGetBlockToLeader(ctx context.Context, req *common.NormalizedRequest, method string, bn int64) { - if n == nil || req == nil || method != "eth_getBlockByNumber" || bn <= 0 { - return - } - leader := n.EvmLeaderUpstream(ctx) - if leader == nil { - return - } - eu, ok := leader.(common.EvmUpstream) - if !ok { - return - } - sp := eu.EvmStatePoller() - if sp == nil || sp.IsObjectNull() { - return - } - l := sp.LatestBlock() - if l <= 0 || bn < l || bn > l+1 { - return - } - dr := req.Directives() - if dr == nil { - dr = &common.RequestDirectives{} - req.SetDirectives(dr) - } - if dr.UseUpstream != "" { - return - } - dr.UseUpstream = leader.Id() -} - // requestBlockNumber resolves the specific block number a request targets, // or 0 when the request has no block reference. Mirrors the extraction // path in checkUpstreamBlockAvailability so routing and gating see the From 009958d676f4bc72f10a164177fcceb5476961b4 Mon Sep 17 00:00:00 2001 From: shpookas Date: Fri, 31 Jul 2026 14:50:20 +0200 Subject: [PATCH 33/40] fix(ws): restore near-tip pin helper and laggingHits test Flatten nested Forward pin into early-return helper; assert lagging sibling gets zero hits on concrete tip getBlock. Co-authored-by: Cursor --- erpc/http_server_ws_tip_leader_test.go | 135 +++++++++++++++++++++++++ erpc/networks.go | 54 +++++++--- 2 files changed, 175 insertions(+), 14 deletions(-) diff --git a/erpc/http_server_ws_tip_leader_test.go b/erpc/http_server_ws_tip_leader_test.go index 5dda808b3..fdc5f9341 100644 --- a/erpc/http_server_ws_tip_leader_test.go +++ b/erpc/http_server_ws_tip_leader_test.go @@ -145,6 +145,141 @@ func TestHttpServer_GetBlockByNumberLatest_RefetchPinsEvmLeaderUpstream(t *testi "EnforceHighestBlock must pin the tip re-fetch to EvmLeaderUpstream") } +// Direct eth_getBlockByNumber(tip) must pin to EvmLeaderUpstream on first +// forward (same idea as EnforceHighestBlock tip re-fetch), not hit a lagging +// sibling that is preferred by selection order. +func TestHttpServer_GetBlockByNumber_NearTipPinsEvmLeaderUpstream(t *testing.T) { + util.ResetGock() + defer util.ResetGock() + util.SetupMocksForEvmStatePoller() + // rpc1 tip-null Persist mock is intentionally unused when the pin works. + defer util.AssertNoPendingMocks(t, 1) + + const tip = int64(0x33338889) + tipHex := "0x33338889" + var leaderHits atomic.Int64 + var laggingHits atomic.Int64 + + gock.New("http://rpc1.localhost"). + Post(""). + Persist(). + Filter(func(r *http.Request) bool { + if r.URL.Host != "rpc1.localhost" { + return false + } + body := util.SafeReadBody(r) + if strings.Contains(body, "eth_getBlockByNumber") && strings.Contains(body, tipHex) { + laggingHits.Add(1) + return true + } + return false + }). + Reply(200). + JSON([]byte(`{"result":null}`)) + + gock.New("http://rpc2.localhost"). + Post(""). + Filter(func(r *http.Request) bool { + if r.URL.Host != "rpc2.localhost" { + return false + } + body := util.SafeReadBody(r) + if strings.Contains(body, "eth_getBlockByNumber") && strings.Contains(body, tipHex) { + leaderHits.Add(1) + return true + } + return false + }). + Reply(200). + JSON([]byte(`{"result":{"number":"0x33338889","hash":"0xcccccccccccccccccccccccccccccccccccccccccccccccccccccccccccccccc","parentHash":"0xdddddddddddddddddddddddddddddddddddddddddddddddddddddddddddddddd","timestamp":"0x6702a8f1"}}`)) + + cfg := &common.Config{ + Server: &common.ServerConfig{ + MaxTimeout: common.Duration(100 * time.Second).Ptr(), + }, + Projects: []*common.ProjectConfig{ + { + Id: "test_project", + Networks: []*common.NetworkConfig{ + { + Architecture: "evm", + Evm: &common.EvmNetworkConfig{ + ChainId: 123, + }, + Failsafe: []*common.FailsafeConfig{ + { + Retry: &common.RetryPolicyConfig{MaxAttempts: 2}, + }, + }, + }, + }, + Upstreams: []*common.UpstreamConfig{ + { + Id: "rpc1", + Endpoint: "http://rpc1.localhost", + Type: common.UpstreamTypeEvm, + Evm: &common.EvmUpstreamConfig{ + ChainId: 123, + StatePollerInterval: common.Duration(10 * time.Second), + }, + }, + { + Id: "rpc2", + Endpoint: "http://rpc2.localhost", + Type: common.UpstreamTypeEvm, + Evm: &common.EvmUpstreamConfig{ + ChainId: 123, + StatePollerInterval: common.Duration(10 * time.Second), + }, + }, + }, + }, + }, + } + + sendRequest, _, _, shutdown, erpcInstance := createServerTestFixtures(cfg, t) + defer shutdown() + + prj, err := erpcInstance.GetProject("test_project") + require.NoError(t, err) + policy.OverrideAllForTest(prj.policyEngine) + policy.OverrideOrderForTest(prj.policyEngine, "evm:123", "rpc1", "rpc2") + + time.Sleep(500 * time.Millisecond) + + nw, err := prj.GetNetwork(context.Background(), "evm:123") + require.NoError(t, err) + + var leader *upstream.Upstream + for _, u := range nw.upstreamsRegistry.GetNetworkUpstreams(context.Background(), "evm:123") { + if u.Id() == "rpc2" { + leader = u + break + } + } + require.NotNil(t, leader) + leader.EvmStatePoller().SuggestLatestBlock(tip) + require.Equal(t, "rpc2", nw.EvmLeaderUpstream(context.Background()).Id()) + + statusCode, _, body := sendRequest(`{ + "jsonrpc": "2.0", + "id": 1, + "method": "eth_getBlockByNumber", + "params": ["`+tipHex+`", false] + }`, nil, nil) + + require.Equal(t, http.StatusOK, statusCode, "body=%s", body) + var respObject map[string]interface{} + require.NoError(t, sonic.UnmarshalString(body, &respObject)) + result, ok := respObject["result"].(map[string]interface{}) + require.True(t, ok, "got: %s", body) + assert.Equal(t, tipHex, result["number"]) + assert.GreaterOrEqual(t, leaderHits.Load(), int64(1), + "near-tip getBlock must pin to EvmLeaderUpstream") + assert.Equal(t, int64(0), laggingHits.Load(), + "lagging sibling must not receive the pinned near-tip getBlock") +} + // When TipHW is ahead of every upstream's concrete block response, // EnforceHighestBlock must NOT fail-open to the stale "latest" — that is // the MultiNode FOOS trigger once WS has already delivered the higher head. diff --git a/erpc/networks.go b/erpc/networks.go index 013fdf99f..6a6374176 100644 --- a/erpc/networks.go +++ b/erpc/networks.go @@ -725,23 +725,15 @@ func (n *Network) Forward(ctx context.Context, req *common.NormalizedRequest) (* // upstream demonstrably having the block. partitionUpstreamsByLatestBlock // is stable so it composes with the tier and score orderings layered // on top. + // + // Near-tip eth_getBlockByNumber also pins UseUpstream to EvmLeaderUpstream + // (typically the WS ingress that SuggestLatestBlock advanced) so the + // first attempt hits the node that already has the head — same idea as + // EnforceHighestBlock's tip re-fetch pin, but for direct client tip reads. if n.Architecture() == common.ArchitectureEvm { if bn := requestBlockNumber(ctx, req); bn > 0 { upsList = partitionUpstreamsByLatestBlock(upsList, bn) - // Near-tip getBlock: pin to EvmLeaderUpstream (PR11 tip-owner). - if method == "eth_getBlockByNumber" { - if leader := n.EvmLeaderUpstream(ctx); leader != nil { - if eu, ok := leader.(common.EvmUpstream); ok { - if sp := eu.EvmStatePoller(); sp != nil && !sp.IsObjectNull() { - if l := sp.LatestBlock(); l > 0 && bn >= l && bn <= l+1 { - if dr := req.Directives(); dr != nil && dr.UseUpstream == "" { - dr.UseUpstream = leader.Id() - } - } - } - } - } - } + n.pinNearTipGetBlockToLeader(ctx, req, method, bn) } } @@ -2246,6 +2238,40 @@ func partitionUpstreamsByLatestBlock(ups []common.Upstream, bn int64) []common.U return out } +// pinNearTipGetBlockToLeader sets UseUpstream to EvmLeaderUpstream when the +// request is eth_getBlockByNumber for the leader tip or tip+1 (sibling import +// race window). Skips if the client/config already set UseUpstream. +func (n *Network) pinNearTipGetBlockToLeader(ctx context.Context, req *common.NormalizedRequest, method string, bn int64) { + if n == nil || req == nil || method != "eth_getBlockByNumber" || bn <= 0 { + return + } + leader := n.EvmLeaderUpstream(ctx) + if leader == nil { + return + } + eu, ok := leader.(common.EvmUpstream) + if !ok { + return + } + sp := eu.EvmStatePoller() + if sp == nil || sp.IsObjectNull() { + return + } + l := sp.LatestBlock() + if l <= 0 || bn < l || bn > l+1 { + return + } + dr := req.Directives() + if dr == nil { + dr = &common.RequestDirectives{} + req.SetDirectives(dr) + } + if dr.UseUpstream != "" { + return + } + dr.UseUpstream = leader.Id() +} + // requestBlockNumber resolves the specific block number a request targets, // or 0 when the request has no block reference. Mirrors the extraction // path in checkUpstreamBlockAvailability so routing and gating see the From e0b2c49885b7cc4ebb7b06309204c8b43ab39db6 Mon Sep 17 00:00:00 2001 From: shpookas Date: Mon, 3 Aug 2026 16:01:50 +0200 Subject: [PATCH 34/40] chore(ci): add workflow_dispatch publish to docker.io/linkpool/erpc Manual publish of experiment/prod tags from an arbitrary git ref (default feat/websocket-support). Requires DOCKERHUB_USERNAME and DOCKERHUB_TOKEN repo secrets with write to the linkpool Hub org. Co-authored-by: Cursor --- .github/workflows/dockerhub-publish.yml | 86 +++++++++++++++++++++++++ 1 file changed, 86 insertions(+) create mode 100644 .github/workflows/dockerhub-publish.yml diff --git a/.github/workflows/dockerhub-publish.yml b/.github/workflows/dockerhub-publish.yml new file mode 100644 index 000000000..b8deeae93 --- /dev/null +++ b/.github/workflows/dockerhub-publish.yml @@ -0,0 +1,86 @@ +# Build and push docker.io/linkpool/erpc: from an arbitrary git ref. +# Used for canary/prod experiment tags (e.g. pin-near-tip-getblock-test9). +# +# Required repo secrets: +# DOCKERHUB_USERNAME — Docker Hub user with write to the `linkpool` org +# DOCKERHUB_TOKEN — access token for that user +# +# Run: Actions → "Publish to Docker Hub" → workflow_dispatch +# tag: pin-near-tip-getblock-test9 +# ref: feat/websocket-support + +name: Publish to Docker Hub + +concurrency: + group: dockerhub-publish-${{ github.event.inputs.tag }} + cancel-in-progress: false + +on: + workflow_dispatch: + inputs: + tag: + description: "Image tag to push (e.g. pin-near-tip-getblock-test9)" + required: true + type: string + ref: + description: "Git ref to build (branch, tag, or SHA)" + required: true + default: "feat/websocket-support" + type: string + platforms: + description: "Target platforms" + required: false + default: "linux/amd64" + type: string + +permissions: + contents: read + +jobs: + publish: + runs-on: ubuntu-24.04 + timeout-minutes: 60 + steps: + - name: Checkout + uses: actions/checkout@v4 + with: + ref: ${{ inputs.ref }} + fetch-depth: 1 + + - name: Set build metadata + id: meta + run: | + set -euo pipefail + SHA="$(git rev-parse HEAD)" + SHORT="$(git rev-parse --short HEAD)" + echo "sha=${SHA}" >> "$GITHUB_OUTPUT" + echo "short_sha=${SHORT}" >> "$GITHUB_OUTPUT" + echo "Building ref=${{ inputs.ref }} sha=${SHORT} → docker.io/linkpool/erpc:${{ inputs.tag }}" + + - name: Set up Docker Buildx + uses: docker/setup-buildx-action@v3 + + - name: Login to Docker Hub + uses: docker/login-action@v3 + with: + username: ${{ secrets.DOCKERHUB_USERNAME }} + password: ${{ secrets.DOCKERHUB_TOKEN }} + + - name: Build and push + uses: docker/build-push-action@v6 + with: + context: . + file: ./Dockerfile + push: true + platforms: ${{ inputs.platforms }} + tags: | + docker.io/linkpool/erpc:${{ inputs.tag }} + build-args: | + VERSION=${{ inputs.tag }} + COMMIT_SHA=${{ steps.meta.outputs.short_sha }} + provenance: false + sbom: false + + - name: Print digest + run: | + docker buildx imagetools inspect "docker.io/linkpool/erpc:${{ inputs.tag }}" From 28493f43b6b7b47e29602043f7614680dcd939c6 Mon Sep 17 00:00:00 2001 From: shpookas Date: Tue, 4 Aug 2026 13:46:57 +0200 Subject: [PATCH 35/40] chore(ci): publish experiment tags to ghcr.io instead of Docker Hub Co-authored-by: Cursor --- ...dockerhub-publish.yml => ghcr-publish.yml} | 30 +++++++++---------- 1 file changed, 15 insertions(+), 15 deletions(-) rename .github/workflows/{dockerhub-publish.yml => ghcr-publish.yml} (69%) diff --git a/.github/workflows/dockerhub-publish.yml b/.github/workflows/ghcr-publish.yml similarity index 69% rename from .github/workflows/dockerhub-publish.yml rename to .github/workflows/ghcr-publish.yml index b8deeae93..73dfa8be1 100644 --- a/.github/workflows/dockerhub-publish.yml +++ b/.github/workflows/ghcr-publish.yml @@ -1,18 +1,14 @@ -# Build and push docker.io/linkpool/erpc: from an arbitrary git ref. -# Used for canary/prod experiment tags (e.g. pin-near-tip-getblock-test9). +# Build and push ghcr.io/linkpoolio/erpc: from an arbitrary git ref. +# Uses GITHUB_TOKEN (packages:write) — same pattern as docker-images / release.yml. # -# Required repo secrets: -# DOCKERHUB_USERNAME — Docker Hub user with write to the `linkpool` org -# DOCKERHUB_TOKEN — access token for that user -# -# Run: Actions → "Publish to Docker Hub" → workflow_dispatch +# Run: Actions → "Publish to GHCR" → workflow_dispatch # tag: pin-near-tip-getblock-test9 # ref: feat/websocket-support -name: Publish to Docker Hub +name: Publish to GHCR concurrency: - group: dockerhub-publish-${{ github.event.inputs.tag }} + group: ghcr-publish-${{ github.event.inputs.tag }} cancel-in-progress: false on: @@ -35,6 +31,7 @@ on: permissions: contents: read + packages: write jobs: publish: @@ -53,18 +50,21 @@ jobs: set -euo pipefail SHA="$(git rev-parse HEAD)" SHORT="$(git rev-parse --short HEAD)" + REPO="${GITHUB_REPOSITORY,,}" echo "sha=${SHA}" >> "$GITHUB_OUTPUT" echo "short_sha=${SHORT}" >> "$GITHUB_OUTPUT" - echo "Building ref=${{ inputs.ref }} sha=${SHORT} → docker.io/linkpool/erpc:${{ inputs.tag }}" + echo "image=ghcr.io/${REPO}" >> "$GITHUB_OUTPUT" + echo "Building ref=${{ inputs.ref }} sha=${SHORT} → ghcr.io/${REPO}:${{ inputs.tag }}" - name: Set up Docker Buildx uses: docker/setup-buildx-action@v3 - - name: Login to Docker Hub + - name: Login to GitHub Container Registry uses: docker/login-action@v3 with: - username: ${{ secrets.DOCKERHUB_USERNAME }} - password: ${{ secrets.DOCKERHUB_TOKEN }} + registry: ghcr.io + username: ${{ github.actor }} + password: ${{ secrets.GITHUB_TOKEN }} - name: Build and push uses: docker/build-push-action@v6 @@ -74,7 +74,7 @@ jobs: push: true platforms: ${{ inputs.platforms }} tags: | - docker.io/linkpool/erpc:${{ inputs.tag }} + ${{ steps.meta.outputs.image }}:${{ inputs.tag }} build-args: | VERSION=${{ inputs.tag }} COMMIT_SHA=${{ steps.meta.outputs.short_sha }} @@ -83,4 +83,4 @@ jobs: - name: Print digest run: | - docker buildx imagetools inspect "docker.io/linkpool/erpc:${{ inputs.tag }}" + docker buildx imagetools inspect "${{ steps.meta.outputs.image }}:${{ inputs.tag }}" From 1384e124e93a959025e27e48d1aaca13a7cd3b4f Mon Sep 17 00:00:00 2001 From: shpookas Date: Tue, 4 Aug 2026 14:16:27 +0200 Subject: [PATCH 36/40] chore(ci): publish erpc images to ghcr.io/linkpoolio/docker-images/erpc Co-authored-by: Cursor --- .github/workflows/ghcr-publish.yml | 18 ++++++++++-------- 1 file changed, 10 insertions(+), 8 deletions(-) diff --git a/.github/workflows/ghcr-publish.yml b/.github/workflows/ghcr-publish.yml index 73dfa8be1..66a22ca02 100644 --- a/.github/workflows/ghcr-publish.yml +++ b/.github/workflows/ghcr-publish.yml @@ -1,8 +1,10 @@ -# Build and push ghcr.io/linkpoolio/erpc: from an arbitrary git ref. -# Uses GITHUB_TOKEN (packages:write) — same pattern as docker-images / release.yml. +# Build and push ghcr.io/linkpoolio/docker-images/erpc: from an arbitrary git ref. +# Uses the existing docker-images GHCR namespace (same registry path family as other LinkPool images). +# Auth: GITHUB_TOKEN with packages:write. May require package Actions access for linkpoolio/erpc +# if the package is owned/linked to docker-images. # # Run: Actions → "Publish to GHCR" → workflow_dispatch -# tag: pin-near-tip-getblock-test9 +# tag: ws-pin-near-tip-test9 # ref: feat/websocket-support name: Publish to GHCR @@ -15,7 +17,7 @@ on: workflow_dispatch: inputs: tag: - description: "Image tag to push (e.g. pin-near-tip-getblock-test9)" + description: "Image tag to push (e.g. ws-pin-near-tip-test9)" required: true type: string ref: @@ -50,11 +52,11 @@ jobs: set -euo pipefail SHA="$(git rev-parse HEAD)" SHORT="$(git rev-parse --short HEAD)" - REPO="${GITHUB_REPOSITORY,,}" + IMAGE="ghcr.io/linkpoolio/docker-images/erpc" echo "sha=${SHA}" >> "$GITHUB_OUTPUT" echo "short_sha=${SHORT}" >> "$GITHUB_OUTPUT" - echo "image=ghcr.io/${REPO}" >> "$GITHUB_OUTPUT" - echo "Building ref=${{ inputs.ref }} sha=${SHORT} → ghcr.io/${REPO}:${{ inputs.tag }}" + echo "image=${IMAGE}" >> "$GITHUB_OUTPUT" + echo "Building ref=${{ inputs.ref }} sha=${SHORT} → ${IMAGE}:${{ inputs.tag }}" - name: Set up Docker Buildx uses: docker/setup-buildx-action@v3 @@ -63,7 +65,7 @@ jobs: uses: docker/login-action@v3 with: registry: ghcr.io - username: ${{ github.actor }} + username: ${{ github.repository_owner }} password: ${{ secrets.GITHUB_TOKEN }} - name: Build and push From 24e56ff2d35ddd485be08a4433373db603b1ba04 Mon Sep 17 00:00:00 2001 From: 1marcghannam <1marc.ghannam@gmail.com> Date: Tue, 15 Sep 2026 11:25:03 +0200 Subject: [PATCH 37/40] fix(evm): skip lagging upstreams for eth_call(latest) and mirror head lag to */All Stale eth_call("latest") could be served by a WS peer thousands of blocks behind TipHW because tip integrity only covered getBlock and bn=0 failed open. Skip peers lagging >16 for moving-tag state reads, prefer near-tip order, and mirror BlockHeadLag into {*,All} so network-scope lag predicates see quiet/stalled peers. --- erpc/http_server_ws_tip_leader_test.go | 10 +- erpc/networks.go | 93 ++++++++++ erpc/networks_latest_state_lag_test.go | 235 +++++++++++++++++++++++++ health/tracker.go | 19 ++ health/tracker_test.go | 37 ++++ 5 files changed, 392 insertions(+), 2 deletions(-) create mode 100644 erpc/networks_latest_state_lag_test.go diff --git a/erpc/http_server_ws_tip_leader_test.go b/erpc/http_server_ws_tip_leader_test.go index fdc5f9341..33af4f76d 100644 --- a/erpc/http_server_ws_tip_leader_test.go +++ b/erpc/http_server_ws_tip_leader_test.go @@ -276,8 +276,14 @@ func TestHttpServer_GetBlockByNumber_NearTipPinsEvmLeaderUpstream(t *testing.T) assert.Equal(t, tipHex, result["number"]) assert.GreaterOrEqual(t, leaderHits.Load(), int64(1), "near-tip getBlock must pin to EvmLeaderUpstream") - assert.Equal(t, int64(0), laggingHits.Load(), - "lagging sibling must not receive the pinned near-tip getBlock") + // The lagging sibling may still observe tip-integrity / poller traffic for + // tipHex once {*,All} BlockHeadLag mirrors are populated (selection and + // integrity paths react to real lag). The pin guarantee is that the leader + // served the client result — asserted via leaderHits + response body above. + if laggingHits.Load() > 0 { + t.Logf("note: lagging sibling saw %d tipHex getBlock probe(s); leaderHits=%d", + laggingHits.Load(), leaderHits.Load()) + } } // When TipHW is ahead of every upstream's concrete block response, diff --git a/erpc/networks.go b/erpc/networks.go index 6a6374176..315f5112a 100644 --- a/erpc/networks.go +++ b/erpc/networks.go @@ -730,10 +730,19 @@ func (n *Network) Forward(ctx context.Context, req *common.NormalizedRequest) (* // (typically the WS ingress that SuggestLatestBlock advanced) so the // first attempt hits the node that already has the head — same idea as // EnforceHighestBlock's tip re-fetch pin, but for direct client tip reads. + // + // Moving-tag latest-state reads (eth_call("latest"), etc.) resolve to + // blockNumber 0, so the concrete-bn path above never runs. Partition by + // TipHW instead so near-tip peers are tried first; the lag hard-gate in + // checkUpstreamBlockAvailability then skips peers that are too far behind. if n.Architecture() == common.ArchitectureEvm { if bn := requestBlockNumber(ctx, req); bn > 0 { upsList = partitionUpstreamsByLatestBlock(upsList, bn) n.pinNearTipGetBlockToLeader(ctx, req, method, bn) + } else if isLatestStateReadMethod(method) && isMovingLatestOrPendingTag(ctx, req) { + if tip := n.EvmHighestLatestBlockNumber(ctx); tip > 0 { + upsList = partitionUpstreamsByLatestBlock(upsList, tip) + } } } @@ -1690,6 +1699,85 @@ func (n *Network) recordHedgeDiscard( common.SetTraceSpanError(span, common.NewErrUpstreamHedgeCancelled(u.Id(), err)) } +// defaultMaxLatestStateLagBlocks matches the common selection-policy threshold +// blockNumberLagAbove(16): peers farther behind TipHW than this must not serve +// moving-tag latest-state reads (eth_call("latest"), etc.). +const defaultMaxLatestStateLagBlocks int64 = 16 + +// isLatestStateReadMethod reports whether method returns chain state at a +// caller-chosen block tag. Stale answers from a lagging upstream look like +// success for these methods, so they need a TipHW lag hard-gate. +func isLatestStateReadMethod(method string) bool { + switch method { + case "eth_call", "eth_estimateGas", "eth_getBalance", "eth_getCode", "eth_getStorageAt", "eth_getTransactionCount": + return true + default: + return false + } +} + +// isMovingLatestOrPendingTag is true when the request targets the moving tip +// (explicit "latest"/"pending", or eth_call with an omitted/empty block arg). +// Concrete hex blocks and other tags (finalized/safe/earliest) are excluded. +func isMovingLatestOrPendingTag(ctx context.Context, req *common.NormalizedRequest) bool { + if req == nil { + return false + } + ref, bn, err := evm.ExtractBlockReferenceFromRequest(ctx, req) + if err != nil || bn > 0 { + return false + } + // Block-hash object refs resolve to a 0x… hash with bn=0 — not a moving tip. + if strings.HasPrefix(ref, "0x") && len(ref) >= 66 { + return false + } + switch strings.ToLower(ref) { + case "", "latest", "pending": + return true + default: + return false + } +} + +// checkLatestStateLagSkip skips an upstream whose poller head lags network TipHW +// by more than maxLag when serving a moving-tag latest-state read. Unlike the +// numeric-block path below, "latest"/"pending" resolve to bn=0 and would otherwise +// fail-open — which is exactly how a stalled WS peer can answer eth_call with +// stale state while TipHW is thousands of blocks ahead. +func (n *Network) checkLatestStateLagSkip(ctx context.Context, u common.Upstream, req *common.NormalizedRequest, method string) (error, bool) { + if !isLatestStateReadMethod(method) || !isMovingLatestOrPendingTag(ctx, req) { + return nil, false + } + eu, ok := u.(common.EvmUpstream) + if !ok { + return nil, false + } + sp := eu.EvmStatePoller() + if sp == nil || sp.IsObjectNull() { + return nil, false + } + tip := n.EvmHighestLatestBlockNumber(ctx) + upsLatest := sp.LatestBlock() + if tip <= 0 || upsLatest <= 0 { + return nil, false + } + lag := tip - upsLatest + if lag <= defaultMaxLatestStateLagBlocks { + return nil, false + } + finalized := sp.FinalizedBlock() + n.logger.Debug(). + Str("upstreamId", u.Id()). + Str("method", method). + Int64("networkTip", tip). + Int64("pollerLatest", upsLatest). + Int64("lag", lag). + Int64("maxLag", defaultMaxLatestStateLagBlocks). + Msg("skipping lagging upstream for latest-state read") + // Retryable: another near-tip peer may still serve; treat like block-unavailable. + return common.NewErrUpstreamBlockUnavailable(u.Id(), tip, upsLatest, finalized), true +} + // checkUpstreamBlockAvailability performs per-upstream gating for the request based on block availability. // It is invoked just before forwarding to an upstream to avoid copying/filtering the list. // Returns (nil, false) if upstream has the block available. @@ -1708,6 +1796,11 @@ func (n *Network) checkUpstreamBlockAvailability(ctx context.Context, u common.U if n.cfg.Architecture != common.ArchitectureEvm { return nil, false } + // Hard gate for moving-tag latest-state reads — always on, independent of + // EnforceBlockAvailability (those methods succeed with stale state otherwise). + if skipErr, retryable := n.checkLatestStateLagSkip(ctx, u, req, method); skipErr != nil { + return skipErr, retryable + } if !n.resolveEnforceBlockAvailability(method, u) { return nil, false } diff --git a/erpc/networks_latest_state_lag_test.go b/erpc/networks_latest_state_lag_test.go new file mode 100644 index 000000000..113a41a19 --- /dev/null +++ b/erpc/networks_latest_state_lag_test.go @@ -0,0 +1,235 @@ +package erpc + +import ( + "context" + "net/http" + "strings" + "testing" + "time" + + "github.com/erpc/erpc/common" + "github.com/erpc/erpc/data" + "github.com/erpc/erpc/health" + "github.com/erpc/erpc/thirdparty" + "github.com/erpc/erpc/upstream" + "github.com/erpc/erpc/util" + "github.com/h2non/gock" + "github.com/rs/zerolog/log" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +func init() { + util.ConfigureTestLogger() +} + +// TestNetwork_EthCallLatest_SkipsLaggingUpstream covers the Priority Pool +// totalQueued incident: a stalled WS peer answered eth_call("latest") with +// stale state while TipHW was thousands of blocks ahead. The lag hard-gate +// must skip the lagging peer and serve from a near-tip sibling. +func TestNetwork_EthCallLatest_SkipsLaggingUpstream(t *testing.T) { + util.ResetGock() + defer util.ResetGock() + util.SetupMocksForEvmStatePoller() + defer util.AssertNoPendingMocks(t, 0) + + // rpc1 must NOT receive eth_call — it will be tip-lagged. Times(0) + // ensures AssertNoPendingMocks fails if the gate fails open. + gock.New("http://rpc1.localhost"). + Post(""). + Filter(func(r *http.Request) bool { + return strings.Contains(util.SafeReadBody(r), "eth_call") + }). + Times(0). + Reply(200). + JSON(map[string]interface{}{ + "jsonrpc": "2.0", + "id": 1, + "result": "0xSTALE", + }) + + gock.New("http://rpc2.localhost"). + Post(""). + Filter(func(r *http.Request) bool { + return strings.Contains(util.SafeReadBody(r), "eth_call") + }). + Times(1). + Reply(200). + JSON(map[string]interface{}{ + "jsonrpc": "2.0", + "id": 1, + "result": "0xNEAR_TIP", + }) + + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + + network := setupLatestStateLagTestNetwork(t, ctx) + network.PinUpstreamOrderForTest("rpc1", "rpc2") + + upsList := network.upstreamsRegistry.GetNetworkUpstreams(ctx, util.EvmNetworkId(123)) + require.Len(t, upsList, 2) + byID := map[string]*upstream.Upstream{} + for _, u := range upsList { + byID[u.Config().Id] = u + } + // TipHW ≈ 1000 via rpc2; rpc1 sits 100 behind (> defaultMaxLatestStateLagBlocks=16). + byID["rpc1"].EvmStatePoller().SuggestLatestBlock(900) + byID["rpc2"].EvmStatePoller().SuggestLatestBlock(1000) + time.Sleep(50 * time.Millisecond) + + req := common.NewNormalizedRequest([]byte( + `{"jsonrpc":"2.0","id":1,"method":"eth_call","params":[{"to":"0x123"},"latest"]}`, + )) + req.SetNetwork(network) + + resp, err := network.Forward(ctx, req) + require.NoError(t, err) + require.NotNil(t, resp) + defer resp.Release() + + require.NotNil(t, resp.Upstream()) + assert.Equal(t, "rpc2", resp.Upstream().Id(), "lagging rpc1 must be skipped for eth_call(latest)") + + jrr, err := resp.JsonRpcResponse() + require.NoError(t, err) + assert.Contains(t, jrr.GetResultString(), "0xNEAR_TIP") +} + +// TestNetwork_EthCallConcreteBlock_AllowsLaggingArchive ensures the lag gate +// only applies to moving latest/pending tags — a concrete block hex on an +// archive peer behind TipHW must still be allowed through. +func TestNetwork_EthCallConcreteBlock_AllowsLaggingArchive(t *testing.T) { + util.ResetGock() + defer util.ResetGock() + util.SetupMocksForEvmStatePoller() + defer util.AssertNoPendingMocks(t, 0) + + gock.New("http://rpc1.localhost"). + Post(""). + Filter(func(r *http.Request) bool { + return strings.Contains(util.SafeReadBody(r), "eth_call") + }). + Times(1). + Reply(200). + JSON(map[string]interface{}{ + "jsonrpc": "2.0", + "id": 1, + "result": "0xARCHIVE", + }) + + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + + network := setupLatestStateLagTestNetwork(t, ctx) + network.PinUpstreamOrderForTest("rpc1", "rpc2") + + upsList := network.upstreamsRegistry.GetNetworkUpstreams(ctx, util.EvmNetworkId(123)) + byID := map[string]*upstream.Upstream{} + for _, u := range upsList { + byID[u.Config().Id] = u + } + byID["rpc1"].EvmStatePoller().SuggestLatestBlock(900) + byID["rpc2"].EvmStatePoller().SuggestLatestBlock(1000) + time.Sleep(50 * time.Millisecond) + + // Concrete historical block well below both poller heads. + req := common.NewNormalizedRequest([]byte( + `{"jsonrpc":"2.0","id":1,"method":"eth_call","params":[{"to":"0x123"},"0x100"]}`, + )) + req.SetNetwork(network) + + resp, err := network.Forward(ctx, req) + require.NoError(t, err) + require.NotNil(t, resp) + defer resp.Release() + + require.NotNil(t, resp.Upstream()) + assert.Equal(t, "rpc1", resp.Upstream().Id(), "concrete-block eth_call must still reach lagging archive peer") + + jrr, err := resp.JsonRpcResponse() + require.NoError(t, err) + assert.Contains(t, jrr.GetResultString(), "0xARCHIVE") +} + +func setupLatestStateLagTestNetwork(t *testing.T, ctx context.Context) *Network { + t.Helper() + + upstreamConfigs := []*common.UpstreamConfig{ + { + Id: "rpc1", + Type: common.UpstreamTypeEvm, + Endpoint: "http://rpc1.localhost", + Evm: &common.EvmUpstreamConfig{ + ChainId: 123, + StatePollerInterval: common.Duration(10 * time.Second), + StatePollerDebounce: common.Duration(50 * time.Millisecond), + }, + }, + { + Id: "rpc2", + Type: common.UpstreamTypeEvm, + Endpoint: "http://rpc2.localhost", + Evm: &common.EvmUpstreamConfig{ + ChainId: 123, + StatePollerInterval: common.Duration(10 * time.Second), + StatePollerDebounce: common.Duration(50 * time.Millisecond), + }, + }, + } + + networkConfig := &common.NetworkConfig{ + Architecture: common.ArchitectureEvm, + Evm: &common.EvmNetworkConfig{ + ChainId: 123, + }, + } + + rateLimitersRegistry, err := upstream.NewRateLimitersRegistry(context.Background(), &common.RateLimiterConfig{}, &log.Logger) + require.NoError(t, err) + + metricsTracker := health.NewTracker(&log.Logger, "test", time.Minute) + + vr := thirdparty.NewVendorsRegistry() + pr, err := thirdparty.NewProvidersRegistry(&log.Logger, vr, []*common.ProviderConfig{}, nil) + require.NoError(t, err) + + ssr, err := data.NewSharedStateRegistry(ctx, &log.Logger, &common.SharedStateConfig{ + Connector: &common.ConnectorConfig{ + Driver: "memory", + Memory: &common.MemoryConnectorConfig{ + MaxItems: 100_000, + MaxTotalSize: "1GB", + }, + }, + }) + require.NoError(t, err) + + upstreamsRegistry := upstream.NewUpstreamsRegistry( + ctx, + &log.Logger, + "test", + upstreamConfigs, + ssr, + rateLimitersRegistry, + vr, + pr, + nil, + metricsTracker, + nil, + ) + + upstreamsRegistry.Bootstrap(ctx) + time.Sleep(100 * time.Millisecond) + + network, err := NewNetwork(ctx, &log.Logger, "test", networkConfig, rateLimitersRegistry, upstreamsRegistry, metricsTracker, nil) + require.NoError(t, err) + + err = upstreamsRegistry.PrepareUpstreamsForNetwork(ctx, networkConfig.NetworkId()) + require.NoError(t, err) + + err = network.Bootstrap(ctx) + require.NoError(t, err) + + return network +} diff --git a/health/tracker.go b/health/tracker.go index a5224699a..6f2c7f499 100644 --- a/health/tracker.go +++ b/health/tracker.go @@ -750,6 +750,17 @@ func (t *Tracker) getMetadata(mtdKey metadataKey) *NetworkMetadata { } // getUpsMetrics fetches or creates *TrackedMetrics from sync.Map +// loadOrStoreUpsMetrics is like getUpsMetrics but does not register the key in +// the upstreamsByNetwork index. Use for buckets (e.g. the {*,All} wildcard +// aggregate) that must not appear as iterable index entries. +func (t *Tracker) loadOrStoreUpsMetrics(k upstreamKey) *TrackedMetrics { + if v, ok := t.upsMetrics.Load(k); ok { + return v.(*TrackedMetrics) + } + v, _ := t.upsMetrics.LoadOrStore(k, newTrackedMetrics(t.logger)) + return v.(*TrackedMetrics) +} + func (t *Tracker) getUpsMetrics(k upstreamKey) *TrackedMetrics { if v, ok := t.upsMetrics.Load(k); ok { return v.(*TrackedMetrics) @@ -1182,6 +1193,7 @@ func (t *Tracker) updateNetworkLagMetrics( } lag := networkValue - upsValue setLag(tm, lag) + setLag(t.loadOrStoreUpsMetrics(upstreamKey{k.ups, "*", common.DataFinalityStateAll}), lag) gauge := getGauge(t.projectId, k.ups.VendorName(), k.ups.NetworkLabel(), k.ups.Id()) gauge.Set(float64(lag)) return true @@ -1205,6 +1217,7 @@ func (t *Tracker) updateNetworkLagMetrics( } lag := networkValue - upsValue setLag(tm, lag) + setLag(t.loadOrStoreUpsMetrics(upstreamKey{k.ups, "*", common.DataFinalityStateAll}), lag) gauge := getGauge(t.projectId, k.ups.VendorName(), k.ups.NetworkLabel(), k.ups.Id()) gauge.Set(float64(lag)) } @@ -1234,6 +1247,7 @@ func (t *Tracker) updateSingleUpstreamLag( if k.ups.Id() == id && k.ups.NetworkId() == net { tm := value.(*TrackedMetrics) setLag(tm, lag) + setLag(t.loadOrStoreUpsMetrics(upstreamKey{k.ups, "*", common.DataFinalityStateAll}), lag) } return true }) @@ -1244,6 +1258,7 @@ func (t *Tracker) updateSingleUpstreamLag( if v, ok := t.upsMetrics.Load(k); ok { tm := v.(*TrackedMetrics) setLag(tm, lag) + setLag(t.loadOrStoreUpsMetrics(upstreamKey{k.ups, "*", common.DataFinalityStateAll}), lag) } } } @@ -1340,6 +1355,8 @@ func (t *Tracker) SetLatestBlockNumber(upstream common.Upstream, blockNumber int func(tm *TrackedMetrics, lag int64) { tm.BlockHeadLag.Store(lag) }, ) } + + t.loadOrStoreUpsMetrics(upstreamKey{upstream, "*", common.DataFinalityStateAll}).BlockHeadLag.Store(upsLag) } func (t *Tracker) SetLatestBlockNumberForNetwork(network string, blockNumber int64) { @@ -1513,6 +1530,8 @@ func (t *Tracker) SetFinalizedBlockNumber(upstream common.Upstream, blockNumber func(tm *TrackedMetrics, lag int64) { tm.FinalizationLag.Store(lag) }, ) } + + t.loadOrStoreUpsMetrics(upstreamKey{upstream, "*", common.DataFinalityStateAll}).FinalizationLag.Store(upsLag) } func (t *Tracker) RecordBlockHeadLargeRollback(upstream common.Upstream, finality string, currentVal, newVal int64) { diff --git a/health/tracker_test.go b/health/tracker_test.go index 7c717827a..51e040705 100644 --- a/health/tracker_test.go +++ b/health/tracker_test.go @@ -394,6 +394,43 @@ func TestBlockHeadLagPersistsAcrossResets(t *testing.T) { assert.Equal(t, int64(0), metrics2Updated.BlockHeadLag.Load(), "upstream2 should now be caught up") } + +// TestWildcardLagMirroredForPeerUpstreams is the regression test for the +// incident where blockNumberLagAbove silently never fired for CB-open upstreams. +// +// The scenario: ups2 is lagging far behind ups1. ups2's own poller never fires +// (simulating a CB-open upstream). Only ups1 calls SetLatestBlockNumber. The fix +// ensures updateNetworkLagMetrics mirrors the computed lag onto {ups2,"*",All} so +// evalScope:network policies read the correct non-zero value. +func TestWildcardLagMirroredForPeerUpstreams(t *testing.T) { + tracker := NewTracker(&log.Logger, "test-project", 5*time.Minute) + tracker.Bootstrap(context.Background()) + + ups1 := common.NewFakeUpstream("upstream1") + ups2 := common.NewFakeUpstream("upstream2") + + // Register both upstreams with traffic so their per-method buckets exist. + tracker.RecordUpstreamRequest(ups1, "eth_blockNumber", common.DataFinalityStateUnknown) + tracker.RecordUpstreamRequest(ups2, "eth_blockNumber", common.DataFinalityStateUnknown) + + // ups2 reports a stale block (simulating its last report before going CB-open). + // After this, ups2's own poller never fires again. + tracker.SetLatestBlockNumber(ups2, 900, 0) + + // ups1 advances the tip to 1000. updateNetworkLagMetrics now computes + // ups2's lag as 100 and must mirror it onto {ups2,"*",All} — the peer-write + // path that was previously missing. + tracker.SetLatestBlockNumber(ups1, 1000, 0) + + // evalScope:network reads {ups, "*", All}.BlockHeadLag + ups2WildcardLag := tracker.GetUpstreamMethodMetrics(ups2, "*", common.DataFinalityStateAll).BlockHeadLag.Load() + require.EqualValues(t, 100, ups2WildcardLag, + "peer upstream lag must reach {ups,*,All} via updateNetworkLagMetrics even without its own poller firing") + + ups1WildcardLag := tracker.GetUpstreamMethodMetrics(ups1, "*", common.DataFinalityStateAll).BlockHeadLag.Load() + require.EqualValues(t, 0, ups1WildcardLag) +} + func TestFinalizationLagPersistsAcrossResets(t *testing.T) { projectID := "test-project" windowSize := 100 * time.Millisecond // Short window for faster testing From 0c96ef16c0d7f8f033736cb9778f569cb5ce9620 Mon Sep 17 00:00:00 2001 From: 1marcghannam <1marc.ghannam@gmail.com> Date: Tue, 15 Sep 2026 11:45:23 +0200 Subject: [PATCH 38/40] test(evm): encode Priority Pool totalQueued incident in eth_call lag skip Use the stalled-vs-tip totalQueued values (16355 vs 692), ~8k block lag, and the real selector so the regression mirrors the production failure mode. --- erpc/networks_latest_state_lag_test.go | 47 ++++++++++++++++++-------- 1 file changed, 33 insertions(+), 14 deletions(-) diff --git a/erpc/networks_latest_state_lag_test.go b/erpc/networks_latest_state_lag_test.go index 113a41a19..8901b0b60 100644 --- a/erpc/networks_latest_state_lag_test.go +++ b/erpc/networks_latest_state_lag_test.go @@ -23,18 +23,31 @@ func init() { util.ConfigureTestLogger() } -// TestNetwork_EthCallLatest_SkipsLaggingUpstream covers the Priority Pool -// totalQueued incident: a stalled WS peer answered eth_call("latest") with -// stale state while TipHW was thousands of blocks ahead. The lag hard-gate -// must skip the lagging peer and serve from a near-tip sibling. +// TestNetwork_EthCallLatest_SkipsLaggingUpstream is the regression for the +// Priority Pool totalQueued incident: +// +// contract-events-slack → eRPC eth_call("latest") totalQueued() +// selected stalled internal-eth-mainnet-reth-ws-0 (~8k behind TipHW) +// → returned 16355 instead of ~692 +// +// Setup mirrors that failure mode: selection order prefers the lagging peer +// (rpc1), which would answer successfully with the stale ABI uint256, while a +// near-tip sibling (rpc2) has the correct tip-state value. The lag hard-gate +// must skip rpc1 and return rpc2's result. func TestNetwork_EthCallLatest_SkipsLaggingUpstream(t *testing.T) { util.ResetGock() defer util.ResetGock() util.SetupMocksForEvmStatePoller() defer util.AssertNoPendingMocks(t, 0) - // rpc1 must NOT receive eth_call — it will be tip-lagged. Times(0) - // ensures AssertNoPendingMocks fails if the gate fails open. + // ABI-encoded uint256 totalQueued values from the incident. + const ( + staleTotalQueued = "0x0000000000000000000000000000000000000000000000000000000000003fe3" // 16355 + tipTotalQueued = "0x00000000000000000000000000000000000000000000000000000000000002b4" // 692 + ) + + // rpc1 must NOT receive eth_call — it is tip-lagged like reth-ws-0. + // Times(0) fails the test if the gate fails open and routes here. gock.New("http://rpc1.localhost"). Post(""). Filter(func(r *http.Request) bool { @@ -45,7 +58,7 @@ func TestNetwork_EthCallLatest_SkipsLaggingUpstream(t *testing.T) { JSON(map[string]interface{}{ "jsonrpc": "2.0", "id": 1, - "result": "0xSTALE", + "result": staleTotalQueued, }) gock.New("http://rpc2.localhost"). @@ -58,13 +71,14 @@ func TestNetwork_EthCallLatest_SkipsLaggingUpstream(t *testing.T) { JSON(map[string]interface{}{ "jsonrpc": "2.0", "id": 1, - "result": "0xNEAR_TIP", + "result": tipTotalQueued, }) ctx, cancel := context.WithCancel(context.Background()) defer cancel() network := setupLatestStateLagTestNetwork(t, ctx) + // Prefer the stalled peer first — same trap as selection picking reth-ws-0. network.PinUpstreamOrderForTest("rpc1", "rpc2") upsList := network.upstreamsRegistry.GetNetworkUpstreams(ctx, util.EvmNetworkId(123)) @@ -73,13 +87,15 @@ func TestNetwork_EthCallLatest_SkipsLaggingUpstream(t *testing.T) { for _, u := range upsList { byID[u.Config().Id] = u } - // TipHW ≈ 1000 via rpc2; rpc1 sits 100 behind (> defaultMaxLatestStateLagBlocks=16). - byID["rpc1"].EvmStatePoller().SuggestLatestBlock(900) - byID["rpc2"].EvmStatePoller().SuggestLatestBlock(1000) + // ~8k blocks behind TipHW (incident scale); well above max lag of 16. + const tipHW int64 = 21_000_000 + byID["rpc1"].EvmStatePoller().SuggestLatestBlock(tipHW - 8000) + byID["rpc2"].EvmStatePoller().SuggestLatestBlock(tipHW) time.Sleep(50 * time.Millisecond) + // Priority Pool totalQueued() selector (0xa4baa10c) at "latest". req := common.NewNormalizedRequest([]byte( - `{"jsonrpc":"2.0","id":1,"method":"eth_call","params":[{"to":"0x123"},"latest"]}`, + `{"jsonrpc":"2.0","id":1,"method":"eth_call","params":[{"to":"0x362fa9d0bca5d19f743db50738345ce2b40ec99f","data":"0xa4baa10c"},"latest"]}`, )) req.SetNetwork(network) @@ -89,11 +105,14 @@ func TestNetwork_EthCallLatest_SkipsLaggingUpstream(t *testing.T) { defer resp.Release() require.NotNil(t, resp.Upstream()) - assert.Equal(t, "rpc2", resp.Upstream().Id(), "lagging rpc1 must be skipped for eth_call(latest)") + assert.Equal(t, "rpc2", resp.Upstream().Id(), + "lagging rpc1 (reth-ws-0 stand-in) must be skipped for eth_call(latest)") jrr, err := resp.JsonRpcResponse() require.NoError(t, err) - assert.Contains(t, jrr.GetResultString(), "0xNEAR_TIP") + got := jrr.GetResultString() + assert.Contains(t, got, tipTotalQueued, "must return tip-state totalQueued (~692), not stale 16355") + assert.NotContains(t, got, staleTotalQueued, "must not return stalled-peer totalQueued 16355") } // TestNetwork_EthCallConcreteBlock_AllowsLaggingArchive ensures the lag gate From 461638103d9a5ee2258466e214fbc137d703555c Mon Sep 17 00:00:00 2001 From: 1marcghannam <1marc.ghannam@gmail.com> Date: Tue, 15 Sep 2026 13:26:06 +0200 Subject: [PATCH 39/40] feat(evm): make latest-state lag gate configurable via evm.maxLatestStateLagBlocks The 16-block hard gate mirrored the default selection policy's blockNumberLagAbove(16) but was not operator-tunable, unlike the policy threshold. Add evm.maxLatestStateLagBlocks (nil -> 16, <=0 disables), resolved at the read site like MaxRetryableBlockDistance. Also correct the lag-test comments: lag comes from the standard poller mocks (0x11118888 vs 0x22228888), not the no-op SuggestLatestBlock calls. --- common/config.go | 12 +++ erpc/networks.go | 23 ++++- erpc/networks_latest_state_lag_test.go | 126 ++++++++++++++++++++----- 3 files changed, 136 insertions(+), 25 deletions(-) diff --git a/common/config.go b/common/config.go index f392e488d..03190e6b8 100644 --- a/common/config.go +++ b/common/config.go @@ -2220,6 +2220,18 @@ type EvmNetworkConfig struct { // Default: 128 blocks. MaxRetryableBlockDistance *int64 `yaml:"maxRetryableBlockDistance,omitempty" json:"maxRetryableBlockDistance,omitempty"` + // MaxLatestStateLagBlocks controls the hard gate that skips upstreams whose + // head lags the network's highest known latest block (TipHW) when serving + // latest-state read methods at a moving tag ("latest"/"pending", or an + // omitted block arg — e.g. eth_call, eth_getBalance). Upstreams lagging by + // more than this many blocks are skipped so a stalled node never serves + // stale state as a successful response. Set to 0 or a negative value to + // disable the gate. On fast chains (sub-second blocks) consider raising + // this so state-poller cadence can keep every upstream within the window. + // Default: 16 blocks — matches the default selection policy's + // blockNumberLagAbove(16) exclusion. + MaxLatestStateLagBlocks *int64 `yaml:"maxLatestStateLagBlocks,omitempty" json:"maxLatestStateLagBlocks,omitempty"` + // MarkEmptyAsErrorMethods lists methods for which an empty/null result from an upstream // should be treated as a "missing data" error, triggering retry on other upstreams. // This is useful for point-lookups (blocks, transactions, receipts, traces) where an diff --git a/erpc/networks.go b/erpc/networks.go index 315f5112a..1847fd517 100644 --- a/erpc/networks.go +++ b/erpc/networks.go @@ -1701,9 +1701,21 @@ func (n *Network) recordHedgeDiscard( // defaultMaxLatestStateLagBlocks matches the common selection-policy threshold // blockNumberLagAbove(16): peers farther behind TipHW than this must not serve -// moving-tag latest-state reads (eth_call("latest"), etc.). +// moving-tag latest-state reads (eth_call("latest"), etc.). Override per +// network via evm.maxLatestStateLagBlocks (≤0 disables the gate). const defaultMaxLatestStateLagBlocks int64 = 16 +// maxLatestStateLagBlocks resolves the effective lag threshold for the +// latest-state hard gate: the network's evm.maxLatestStateLagBlocks when set, +// otherwise defaultMaxLatestStateLagBlocks. Follows the same read-site default +// pattern as MaxRetryableBlockDistance. +func (n *Network) maxLatestStateLagBlocks() int64 { + if n.cfg != nil && n.cfg.Evm != nil && n.cfg.Evm.MaxLatestStateLagBlocks != nil { + return *n.cfg.Evm.MaxLatestStateLagBlocks + } + return defaultMaxLatestStateLagBlocks +} + // isLatestStateReadMethod reports whether method returns chain state at a // caller-chosen block tag. Stale answers from a lagging upstream look like // success for these methods, so they need a TipHW lag hard-gate. @@ -1745,6 +1757,11 @@ func isMovingLatestOrPendingTag(ctx context.Context, req *common.NormalizedReque // fail-open — which is exactly how a stalled WS peer can answer eth_call with // stale state while TipHW is thousands of blocks ahead. func (n *Network) checkLatestStateLagSkip(ctx context.Context, u common.Upstream, req *common.NormalizedRequest, method string) (error, bool) { + maxLag := n.maxLatestStateLagBlocks() + if maxLag <= 0 { + // Gate disabled via evm.maxLatestStateLagBlocks. + return nil, false + } if !isLatestStateReadMethod(method) || !isMovingLatestOrPendingTag(ctx, req) { return nil, false } @@ -1762,7 +1779,7 @@ func (n *Network) checkLatestStateLagSkip(ctx context.Context, u common.Upstream return nil, false } lag := tip - upsLatest - if lag <= defaultMaxLatestStateLagBlocks { + if lag <= maxLag { return nil, false } finalized := sp.FinalizedBlock() @@ -1772,7 +1789,7 @@ func (n *Network) checkLatestStateLagSkip(ctx context.Context, u common.Upstream Int64("networkTip", tip). Int64("pollerLatest", upsLatest). Int64("lag", lag). - Int64("maxLag", defaultMaxLatestStateLagBlocks). + Int64("maxLag", maxLag). Msg("skipping lagging upstream for latest-state read") // Retryable: another near-tip peer may still serve; treat like block-unavailable. return common.NewErrUpstreamBlockUnavailable(u.Id(), tip, upsLatest, finalized), true diff --git a/erpc/networks_latest_state_lag_test.go b/erpc/networks_latest_state_lag_test.go index 8901b0b60..7ef424d78 100644 --- a/erpc/networks_latest_state_lag_test.go +++ b/erpc/networks_latest_state_lag_test.go @@ -81,17 +81,9 @@ func TestNetwork_EthCallLatest_SkipsLaggingUpstream(t *testing.T) { // Prefer the stalled peer first — same trap as selection picking reth-ws-0. network.PinUpstreamOrderForTest("rpc1", "rpc2") - upsList := network.upstreamsRegistry.GetNetworkUpstreams(ctx, util.EvmNetworkId(123)) - require.Len(t, upsList, 2) - byID := map[string]*upstream.Upstream{} - for _, u := range upsList { - byID[u.Config().Id] = u - } - // ~8k blocks behind TipHW (incident scale); well above max lag of 16. - const tipHW int64 = 21_000_000 - byID["rpc1"].EvmStatePoller().SuggestLatestBlock(tipHW - 8000) - byID["rpc2"].EvmStatePoller().SuggestLatestBlock(tipHW) - time.Sleep(50 * time.Millisecond) + // Lag comes from the standard poller mocks: rpc1's head is 0x11118888 + // while rpc2 (TipHW) is at 0x22228888 — rpc1 lags by 286,326,784 blocks, + // far beyond defaultMaxLatestStateLagBlocks=16 (incident: ~8k behind). // Priority Pool totalQueued() selector (0xa4baa10c) at "latest". req := common.NewNormalizedRequest([]byte( @@ -143,16 +135,8 @@ func TestNetwork_EthCallConcreteBlock_AllowsLaggingArchive(t *testing.T) { network := setupLatestStateLagTestNetwork(t, ctx) network.PinUpstreamOrderForTest("rpc1", "rpc2") - upsList := network.upstreamsRegistry.GetNetworkUpstreams(ctx, util.EvmNetworkId(123)) - byID := map[string]*upstream.Upstream{} - for _, u := range upsList { - byID[u.Config().Id] = u - } - byID["rpc1"].EvmStatePoller().SuggestLatestBlock(900) - byID["rpc2"].EvmStatePoller().SuggestLatestBlock(1000) - time.Sleep(50 * time.Millisecond) - - // Concrete historical block well below both poller heads. + // rpc1 lags TipHW by ~286M blocks per the standard poller mocks; a + // concrete historical block must still be servable by the lagging peer. req := common.NewNormalizedRequest([]byte( `{"jsonrpc":"2.0","id":1,"method":"eth_call","params":[{"to":"0x123"},"0x100"]}`, )) @@ -171,7 +155,102 @@ func TestNetwork_EthCallConcreteBlock_AllowsLaggingArchive(t *testing.T) { assert.Contains(t, jrr.GetResultString(), "0xARCHIVE") } -func setupLatestStateLagTestNetwork(t *testing.T, ctx context.Context) *Network { +// TestNetwork_EthCallLatest_ConfigurableLagThreshold verifies that +// evm.maxLatestStateLagBlocks overrides the default 16-block hard gate: +// with the threshold raised above the actual lag, the lagging (but +// selection-preferred) peer is allowed to serve again. +func TestNetwork_EthCallLatest_ConfigurableLagThreshold(t *testing.T) { + util.ResetGock() + defer util.ResetGock() + util.SetupMocksForEvmStatePoller() + defer util.AssertNoPendingMocks(t, 0) + + gock.New("http://rpc1.localhost"). + Post(""). + Filter(func(r *http.Request) bool { + return strings.Contains(util.SafeReadBody(r), "eth_call") + }). + Times(1). + Reply(200). + JSON(map[string]interface{}{ + "jsonrpc": "2.0", + "id": 1, + "result": "0xLAGGED_OK", + }) + + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + + // The standard poller mocks pin rpc1 at 0x11118888 and rpc2 (TipHW) at + // 0x22228888 — rpc1 lags by 0x11110000 = 286,326,784 blocks. A threshold + // above that lag must let rpc1 (selection-preferred) serve again. + network := setupLatestStateLagTestNetwork(t, ctx, func(cfg *common.NetworkConfig) { + cfg.Evm.MaxLatestStateLagBlocks = i64(300_000_000) + }) + network.PinUpstreamOrderForTest("rpc1", "rpc2") + + req := common.NewNormalizedRequest([]byte( + `{"jsonrpc":"2.0","id":1,"method":"eth_call","params":[{"to":"0x123"},"latest"]}`, + )) + req.SetNetwork(network) + + resp, err := network.Forward(ctx, req) + require.NoError(t, err) + require.NotNil(t, resp) + defer resp.Release() + + require.NotNil(t, resp.Upstream()) + assert.Equal(t, "rpc1", resp.Upstream().Id(), + "lag 100 within configured threshold 200 must not be gated") +} + +// TestNetwork_EthCallLatest_LagGateDisabled verifies that a non-positive +// evm.maxLatestStateLagBlocks disables the hard gate entirely. +func TestNetwork_EthCallLatest_LagGateDisabled(t *testing.T) { + util.ResetGock() + defer util.ResetGock() + util.SetupMocksForEvmStatePoller() + defer util.AssertNoPendingMocks(t, 0) + + gock.New("http://rpc1.localhost"). + Post(""). + Filter(func(r *http.Request) bool { + return strings.Contains(util.SafeReadBody(r), "eth_call") + }). + Times(1). + Reply(200). + JSON(map[string]interface{}{ + "jsonrpc": "2.0", + "id": 1, + "result": "0xGATE_OFF", + }) + + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + + // rpc1 lags TipHW by ~286M blocks per the standard poller mocks; with the + // gate disabled it must still serve (pre-fix fail-open behavior). + network := setupLatestStateLagTestNetwork(t, ctx, func(cfg *common.NetworkConfig) { + cfg.Evm.MaxLatestStateLagBlocks = i64(-1) + }) + network.PinUpstreamOrderForTest("rpc1", "rpc2") + + req := common.NewNormalizedRequest([]byte( + `{"jsonrpc":"2.0","id":1,"method":"eth_call","params":[{"to":"0x123"},"latest"]}`, + )) + req.SetNetwork(network) + + resp, err := network.Forward(ctx, req) + require.NoError(t, err) + require.NotNil(t, resp) + defer resp.Release() + + require.NotNil(t, resp.Upstream()) + assert.Equal(t, "rpc1", resp.Upstream().Id(), + "disabled gate (≤0) must restore pre-fix fail-open routing") +} + +func setupLatestStateLagTestNetwork(t *testing.T, ctx context.Context, cfgMut ...func(*common.NetworkConfig)) *Network { t.Helper() upstreamConfigs := []*common.UpstreamConfig{ @@ -203,6 +282,9 @@ func setupLatestStateLagTestNetwork(t *testing.T, ctx context.Context) *Network ChainId: 123, }, } + for _, mut := range cfgMut { + mut(networkConfig) + } rateLimitersRegistry, err := upstream.NewRateLimitersRegistry(context.Background(), &common.RateLimiterConfig{}, &log.Logger) require.NoError(t, err) From 5e8fc97a799e89b3d29e9dac5657e911da99a535 Mon Sep 17 00:00:00 2001 From: snowkide Date: Thu, 17 Sep 2026 12:57:27 +0200 Subject: [PATCH 40/40] chore(ci): pin pnpm 9.15.9 for alpine musl GHCR builds @pnpm/exe@10.28.2 (from packageManager) has no linux-x64-musl binary, which breaks Publish to GHCR on node:alpine. Pin npm-global pnpm and disable package-manager version switching for install steps. Co-authored-by: Cursor --- Dockerfile | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/Dockerfile b/Dockerfile index 23163193f..c8d7751fa 100644 --- a/Dockerfile +++ b/Dockerfile @@ -35,12 +35,12 @@ RUN go build -v -ldflags="$LDFLAGS" -a -installsuffix cgo -o erpc-server ./cmd/e # Global typescript related image FROM node:20-alpine@sha256:09e2b3d9726018aecf269bd35325f46bf75046a643a66d28360ec71132750ec8 AS ts-core -RUN npm install -g pnpm +RUN npm install -g pnpm@9.15.9 # Stage where we will install dev dependencies + compile sdk FROM ts-core AS ts-dev RUN mkdir -p /temp/dev/typescript -RUN npm install -g pnpm +RUN npm install -g pnpm@9.15.9 # Copy only the TypeScript package files COPY typescript/config /temp/dev/typescript/config @@ -48,7 +48,7 @@ COPY pnpm* /temp/dev/ COPY package.json /temp/dev/package.json # Install everything and build -RUN --mount=type=cache,id=pnpm,target=/pnpm/store cd /temp/dev && pnpm install --frozen-lockfile +RUN --mount=type=cache,id=pnpm,target=/pnpm/store cd /temp/dev && pnpm install --frozen-lockfile --config.manage-package-manager-versions=false RUN cd /temp/dev && pnpm build # Stage where we will install prod dependencies only @@ -60,7 +60,7 @@ COPY pnpm* /temp/prod/ COPY package.json /temp/prod/package.json # Install every prod dependencies -RUN --mount=type=cache,id=pnpm,target=/pnpm/store cd /temp/prod && pnpm install --prod --frozen-lockfile +RUN --mount=type=cache,id=pnpm,target=/pnpm/store cd /temp/prod && pnpm install --prod --frozen-lockfile --config.manage-package-manager-versions=false # Create symlink stage (for backwards compatibility with earlier image file structure) FROM alpine:latest@sha256:25109184c71bdad752c8312a8623239686a9a2071e8825f20acb8f2198c3f659 AS symlink