Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
29 changes: 29 additions & 0 deletions server/cmd/api/main.go
Original file line number Diff line number Diff line change
Expand Up @@ -35,6 +35,7 @@ import (
"github.com/kernel/kernel-images/server/lib/metrics"
"github.com/kernel/kernel-images/server/lib/nekoclient"
oapi "github.com/kernel/kernel-images/server/lib/oapi"
"github.com/kernel/kernel-images/server/lib/pagerecovery"
"github.com/kernel/kernel-images/server/lib/recorder"
"github.com/kernel/kernel-images/server/lib/scaletozero"
"github.com/kernel/kernel-images/server/lib/sysmon"
Expand Down Expand Up @@ -210,6 +211,21 @@ func main() {
os.Exit(1)
}

// Navigation retry runs on its own CDP connection so it is unaffected by
// whether customer telemetry is capturing. It stays nil when off, and the
// metrics below then report zeros and a down gauge rather than disappearing.
var recoverer *pagerecovery.Recoverer
if config.PageRecoveryEnabled {
recoverer = pagerecovery.New(upstreamMgr, pagerecovery.Config{
MaxAttempts: config.PageRecoveryMaxAttempts,
Budget: config.PageRecoveryBudget,
}, slogger)
if err := recoverer.Start(context.Background()); err != nil {
slogger.Error("failed to start page recovery", "err", err)
os.Exit(1)
}
}

// api_call event emission. Off until the telemetry handlers flip it on.
r.Use(api.TelemetryHTTPMiddleware(telemetrySession.Publish))
r.Use(api.WebMCPRequestSizeMiddleware)
Expand Down Expand Up @@ -339,6 +355,13 @@ func main() {
rMetrics.Use(chiMiddleware.Recoverer)
metricsCollectors := []metrics.Collector{
metrics.NewNetworkCollector(apiService.NetworkMetrics),
metrics.NewPageRecoveryCollector(func() (retries, recovered, exhausted uint64, up bool) {
if recoverer == nil {
return 0, 0, 0, false
}
snapshot := recoverer.SnapshotMetrics()
return snapshot.Retries, snapshot.Recovered, snapshot.Exhausted, snapshot.Up
}),
metrics.NewChromeCollector(upstreamMgr),
metrics.NewGPUCollector(),
metrics.NewSystemCollector(),
Expand Down Expand Up @@ -398,6 +421,12 @@ func main() {
g.Go(func() error {
return apiService.Shutdown(shutdownCtx)
})
g.Go(func() error {
if recoverer != nil {
recoverer.Stop()
}
return nil
})
g.Go(func() error {
if n := wsRegistry.CloseAll(websocket.StatusGoingAway, "browser shutting down"); n > 0 {
slogger.Info("closed active websocket connections for shutdown", "count", n)
Expand Down
7 changes: 7 additions & 0 deletions server/cmd/config/config.go
Original file line number Diff line number Diff line change
Expand Up @@ -42,6 +42,13 @@ type Config struct {
// How long to wait after the last active request before re-enabling scale-to-zero.
ScaleToZeroCooldown time.Duration `envconfig:"SCALE_TO_ZERO_COOLDOWN" default:"1s"`

// Navigation retry: replay a top-level document that the site refused, so
// the caller driving the browser does not have to. Off by default; the
// budget bounds how much latency one navigation may spend retrying.
PageRecoveryEnabled bool `envconfig:"PAGE_RECOVERY_ENABLED" default:"false"`
PageRecoveryMaxAttempts int `envconfig:"PAGE_RECOVERY_MAX_ATTEMPTS" default:"2"`
PageRecoveryBudget time.Duration `envconfig:"PAGE_RECOVERY_BUDGET" default:"8s"`

// ChromeDriver proxy: external port where the proxy listens.
ChromeDriverProxyPort int `envconfig:"CHROMEDRIVER_PROXY_PORT" default:"9224"`
// Internal ChromeDriver upstream used by the ChromeDriver proxy.
Expand Down
6 changes: 6 additions & 0 deletions server/cmd/config/config_test.go
Original file line number Diff line number Diff line change
Expand Up @@ -29,6 +29,8 @@ func TestLoad(t *testing.T) {
PathToFFmpeg: "ffmpeg",
DevToolsProxyPort: 9222,
ScaleToZeroCooldown: time.Second,
PageRecoveryMaxAttempts: 2,
PageRecoveryBudget: 8 * time.Second,
ChromeDriverProxyPort: 9224,
ChromeDriverUpstreamAddr: "127.0.0.1:9225",
DevToolsProxyAddr: "127.0.0.1:9222",
Expand Down Expand Up @@ -68,6 +70,8 @@ func TestLoad(t *testing.T) {
PathToFFmpeg: "/usr/local/bin/ffmpeg",
DevToolsProxyPort: 9876,
ScaleToZeroCooldown: 5 * time.Second,
PageRecoveryMaxAttempts: 2,
PageRecoveryBudget: 8 * time.Second,
ChromeDriverProxyPort: 5432,
ChromeDriverUpstreamAddr: "127.0.0.1:9999",
DevToolsProxyAddr: "127.0.0.1:9876",
Expand Down Expand Up @@ -96,6 +100,8 @@ func TestLoad(t *testing.T) {
PathToFFmpeg: "ffmpeg",
DevToolsProxyPort: 7777,
ScaleToZeroCooldown: time.Second,
PageRecoveryMaxAttempts: 2,
PageRecoveryBudget: 8 * time.Second,
ChromeDriverProxyPort: 9224,
ChromeDriverUpstreamAddr: "127.0.0.1:9225",
DevToolsProxyAddr: "10.0.0.1:1234",
Expand Down
37 changes: 37 additions & 0 deletions server/lib/metrics/pagerecovery.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,37 @@
package metrics

import "context"

// PageRecoveryCollector exposes the in-memory navigation-retry counters. Like
// the network collector it never dials Chrome, so it keeps reporting while the
// recoverer's own connection is down — which is exactly when the gauge matters.
type PageRecoveryCollector struct {
snapshot func() (retries, recovered, exhausted uint64, up bool)
}

func NewPageRecoveryCollector(snapshot func() (retries, recovered, exhausted uint64, up bool)) *PageRecoveryCollector {
return &PageRecoveryCollector{snapshot: snapshot}
}

func (*PageRecoveryCollector) Name() string { return "page_recovery" }

func (c *PageRecoveryCollector) Collect(_ context.Context, w *Writer) error {
retries, recovered, exhausted, healthy := c.snapshot()
const retriesName = "kernel_page_recovery_retries_total"
const recoveredName = "kernel_page_recovery_recovered_total"
const exhaustedName = "kernel_page_recovery_exhausted_total"
const upName = "kernel_page_recovery_up"
w.Metric(retriesName, "Top-level navigations replayed after a refused response.", "counter")
w.Sample(retriesName, nil, float64(retries))
w.Metric(recoveredName, "Replayed navigations that went on to answer below 400.", "counter")
w.Sample(recoveredName, nil, float64(recovered))
w.Metric(exhaustedName, "Refusals passed through with the retry budget spent.", "counter")
w.Sample(exhaustedName, nil, float64(exhausted))
w.Metric(upName, "Whether navigation-retry interception is installed.", "gauge")
value := float64(0)
if healthy {
value = 1
}
w.Sample(upName, nil, value)
return nil
}
108 changes: 108 additions & 0 deletions server/lib/pagerecovery/README.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,108 @@
# Page recovery

Replays a top-level navigation that the site refused, so the caller driving the
browser does not have to.

A client asks for a page, the site answers `429 Too Many Requests`, and the
client is left holding a block page. A person in that position reloads. An agent
usually does not: it reads the document it was given, concludes the site is
unavailable, and says so. The work this package does is that reload, early
enough that the caller never sees the refusal.

Off by default. `PAGE_RECOVERY_ENABLED=true` turns it on for a browser.

## What it acts on

One `Fetch.requestPaused` notification, at the response stage, for a main-frame
document. Everything the decision needs is in the response:

| Condition | Replayed |
| --- | --- |
| `408`, `429`, `502`, `503`, `504`, `507` | yes |
| `ConnectionReset`, `ConnectionClosed`, `ConnectionFailed`, `ConnectionAborted`, `TimedOut` | yes |
| `403` | no |
| `502` carrying `X-Kernel-Proxy-Error` | no |
| Any status on a subresource, an iframe document, or a non-`GET` navigation | no |

`403` is absent because it is as often a settled answer about the session as a
throttle, and replaying a settled answer adds latency and requests without
changing anything. A branded `502` is Kernel's own egress failing rather than
the site refusing; `cdpmonitor` already reports those as typed `proxy_error`
events, and retrying one would spend the budget hiding a Kernel-side signal.
Non-`GET` is excluded because the replay preserves method and body: a POST a
gateway refused may still have been recorded upstream, and a duplicate order is
a worse outcome than a visible refusal.

## How the replay stays invisible

The refusal is answered with a `307` back to the same URL rather than being let
through. Chromium treats that as one more hop in the navigation that is already
in flight, so the client's `Page.navigate` — a Playwright `goto`, a Puppeteer
`goto` — resolves once, on the page it asked for, having waited out the retries.
It is not a second navigation, so nothing the client is waiting on is
interrupted.

Cookies the refusal set are still applied. The network stack processes
`Set-Cookie` before the request is paused, so a clearance cookie handed out by a
block page is present on the replay, which is the mechanism that makes a manual
reload work in the first place.

A fulfillment must carry a body, even an empty one. Chromium treats a
fulfillment without one as no fulfillment at all and lets the original response
through.

## Budget

Retries are bounded by attempts (`PAGE_RECOVERY_MAX_ATTEMPTS`, default 2) and by
wall clock (`PAGE_RECOVERY_BUDGET`, default 8s), per CDP session and URL. The
budget is sized against what the caller is waiting on: a `goto` typically
carries a 30s timeout with the real page load still to come. A block that needs
longer than the budget is a block the caller should see, so the refusal goes
through rather than turning into a navigation that looks hung.

`Retry-After` is honoured when the site sends one and it fits in what the budget
has left; otherwise the wait is exponential with full jitter from 300 ms. The
jitter matters more than the growth — many browser VMs retrying one throttled
origin in lockstep is how a short block becomes a long one.

A URL's budget is released once it answers with something that was not retried,
so the next navigation to it is judged on its own.

## Boundaries

It runs on its own CDP connection and browser-surface tracker, sharing no state
with `cdpmonitor` or WebMCP: interception has to run whether or not customer
telemetry is capturing, and a connection of its own is what keeps the two
lifecycles from having to agree. A client's own `Fetch` interception — what
Playwright's `page.route` installs — coexists with it; both see every request.

It does not cover a block that answers `200` and puts an interstitial in the
document, where the evidence is the rendered page rather than the status line.
Recognising those is the anti-bot extension's reading, and acting on one means a
real reload after the document has run, not a redirect before it. This package
deliberately handles only the half that can be decided from the response itself.

## Metrics

Served label-free on the existing `GET /metrics`:

| Metric | Meaning |
| --- | --- |
| `kernel_page_recovery_retries_total` | Navigations replayed after a refused response. |
| `kernel_page_recovery_recovered_total` | Replayed navigations that went on to answer below 400. |
| `kernel_page_recovery_exhausted_total` | Refusals passed through with the budget spent. |
| `kernel_page_recovery_up` | Whether interception is installed. Zero while off, during setup, and across reconnects. |

## Tests

```sh
go test -race ./lib/pagerecovery
KERNEL_PAGERECOVERY_CHROME_E2E=1 go test ./lib/pagerecovery -count=1 -v
```

The real-Chromium suite uses a local origin that refuses a fixed number of
requests per path. It checks the transparent case, an exhausted budget landing
the caller on the site's own answer, an unrefused navigation going untouched, a
`403` left alone, a client interceptor still seeing every request, and six tabs
refused at once each recovering independently. It does not cover image restart,
snapshot fork, or a site that blocks with a rendered page.
36 changes: 36 additions & 0 deletions server/lib/pagerecovery/cdp_proto.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,36 @@
package pagerecovery

// CDP payloads this package reads, holding only the fields it acts on. The
// wider PDL-faithful types live in cdpmonitor, which reports on every field;
// here an unread field would be a field nothing tests.

type cdpHeaderEntry struct {
Name string `json:"name"`
Value string `json:"value"`
}

type cdpFetchRequest struct {
URL string `json:"url"`
Method string `json:"method"`
}

// cdpFetchRequestPaused mirrors the response-stage form of the notification.
// ResponseStatusCode and ResponseErrorReason are mutually exclusive: a request
// that died has no status, and one that answered has no error reason.
type cdpFetchRequestPaused struct {
RequestID string `json:"requestId"`
Request cdpFetchRequest `json:"request"`
FrameID string `json:"frameId"`
ResourceType string `json:"resourceType"`
ResponseStatusCode int `json:"responseStatusCode"`
ResponseErrorReason string `json:"responseErrorReason"`
ResponseHeaders []cdpHeaderEntry `json:"responseHeaders"`
}

type cdpPageGetFrameTreeResult struct {
FrameTree struct {
Frame struct {
ID string `json:"id"`
} `json:"frame"`
} `json:"frameTree"`
}
Loading
Loading