diff --git a/cmd/silo/abs_listener.go b/cmd/silo/abs_listener.go index 3f10fdee33..33c7ae19c0 100644 --- a/cmd/silo/abs_listener.go +++ b/cmd/silo/abs_listener.go @@ -10,6 +10,7 @@ import ( "github.com/Silo-Server/silo-server/internal/audiobooks/abs" "github.com/Silo-Server/silo-server/internal/clientip" "github.com/Silo-Server/silo-server/internal/httpstream" + "github.com/Silo-Server/silo-server/internal/netaccess" ) // absMounter is the narrow interface the listener needs from the @@ -36,11 +37,14 @@ type absMounter interface { // artwork storage the client follows them on this port; without the route // here every locally stored cover would 404. Mirrors the Jellyfin listener. // A nil handler mounts nothing. -func newAudiobookshelfListener(listen string, handler absMounter, artwork http.Handler, ipResolver *clientip.Resolver) *http.Server { +func newAudiobookshelfListener(listen string, handler absMounter, artwork http.Handler, ipResolver *clientip.Resolver, ingressTokens *netaccess.Registry) *http.Server { absRouter := chi.NewRouter() if ipResolver != nil { absRouter.Use(clientip.Middleware(ipResolver)) } + if ingressTokens != nil { + absRouter.Use(netaccess.Middleware(ingressTokens)) + } absRouter.Use(chimiddleware.Recoverer) absRouter.Use(httpstream.CompressExcept(5, abs.SkipMediaCompression)) handler.Mount(absRouter) diff --git a/cmd/silo/artwork_delivery_test.go b/cmd/silo/artwork_delivery_test.go index 930f5c8508..dcf80f2445 100644 --- a/cmd/silo/artwork_delivery_test.go +++ b/cmd/silo/artwork_delivery_test.go @@ -56,7 +56,7 @@ func TestAudiobookshelfListenerServesSignedArtwork(t *testing.T) { t.Fatal(err) } signer := artworkurl.NewSigner("test-secret", time.Hour) - srv := newAudiobookshelfListener(":0", noopABSMounter{}, apiv2.NewArtworkHandler(store, signer, nil), nil) + srv := newAudiobookshelfListener(":0", noopABSMounter{}, apiv2.NewArtworkHandler(store, signer, nil), nil, nil) u, _ := signer.Sign(key, time.Now()) for _, method := range []string{http.MethodGet, http.MethodHead} { rec := httptest.NewRecorder() @@ -73,7 +73,7 @@ func TestAudiobookshelfListenerServesSignedArtwork(t *testing.T) { if rec.Code != http.StatusNotFound { t.Fatalf("unsigned artwork on the ABS listener: %d", rec.Code) } - unmounted := newAudiobookshelfListener(":0", noopABSMounter{}, nil, nil) + unmounted := newAudiobookshelfListener(":0", noopABSMounter{}, nil, nil, nil) rec = httptest.NewRecorder() unmounted.Handler.ServeHTTP(rec, httptest.NewRequest(http.MethodGet, u, nil)) if rec.Code != http.StatusNotFound { diff --git a/cmd/silo/main.go b/cmd/silo/main.go index b52c3312c1..58564eb834 100644 --- a/cmd/silo/main.go +++ b/cmd/silo/main.go @@ -11,6 +11,7 @@ import ( "io/fs" "log" "log/slog" + "net" "net/http" "os" "os/signal" @@ -73,6 +74,7 @@ import ( "github.com/Silo-Server/silo-server/internal/markers" "github.com/Silo-Server/silo-server/internal/mdblist" "github.com/Silo-Server/silo-server/internal/metadata" + "github.com/Silo-Server/silo-server/internal/netaccess" // Built-in metadata providers self-register into the metadata package's // builtin registry on import; buildProviders resolves their seeded chain @@ -996,6 +998,7 @@ func main() { var handler http.Handler var shutdownStandalone func(context.Context) error + var standaloneHooks standaloneServerHooks if mode == "proxy" { srv := proxy.NewServer(watcher, tracker) proxyIPResolver, resolverErr := clientIPResolverFromConfig(watcher.Config()) @@ -1005,6 +1008,13 @@ func main() { registerClientIPConfigReload(watcher, proxyIPResolver) srv.SetClientIPResolver(proxyIPResolver) srv.SetStreamTelemetry(streamTelemetryRegistry) + // Network access providers run on this proxy too, one instance per + // node with its own overlay identity; see newProxyPluginHost. + proxyPlugins := newProxyPluginHost(appCtx, pool, dataCipher, eventBus, watcher, nodeName, cfg.Server.Listen, resolvePluginCacheDir()) + srv.SetIngressTokens(proxyPlugins.broker.Registry) + srv.SetNetworkAccessStatus(proxyPlugins.broker.Status) + srv.SetNetworkAccessProviderHost(proxyPlugins.service) + standaloneHooks = proxyPlugins.hooks() // Serve header-authenticated sessions: the recipe comes from the // shared grant store central wrote at plan time, and the caller's // own access token is re-checked against the live login session in @@ -1058,7 +1068,7 @@ func main() { _ = operationalWriter _ = opsRepo - startStandaloneServer(cfg.Server.Listen, handler, appCancel, shutdownStandalone) + startStandaloneServer(cfg.Server.Listen, handler, appCancel, shutdownStandalone, standaloneHooks) return } @@ -1396,11 +1406,55 @@ func main() { var pluginHTTPProxy *plugins.HTTPProxy pluginAutoUpdateDone := make(chan struct{}) var pluginAutoUpdater *plugins.AutoUpdateService + // Network access providers: ingress tokens issued per plugin start and the + // providers' last reported status. Shared by the plugin host (issue, + // revoke, status pushes) and all three listeners (token validation). + networkAccess := netaccess.NewBroker() + deps.NetworkAccess = networkAccess if deps.DB != nil { pluginCacheDir := resolvePluginCacheDir() repositoryStore := plugins.NewRepositoryStore(deps.DB) installationStore := plugins.NewInstallationStore(deps.DB) runtimeConfigStore := plugins.NewRuntimeConfigStore(deps.DB, deps.SecretCipher) + // This process is the api host: its resident plugins keep their + // per-instance state (overlay node keys) under the "api" scope. + instanceStateStore := plugins.NewInstanceStateStore(deps.DB, deps.SecretCipher).ForScope(plugins.HostScopeAPI) + hostInfo := func(ctx context.Context) (pluginhost.HostInfo, error) { + live := configWatcher.Config() + if live == nil { + live = cfg + } + name, _ := settingsRepo.Get(ctx, branding.KeyServerName) + if strings.TrimSpace(name) == "" { + name, _ = os.Hostname() + } + info := pluginhost.HostInfo{ + PublicBaseURL: live.Server.PublicURL, + PluginContentPrefix: plugins.ContentPrefix, + Role: pluginhost.HostRoleAPI, + Name: name, + Listeners: []pluginhost.HostListener{{ + Name: pluginhost.ListenerAPI, + Address: pluginhost.LoopbackDialAddress(live.Server.Listen), + DefaultPort: pluginhost.DefaultPortAPI, + }}, + } + if live.JellyfinCompat.Enabled && live.JellyfinCompat.Listen != "" { + info.Listeners = append(info.Listeners, pluginhost.HostListener{ + Name: pluginhost.ListenerJellyfin, + Address: pluginhost.LoopbackDialAddress(live.JellyfinCompat.Listen), + DefaultPort: pluginhost.DefaultPortJellyfin, + }) + } + if absCompatEnabled && live.AudiobookshelfCompat.Listen != "" { + info.Listeners = append(info.Listeners, pluginhost.HostListener{ + Name: pluginhost.ListenerABS, + Address: pluginhost.LoopbackDialAddress(live.AudiobookshelfCompat.Listen), + DefaultPort: pluginhost.DefaultPortABS, + }) + } + return info, nil + } catalogService := plugins.NewCatalogService(repositoryStore, plugins.CatalogServiceOptions{ SiloAPIVersion: plugins.DefaultSiloAPIVersion, }) @@ -1494,6 +1548,9 @@ func main() { return runtimeConfigStore.PutGlobalConfig(ctx, installationID, key, value) }, ), + HostInfo: hostInfo, + InstanceState: instanceStateStore, + NetworkAccess: networkAccess, Logger: hclog.New(&hclog.LoggerOptions{ Name: "plugin-host", Level: hclog.Info, @@ -1508,6 +1565,10 @@ func main() { installer, plugins.NewHostAdapter(pluginHost), ) + // Crashes of resident plugins (network access providers) reach the + // supervisor through the host's exit watcher so they restart with + // backoff instead of waiting for the next lazy RPC. + pluginHost.SetExitHandler(pluginService.HandleResidentExit) if watchProviderRegistry != nil { reloadWatchProviders := func(ctx context.Context) { if err := reloadWatchSyncPluginProviders(ctx, watchProviderRegistry, installationStore, pluginService, watchProviderRepo); err != nil { @@ -1565,6 +1626,13 @@ func main() { pluginHTTPProxy = pluginHTTPProxy.WithUserThemeLookup(plugins.NewPgUserThemeLookup(deps.DB)) pluginHTTPProxy = pluginHTTPProxy.WithUserIdentityLookup(plugins.NewPgUserIdentityLookup(deps.DB)) } + // The admin network access reads name this process as the "api" host + // and refresh the shared status cache with what the provider answers. + pluginService.SetNetworkAccessHostInfo(hostInfo) + pluginService.SetNetworkAccessStatusSink(networkAccess) + // Proxy nodes run the same resident installations; every lifecycle + // change here is announced so they reconcile at once. + pluginService.PublishLifecycleChanges(eventBus) deps.PluginService = pluginService deps.PluginHTTPProxy = pluginHTTPProxy defer func() { @@ -3013,6 +3081,7 @@ func main() { DB: deps.DB, SecretCipher: dataCipher, ClientIPResolver: ipResolver, + IngressTokens: networkAccess.Registry, StreamTelemetry: streamTelemetryRegistry, NodePlanner: deps.NodePlanner, JWTSecret: cfg.Auth.JWTSecret, @@ -3155,7 +3224,7 @@ func main() { var absSrv *http.Server if (mode == "integrated" || mode == "api") && deps.ABSHandler != nil && cfg.AudiobookshelfCompat.Listen != "" { absSrv = newAudiobookshelfListener(cfg.AudiobookshelfCompat.Listen, deps.ABSHandler, - apiv2.NewArtworkHandler(deps.Artwork, deps.ArtworkSigner, deps.ArtworkRepair), ipResolver) + apiv2.NewArtworkHandler(deps.Artwork, deps.ArtworkSigner, deps.ArtworkRepair), ipResolver, networkAccess.Registry) } // Run non-critical startup work in the background so it doesn't delay the @@ -3183,12 +3252,25 @@ func main() { } errCh := make(chan error, 3) - go func() { - slog.Info("HTTP server listening", "addr", cfg.Server.Listen) - if listenErr := srv.ListenAndServe(); listenErr != nil && listenErr != http.ErrServerClosed { - errCh <- fmt.Errorf("HTTP server error: %w", listenErr) + // Bind before serving so resident plugins, which reverse-proxy to this + // listener, are only started once it exists. + apiListener, apiListenErr := net.Listen("tcp", cfg.Server.Listen) + if apiListenErr != nil { + errCh <- fmt.Errorf("HTTP server listen: %w", apiListenErr) + } else { + go func() { + slog.Info("HTTP server listening", "addr", cfg.Server.Listen) + if serveErr := srv.Serve(apiListener); serveErr != nil && serveErr != http.ErrServerClosed { + errCh <- fmt.Errorf("HTTP server error: %w", serveErr) + } + }() + if pluginService != nil { + pluginService.StartResidents(appCtx) + if mode == "api" { + warnOnMultipleAPIReplicas(appCtx, cache.NewAPIReplicaPresence(apiRedisClient, nodeID), pluginService) + } } - }() + } if compatSrv != nil { go func() { slog.Info("Jellyfin compat server listening", "addr", compatSrv.Addr) @@ -3227,6 +3309,16 @@ func main() { shutdownCtx, shutdownCancel := context.WithTimeout(context.Background(), 30*time.Second) defer shutdownCancel() + // 0. Stop resident plugins first: their overlay listeners front the HTTP + // servers, so ingress goes away before the servers drain. + if pluginService != nil { + residentCtx, residentCancel := context.WithTimeout(shutdownCtx, 10*time.Second) + if stopErr := pluginService.StopResidents(residentCtx); stopErr != nil { + slog.Error("resident plugin shutdown error", "error", stopErr) + } + residentCancel() + } + // 1. Stop accepting new requests. if shutdownErr := srv.Shutdown(shutdownCtx); shutdownErr != nil { slog.Error("HTTP shutdown error", "error", shutdownErr) @@ -3312,9 +3404,18 @@ func runShutdownWorkWithTimeout(timeout time.Duration, work func(context.Context return work(ctx) } +// standaloneServerHooks lets a standalone mode run work that must bracket the +// listener's lifetime: afterListen runs once the address is bound (resident +// plugins reverse-proxy to it, so they start only then), beforeDrain runs +// before the HTTP server drains (their overlay ingress goes away first). +type standaloneServerHooks struct { + afterListen func() + beforeDrain func(context.Context) +} + // startStandaloneServer runs a standalone HTTP server for proxy/transcode modes. // It listens on the given address, handles graceful shutdown on SIGTERM/SIGINT. -func startStandaloneServer(addr string, handler http.Handler, appCancel context.CancelFunc, shutdownWork func(context.Context) error) { +func startStandaloneServer(addr string, handler http.Handler, appCancel context.CancelFunc, shutdownWork func(context.Context) error, hooks standaloneServerHooks) { srv := &http.Server{ Addr: addr, Handler: handler, @@ -3324,12 +3425,22 @@ func startStandaloneServer(addr string, handler http.Handler, appCancel context. } errCh := make(chan error, 1) - go func() { - slog.Info("HTTP server listening", "addr", addr) - if err := srv.ListenAndServe(); err != nil && err != http.ErrServerClosed { - errCh <- fmt.Errorf("HTTP server error: %w", err) + // Bind before serving so the after-listen hook runs against a listener + // that exists. + listener, listenErr := net.Listen("tcp", addr) + if listenErr != nil { + errCh <- fmt.Errorf("HTTP server listen: %w", listenErr) + } else { + go func() { + slog.Info("HTTP server listening", "addr", addr) + if err := srv.Serve(listener); err != nil && err != http.ErrServerClosed { + errCh <- fmt.Errorf("HTTP server error: %w", err) + } + }() + if hooks.afterListen != nil { + hooks.afterListen() } - }() + } sigCh := make(chan os.Signal, 1) signal.Notify(sigCh, syscall.SIGTERM, syscall.SIGINT) @@ -3345,6 +3456,11 @@ func startStandaloneServer(addr string, handler http.Handler, appCancel context. shutdownCtx, cancel := context.WithTimeout(context.Background(), 30*time.Second) defer cancel() + if hooks.beforeDrain != nil { + drainCtx, drainCancel := context.WithTimeout(shutdownCtx, 10*time.Second) + hooks.beforeDrain(drainCtx) + drainCancel() + } if err := srv.Shutdown(shutdownCtx); err != nil { slog.Error("HTTP shutdown error", "error", err) } diff --git a/cmd/silo/network_access_multinode_test.go b/cmd/silo/network_access_multinode_test.go new file mode 100644 index 0000000000..570e6ef7cc --- /dev/null +++ b/cmd/silo/network_access_multinode_test.go @@ -0,0 +1,352 @@ +package main + +import ( + "bytes" + "context" + "errors" + "net" + "net/http" + "os" + "os/exec" + "path/filepath" + "sync" + "testing" + "time" + + "github.com/hashicorp/go-hclog" + "github.com/jackc/pgx/v5" + "github.com/jackc/pgx/v5/pgxpool" + + "github.com/Silo-Server/silo-server/internal/api/handlers" + "github.com/Silo-Server/silo-server/internal/cache" + "github.com/Silo-Server/silo-server/internal/netaccess" + "github.com/Silo-Server/silo-server/internal/nodeconfig" + "github.com/Silo-Server/silo-server/internal/nodepool" + "github.com/Silo-Server/silo-server/internal/pluginhost" + "github.com/Silo-Server/silo-server/internal/plugins" + "github.com/Silo-Server/silo-server/internal/proxy" + "github.com/Silo-Server/silo-server/internal/secret" +) + +// memoryEventBus is the Redis event bus for one process: what an API-mode +// host and a proxy-mode host share when they run in one test binary. +type memoryEventBus struct { + mu sync.Mutex + handlers map[string][]cache.EventHandler +} + +func newMemoryEventBus() *memoryEventBus { + return &memoryEventBus{handlers: make(map[string][]cache.EventHandler)} +} + +func (b *memoryEventBus) Publish(_ context.Context, channel string, event cache.Event) error { + b.mu.Lock() + handlers := append([]cache.EventHandler(nil), b.handlers[channel]...) + b.mu.Unlock() + for _, handler := range handlers { + handler(event) + } + return nil +} + +func (b *memoryEventBus) Subscribe(_ context.Context, channel string, handler cache.EventHandler) error { + b.mu.Lock() + b.handlers[channel] = append(b.handlers[channel], handler) + b.mu.Unlock() + return nil +} + +func (b *memoryEventBus) Close() error { return nil } + +func multinodeTestPool(t *testing.T) *pgxpool.Pool { + t.Helper() + dsn := os.Getenv("SILO_TEST_DATABASE_URL") + if dsn == "" { + t.Skip("SILO_TEST_DATABASE_URL is not set") + } + pool, err := pgxpool.New(context.Background(), dsn) + if err != nil { + t.Fatalf("connect test database: %v", err) + } + t.Cleanup(pool.Close) + return pool +} + +// setServerSetting writes one server_settings row for the test and restores +// the previous value (or absence) afterwards. +func setServerSetting(t *testing.T, pool *pgxpool.Pool, key, value string) { + t.Helper() + ctx := context.Background() + var previous *string + if err := pool.QueryRow(ctx, `SELECT value FROM server_settings WHERE key = $1`, key).Scan(&previous); err != nil { + if !errors.Is(err, pgx.ErrNoRows) { + t.Fatalf("read %s: %v", key, err) + } + previous = nil + } + if _, err := pool.Exec(ctx, `INSERT INTO server_settings (key, value) VALUES ($1, $2) ON CONFLICT (key) DO UPDATE SET value = EXCLUDED.value`, key, value); err != nil { + t.Fatalf("set %s: %v", key, err) + } + t.Cleanup(func() { + if previous == nil { + _, _ = pool.Exec(ctx, `DELETE FROM server_settings WHERE key = $1`, key) + return + } + _, _ = pool.Exec(ctx, `UPDATE server_settings SET value = $2 WHERE key = $1`, key, *previous) + }) +} + +func buildResidentFixtureBinary(t *testing.T) []byte { + t.Helper() + dir := t.TempDir() + bin := filepath.Join(dir, "residentplugin") + build := exec.Command("go", "build", "-o", bin, "../../internal/plugins/testdata/residentplugin") + if out, err := build.CombinedOutput(); err != nil { + t.Fatalf("build residentplugin: %v\n%s", err, out) + } + data, err := os.ReadFile(bin) + if err != nil { + t.Fatal(err) + } + return data +} + +func waitFor(t *testing.T, what string, timeout time.Duration, cond func() bool) { + t.Helper() + deadline := time.Now().Add(timeout) + for { + if cond() { + return + } + if time.Now().After(deadline) { + t.Fatalf("timed out waiting for %s", what) + } + time.Sleep(20 * time.Millisecond) + } +} + +// TestNetworkAccessProxyReportsOriginTheAPIStores runs the api-mode plugin +// host and a proxy-mode plugin host in one process against the shared +// database and event bus, with the resident fixture installed once: +// +// - the proxy rehydrates the plugin from plugin_archives, runs it under the +// node: state scope and reports it as host node:; +// - the API's admin connect fans out to the proxy over its bearer route and +// the proxy's overlay origin comes back in the report; +// - the proxy's /health carries that origin and the node health update +// stores it on the stream_nodes row, where ClientURLFor hands it to +// clients on the provider's path; +// - disabling the installation on the API side reaches the proxy through +// the bus and stops its instance. +func TestNetworkAccessProxyReportsOriginTheAPIStores(t *testing.T) { + pool := multinodeTestPool(t) + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + const jwtSecret = "multinode-network-access-secret" + setServerSetting(t, pool, "auth.jwt_secret", jwtSecret) + + cipher, err := secret.New(bytes.Repeat([]byte("k"), secret.MinMasterKeyLen)) + if err != nil { + t.Fatal(err) + } + bus := newMemoryEventBus() + + // --- API host: install and enable the fixture ------------------------- + installationStore := plugins.NewInstallationStore(pool) + runtimeConfigStore := plugins.NewRuntimeConfigStore(pool, cipher) + repositoryStore := plugins.NewRepositoryStore(pool) + installer := plugins.NewInstaller(installationStore, plugins.InstallerOptions{BaseDir: t.TempDir()}) + apiBroker := netaccess.NewBroker() + apiHost := pluginhost.NewHost(pluginhost.Config{ + Logger: hclog.NewNullLogger(), + HostInfo: func(context.Context) (pluginhost.HostInfo, error) { + return pluginhost.HostInfo{Role: pluginhost.HostRoleAPI, Name: "api"}, nil + }, + InstanceState: plugins.NewInstanceStateStore(pool, cipher).ForScope(plugins.HostScopeAPI), + NetworkAccess: apiBroker, + }) + apiService := plugins.NewService(repositoryStore, installationStore, runtimeConfigStore, + plugins.NewCatalogService(repositoryStore, plugins.CatalogServiceOptions{SiloAPIVersion: plugins.DefaultSiloAPIVersion}), + installer, plugins.NewHostAdapter(apiHost)) + apiHost.SetExitHandler(apiService.HandleResidentExit) + apiService.SetNetworkAccessHostInfo(func(context.Context) (pluginhost.HostInfo, error) { + return pluginhost.HostInfo{Role: pluginhost.HostRoleAPI, Name: "api"}, nil + }) + apiService.SetNetworkAccessStatusSink(apiBroker) + apiService.PublishLifecycleChanges(bus) + t.Cleanup(func() { + stop, stopCancel := context.WithTimeout(context.Background(), 10*time.Second) + defer stopCancel() + _ = apiService.StopResidents(stop) + _ = apiHost.Shutdown(stop) + }) + + result, err := apiService.InstallBinaryUpload(ctx, buildResidentFixtureBinary(t)) + if err != nil { + t.Fatalf("install fixture: %v", err) + } + installationID := result.Installation.ID + t.Cleanup(func() { _ = installationStore.Delete(context.Background(), installationID) }) + enabled := true + if err := installationStore.Update(ctx, installationID, plugins.UpdateInstallationInput{Enabled: &enabled}); err != nil { + t.Fatal(err) + } + apiService.OnLifecycleChange(ctx) + + // --- Proxy node row and listener --------------------------------------- + listener, err := net.Listen("tcp", "127.0.0.1:0") + if err != nil { + t.Fatal(err) + } + proxyURL := "http://" + listener.Addr().String() + nodeRepo := nodepool.NewRepository(pool) + node, err := nodeRepo.Create(ctx, nodepool.CreateNodeInput{Name: "proxy-1", Type: nodepool.NodeTypeProxy, URL: proxyURL}) + if err != nil { + t.Fatal(err) + } + t.Cleanup(func() { _ = nodeRepo.Delete(context.Background(), node.ID) }) + nodeScope := plugins.NodeHostScope(int64(node.ID)) + + // The proxy shares this test's filesystem, so drop the installed files to + // prove it rehydrates the plugin from plugin_archives on its own, into + // its own cache dir rather than the API server's install path. + if err := os.RemoveAll(filepath.Dir(result.Installation.InstallPath)); err != nil { + t.Fatal(err) + } + proxyCacheDir := filepath.Join(t.TempDir(), "proxy-plugins") + + watcher := nodeconfig.NewWatcher(pool, cipher, bus, nodeconfig.BootstrapOverrides{ + Listen: listener.Addr().String(), Mode: "proxy", NodeURL: proxyURL, NodeName: "proxy-1", + }) + if err := watcher.Start(ctx); err != nil { + t.Fatalf("watcher start: %v", err) + } + if id, ok := watcher.NodeRowID(); !ok || id != node.ID { + t.Fatalf("watcher resolved node row %d (ok=%v), want %d", id, ok, node.ID) + } + + proxyPlugins := newProxyPluginHost(ctx, pool, cipher, bus, watcher, "proxy-1", listener.Addr().String(), proxyCacheDir) + proxyServer := proxy.NewServer(watcher, nil) + proxyServer.SetIngressTokens(proxyPlugins.broker.Registry) + proxyServer.SetNetworkAccessStatus(proxyPlugins.broker.Status) + proxyServer.SetNetworkAccessProviderHost(proxyPlugins.service) + httpServer := &http.Server{Handler: proxyServer.Handler()} + go func() { _ = httpServer.Serve(listener) }() + hooks := proxyPlugins.hooks() + hooks.afterListen() + t.Cleanup(func() { + stop, stopCancel := context.WithTimeout(context.Background(), 10*time.Second) + defer stopCancel() + hooks.beforeDrain(stop) + _ = httpServer.Shutdown(stop) + }) + + // --- API fan-out -------------------------------------------------------- + nodeHandler := handlers.NewNodeHandler(nodeRepo, nil, nil, nodeRepo, bus, nil, jwtSecret) + apiService.SetNetworkAccessNodes(nodeHandler) + + waitFor(t, "proxy resident running", 60*time.Second, func() bool { + state, ok := proxyPlugins.service.Residents().State(installationID) + return ok && state.State == plugins.ResidentRunning + }) + if matches, _ := filepath.Glob(filepath.Join(proxyCacheDir, "silo.test.resident", "0.1.0", "*", "plugin")); len(matches) != 1 { + t.Fatalf("proxy cache dir holds %v, want one rehydrated binary", matches) + } + if _, err := os.Stat(result.Installation.InstallPath); !errors.Is(err, os.ErrNotExist) { + t.Fatalf("proxy wrote to the API server's install path: %v", err) + } + + report, err := apiService.ConnectNetworkAccess(ctx, "stub", []string{nodeScope}) + if err != nil { + t.Fatalf("connect on the proxy: %v", err) + } + var proxyRow *plugins.NetworkAccessHostStatus + for i := range report.Hosts { + if report.Hosts[i].Host.ID == nodeScope { + proxyRow = &report.Hosts[i] + } + } + if proxyRow == nil { + t.Fatalf("proxy host %s missing from report %+v", nodeScope, report.Hosts) + } + if proxyRow.Host.Role != pluginhost.HostRoleProxy || proxyRow.Host.Name != "proxy-1" { + t.Fatalf("proxy host = %+v", proxyRow.Host) + } + if proxyRow.Status.State != netaccess.StateConnected || proxyRow.Status.Origin != "https://silo.stub.test" || proxyRow.Status.InstallationID != installationID { + t.Fatalf("proxy status after connect = %+v", proxyRow.Status) + } + + // --- Health pull stores the origin on the node row ---------------------- + healthy, activeJobs, egress, _, lastStats, networkAccess := nodepool.CheckNode(ctx, node) + if !healthy { + t.Fatal("proxy reported unhealthy") + } + if origin, ok := networkAccess.ConnectedOrigin("stub"); !ok || origin != "https://silo.stub.test" { + t.Fatalf("health network_access = %+v", networkAccess) + } + if err := nodeRepo.UpdateHealth(ctx, node.ID, node.URL, healthy, activeJobs, egress, lastStats, networkAccess); err != nil { + t.Fatal(err) + } + stored, err := nodeRepo.GetByID(ctx, node.ID) + if err != nil { + t.Fatal(err) + } + if got := stored.ClientURLFor(netaccess.Path{Provider: "stub"}); got != "https://silo.stub.test" { + t.Fatalf("ClientURLFor(stub) = %q from row %+v", got, stored.NetworkAccess) + } + if got := stored.ClientURLFor(netaccess.Path{}); got != proxyURL { + t.Fatalf("ClientURLFor(default) = %q, want %q", got, proxyURL) + } + + // The proxy validates the ingress token it issued to its own instance, + // and refuses a forged one. + token, ok := proxyPlugins.broker.IngressToken(installationID) + if !ok { + t.Fatal("proxy issued no ingress token") + } + for _, tc := range []struct { + token string + want int + }{{token, http.StatusOK}, {"forged", http.StatusForbidden}} { + req, _ := http.NewRequestWithContext(ctx, http.MethodGet, proxyURL+"/api/v1/health", nil) + req.Header.Set(netaccess.IngressTokenHeader, tc.token) + resp, err := http.DefaultClient.Do(req) + if err != nil { + t.Fatal(err) + } + _ = resp.Body.Close() + if resp.StatusCode != tc.want { + t.Fatalf("health with token %q: %d, want %d", tc.token, resp.StatusCode, tc.want) + } + } + + // --- Lifecycle propagation ---------------------------------------------- + disabled := false + if err := installationStore.Update(ctx, installationID, plugins.UpdateInstallationInput{Enabled: &disabled}); err != nil { + t.Fatal(err) + } + apiService.OnLifecycleChange(ctx) + waitFor(t, "proxy resident stopped after the disable", 60*time.Second, func() bool { + _, tracked := proxyPlugins.service.Residents().State(installationID) + _, clientErr := proxyPlugins.host.Client(installationID) + return !tracked && errors.Is(clientErr, pluginhost.ErrClientNotFound) + }) + if origins := proxyPlugins.broker.Status.ConnectedOrigins(); len(origins) != 0 { + t.Fatalf("proxy still advertises origins after the disable: %v", origins) + } + // The next health pull clears the row's origin. + healthy, activeJobs, egress, _, lastStats, networkAccess = nodepool.CheckNode(ctx, node) + if !healthy || len(networkAccess) != 0 { + t.Fatalf("health after disable: healthy=%v network_access=%+v", healthy, networkAccess) + } + if err := nodeRepo.UpdateHealth(ctx, node.ID, node.URL, healthy, activeJobs, egress, lastStats, networkAccess); err != nil { + t.Fatal(err) + } + stored, err = nodeRepo.GetByID(ctx, node.ID) + if err != nil { + t.Fatal(err) + } + if got := stored.ClientURLFor(netaccess.Path{Provider: "stub"}); got != "" { + t.Fatalf("ClientURLFor(stub) after disable = %q", got) + } +} diff --git a/cmd/silo/proxy_plugins.go b/cmd/silo/proxy_plugins.go new file mode 100644 index 0000000000..578caf5df8 --- /dev/null +++ b/cmd/silo/proxy_plugins.go @@ -0,0 +1,230 @@ +package main + +import ( + "context" + "errors" + "fmt" + "log/slog" + "os" + "time" + + "github.com/hashicorp/go-hclog" + "github.com/jackc/pgx/v5/pgxpool" + + "github.com/Silo-Server/silo-server/internal/cache" + "github.com/Silo-Server/silo-server/internal/netaccess" + "github.com/Silo-Server/silo-server/internal/nodeconfig" + "github.com/Silo-Server/silo-server/internal/nodepool" + "github.com/Silo-Server/silo-server/internal/pluginhost" + "github.com/Silo-Server/silo-server/internal/plugins" + "github.com/Silo-Server/silo-server/internal/secret" +) + +// proxyPluginHost is the plugin runtime a proxy node carries: the resident +// network access providers and nothing else. Each proxy runs its own instance +// of every enabled provider installation with its own overlay identity, kept +// under the instance-state scope node:, and reports that +// identity to the API through /health and the bearer network-access routes. +// +// No admin routes, no catalog, no installer, no metadata or other capability +// dispatch: lifecycle mutations happen on the API server and reach this node +// as cache.EventPluginsChanged, with a poll as the backstop. +type proxyPluginHost struct { + host *pluginhost.Host + service *plugins.Service + broker *netaccess.Broker + bus cache.EventBus + ctx context.Context +} + +// errProxyNodeRowUnknown is why a proxy runs no providers before the config +// watcher has matched this process to its stream_nodes row: without the row +// id there is no state scope to keep the overlay node key under. +var errProxyNodeRowUnknown = errors.New("this proxy's stream_nodes row is not known yet (NODE_URL or NODE_NAME must match an enabled proxy node); network access providers start once it resolves") + +// errProxyNodeDisabled is why a proxy stops its providers once an operator +// disables its row or changes its type: the API no longer lists the node as +// a network-access host, so nothing could disconnect a provider it kept +// running, and its overlay origin must not outlive the node's eligibility. +var errProxyNodeDisabled = errors.New("this proxy's stream_nodes row is disabled or no longer a proxy; network access providers stay stopped until it is enabled again") + +// newProxyPluginHost builds the proxy's plugin host, service and supervisor. +// It does not start anything; hooks() returns the listener-bracketing +// callbacks that do. cacheDir is this node's plugin cache root. +func newProxyPluginHost( + ctx context.Context, + pool *pgxpool.Pool, + cipher *secret.Cipher, + bus cache.EventBus, + watcher *nodeconfig.Watcher, + nodeName string, + listen string, + cacheDir string, +) *proxyPluginHost { + broker := netaccess.NewBroker() + installationStore := plugins.NewInstallationStore(pool) + runtimeConfigStore := plugins.NewRuntimeConfigStore(pool, cipher) + instanceState := plugins.NewInstanceStateStore(pool, cipher) + + hostInfo := func(context.Context) (pluginhost.HostInfo, error) { + live := watcher.Config() + address := listen + if live != nil && live.Server.Listen != "" { + address = live.Server.Listen + } + info := pluginhost.HostInfo{ + Role: pluginhost.HostRoleProxy, + Name: nodeName, + // A proxy exposes only its own listener: clients fetch media here + // and talk to the API server for everything else. + Listeners: []pluginhost.HostListener{{ + Name: pluginhost.ListenerAPI, + Address: pluginhost.LoopbackDialAddress(address), + DefaultPort: pluginhost.DefaultPortAPI, + }}, + } + if id, ok := watcher.NodeRowID(); ok { + info.NodeID = int64(id) + } + return info, nil + } + + host := pluginhost.NewHost(pluginhost.Config{ + RuntimeHostForStart: func(ctx context.Context) (pluginhost.HostInfoFunc, pluginhost.InstanceStateStore, error) { + info, err := hostInfo(ctx) + if err != nil { + return nil, nil, err + } + if info.NodeID <= 0 { + return nil, nil, errProxyNodeRowUnknown + } + return func(context.Context) (pluginhost.HostInfo, error) { return info, nil }, + instanceState.ForScope(plugins.NodeHostScope(info.NodeID)), nil + }, + NetworkAccess: broker, + GlobalConfigSetter: pluginhost.GlobalConfigSetterFunc( + func(ctx context.Context, installationID int, key string, value map[string]any) error { + return runtimeConfigStore.PutGlobalConfig(ctx, installationID, key, value) + }, + ), + Logger: hclog.New(&hclog.LoggerOptions{ + Name: "plugin-host", + Level: hclog.Info, + Output: os.Stderr, + }), + }) + // Plugin binaries are rehydrated from plugin_archives into this node's + // own cache dir (SILO_PLUGIN_CACHE_DIR): the API server's install paths + // name directories on its machine, not this one. + service := plugins.NewNodeService(installationStore, runtimeConfigStore, plugins.NewHostAdapter(host), cacheDir) + host.SetExitHandler(service.HandleResidentExit) + service.SetNetworkAccessHostInfo(hostInfo) + service.SetNetworkAccessStatusSink(broker) + service.SetResidentHostIdentity(func() string { + id, ok := watcher.NodeRowID() + if !ok { + return "" + } + return plugins.NodeHostScope(int64(id)) + }) + // A proxy whose row is unknown must not start providers: their node keys + // would have no scope. The gate is re-evaluated on every reconcile, so + // the poll picks the row up once the watcher resolves it. + // A provider process keeps the node identity it started under in memory + // (its overlay node key, the node id it reports). If the operator deletes + // and re-registers this proxy the watcher resolves a new row id; the + // state scope follows it per call, so the process must be replaced + // rather than left reading another scope's keys. + service.SetResidentHostIdentity(func() string { + id, ok := watcher.NodeRowID() + if !ok { + return "" + } + return plugins.NodeHostScope(int64(id)) + }) + nodes := nodepool.NewRepository(pool) + service.SetResidentGate(newProxyResidentGate(watcher.NodeRowID, nodes.GetByID)) + return &proxyPluginHost{host: host, service: service, broker: broker, bus: bus, ctx: ctx} +} + +// Reconcile calls this gate serially. A transient read failure preserves the +// last confirmed decision for the same node identity, while a missing, +// disabled, or retyped row closes the gate immediately. +func newProxyResidentGate(nodeRowID func() (int, bool), getNode func(context.Context, int) (*nodepool.Node, error)) func(context.Context) error { + var confirmedID int + var confirmedErr error + return func(ctx context.Context) error { + id, ok := nodeRowID() + if !ok || id != confirmedID { + confirmedID, confirmedErr = 0, nil + } + if !ok { + return errProxyNodeRowUnknown + } + node, err := getNode(ctx, id) + if err != nil && !errors.Is(err, nodepool.ErrNodeNotFound) { + if confirmedID == id { + slog.WarnContext(ctx, "proxy resident gate retaining its confirmed state after a node read failure", "component", "plugins", "error", err) + return confirmedErr + } + return fmt.Errorf("read this proxy's stream_nodes row: %w", err) + } + confirmedID, confirmedErr = id, nil + if errors.Is(err, nodepool.ErrNodeNotFound) || node == nil || !node.Enabled || node.Type != nodepool.NodeTypeProxy { + confirmedErr = errProxyNodeDisabled + } + return confirmedErr + } +} + +// hooks returns the callbacks startStandaloneServer runs around the listener: +// residents start once the address is bound and stop before it drains. +func (p *proxyPluginHost) hooks() standaloneServerHooks { + return standaloneServerHooks{ + afterListen: func() { + p.service.StartResidents(p.ctx) + if err := p.service.FollowLifecycleChanges(p.ctx, p.bus, plugins.DefaultLifecyclePollInterval); err != nil { + slog.Warn("proxy plugin lifecycle subscription failed; reconciling on the poll only", "component", "plugins", "error", err) + } + }, + beforeDrain: func(ctx context.Context) { + if err := p.service.StopResidents(ctx); err != nil { + slog.ErrorContext(ctx, "resident plugin shutdown error", "component", "plugins", "error", err) + } + shutdownCtx, cancel := context.WithTimeout(context.Background(), 5*time.Second) + defer cancel() + if err := p.host.Shutdown(shutdownCtx); err != nil { + slog.WarnContext(ctx, "failed to shut down plugin host", "component", "plugins", "error", err) + } + }, + } +} + +// warnOnMultipleAPIReplicas registers this API replica's presence and logs +// the single-API constraint when another live replica is seen: network +// access providers keep their overlay node key under one shared "api" state +// scope, so two replicas would present one overlay identity from two +// machines. Best effort and never fatal; a Redis-less deployment cannot have +// a second replica reading the same state. +func warnOnMultipleAPIReplicas(ctx context.Context, presence *cache.APIReplicaPresence, service *plugins.Service) { + if presence == nil || service == nil { + return + } + if err := presence.Register(ctx); err != nil { + slog.DebugContext(ctx, "api replica presence unavailable", "component", "plugins", "error", err) + return + } + providers, err := service.ListNetworkAccessProviders(ctx) + if err != nil || len(providers) == 0 { + return + } + count, err := presence.Count(ctx) + if err != nil { + slog.DebugContext(ctx, "api replica census unavailable", "component", "plugins", "error", err) + return + } + if count > 1 { + slog.WarnContext(ctx, "network access providers are installed and more than one API replica is live; they share one overlay identity under the api state scope, which phase one does not support — run a single API server or disable the providers", + "component", "plugins", "replicas", count) + } +} diff --git a/cmd/silo/proxy_plugins_test.go b/cmd/silo/proxy_plugins_test.go new file mode 100644 index 0000000000..3bf98d9933 --- /dev/null +++ b/cmd/silo/proxy_plugins_test.go @@ -0,0 +1,76 @@ +package main + +import ( + "context" + "errors" + "testing" + + "github.com/Silo-Server/silo-server/internal/nodepool" +) + +func TestProxyResidentGatePreservesConfirmedStateDuringReadErrors(t *testing.T) { + for _, closed := range []string{"missing", "nil", "disabled", "retyped"} { + t.Run(closed, func(t *testing.T) { + id := 7 + var node *nodepool.Node + readErr := errors.New("database unavailable") + gate := newProxyResidentGate(func() (int, bool) { return id, id > 0 }, func(context.Context, int) (*nodepool.Node, error) { + return node, readErr + }) + if err := gate(t.Context()); err == nil { + t.Fatal("gate opened before the first successful read") + } + readErr = nil + node = &nodepool.Node{ID: id, Type: nodepool.NodeTypeProxy, Enabled: true} + if err := gate(t.Context()); err != nil { + t.Fatalf("enabled proxy: %v", err) + } + readErr = errors.New("database unavailable") + if err := gate(t.Context()); err != nil { + t.Errorf("transient read failure closed a confirmed enabled gate: %v", err) + } + readErr = nil + switch closed { + case "missing": + node, readErr = nil, nodepool.ErrNodeNotFound + case "nil": + node = nil + case "disabled": + node.Enabled = false + case "retyped": + node.Type = nodepool.NodeTypeTranscode + } + if err := gate(t.Context()); !errors.Is(err, errProxyNodeDisabled) { + t.Errorf("confirmed %s gate error = %v, want errProxyNodeDisabled", closed, err) + } + readErr = errors.New("database unavailable") + if err := gate(t.Context()); !errors.Is(err, errProxyNodeDisabled) { + t.Errorf("read failure lost the confirmed closed state: %v", err) + } + }) + } +} + +func TestProxyResidentGateDoesNotReuseAnotherIdentity(t *testing.T) { + id := 7 + var readErr error + gate := newProxyResidentGate(func() (int, bool) { return id, id > 0 }, func(context.Context, int) (*nodepool.Node, error) { + return &nodepool.Node{ID: id, Type: nodepool.NodeTypeProxy, Enabled: true}, readErr + }) + if err := gate(t.Context()); err != nil { + t.Fatal(err) + } + id = 8 + readErr = errors.New("database unavailable") + if err := gate(t.Context()); err == nil { + t.Fatal("new row reused the previous row's confirmed enabled state") + } + id = 0 + if err := gate(t.Context()); !errors.Is(err, errProxyNodeRowUnknown) { + t.Fatalf("unknown identity error = %v", err) + } + id = 7 + if err := gate(t.Context()); err == nil { + t.Fatal("forgotten identity reused its earlier enabled state") + } +} diff --git a/cmd/silo/session_sync.go b/cmd/silo/session_sync.go index ae7bc3c3f1..e25c1ee8e0 100644 --- a/cmd/silo/session_sync.go +++ b/cmd/silo/session_sync.go @@ -39,6 +39,7 @@ func buildLiveSessionSync(s *playback.Session, reportingNode string) worker.Sess TargetBitrateKbps: s.TargetBitrateKbps, TranscodeHWAccel: s.TranscodeHWAccel, ToneMapMode: string(s.ToneMapMode), + RoutingNetworkProvider: s.RoutingNetworkProvider, RoutingWorkload: s.RoutingWorkload, RoutingExecution: s.RoutingExecution, RoutingExecutionNodeID: s.RoutingExecutionNodeID, diff --git a/contracts/api/v2/fixtures/admin_network_access_connect_accepted.json b/contracts/api/v2/fixtures/admin_network_access_connect_accepted.json new file mode 100644 index 0000000000..d45a9fc95c --- /dev/null +++ b/contracts/api/v2/fixtures/admin_network_access_connect_accepted.json @@ -0,0 +1,21 @@ +{ + "provider": "stub", + "hosts": [ + { + "host": { + "id": "api", + "role": "api", + "name": "Living Room" + }, + "state": "awaiting_authorization", + "hostname": "silo.overlay.example.test", + "origin": "https://silo.overlay.example.test", + "addresses": [ + "127.0.0.1" + ], + "auth_url": "https://login.example.test/a/abc", + "provider_version": "stub 0.1.0", + "updated_at": "2026-01-02T03:04:05.678Z" + } + ] +} diff --git a/contracts/api/v2/fixtures/admin_network_access_disconnect_accepted.json b/contracts/api/v2/fixtures/admin_network_access_disconnect_accepted.json new file mode 100644 index 0000000000..6d7395eb38 --- /dev/null +++ b/contracts/api/v2/fixtures/admin_network_access_disconnect_accepted.json @@ -0,0 +1,20 @@ +{ + "provider": "stub", + "hosts": [ + { + "host": { + "id": "api", + "role": "api", + "name": "Living Room" + }, + "state": "disconnected", + "hostname": "silo.overlay.example.test", + "origin": "https://silo.overlay.example.test", + "addresses": [ + "127.0.0.1" + ], + "provider_version": "stub 0.1.0", + "updated_at": "2026-01-02T03:04:05.678Z" + } + ] +} diff --git a/contracts/api/v2/fixtures/admin_network_access_status_not_found.json b/contracts/api/v2/fixtures/admin_network_access_status_not_found.json new file mode 100644 index 0000000000..5a1021b4fd --- /dev/null +++ b/contracts/api/v2/fixtures/admin_network_access_status_not_found.json @@ -0,0 +1,7 @@ +{ + "type": "https://siloserver.org/docs/api/v2/problems/not_found", + "title": "Not found", + "status": 404, + "detail": "Network access provider not found.", + "instance": "urn:silo:request:000000000000000000000285" +} diff --git a/contracts/api/v2/fixtures/admin_network_access_status_ok.json b/contracts/api/v2/fixtures/admin_network_access_status_ok.json new file mode 100644 index 0000000000..6d7395eb38 --- /dev/null +++ b/contracts/api/v2/fixtures/admin_network_access_status_ok.json @@ -0,0 +1,20 @@ +{ + "provider": "stub", + "hosts": [ + { + "host": { + "id": "api", + "role": "api", + "name": "Living Room" + }, + "state": "disconnected", + "hostname": "silo.overlay.example.test", + "origin": "https://silo.overlay.example.test", + "addresses": [ + "127.0.0.1" + ], + "provider_version": "stub 0.1.0", + "updated_at": "2026-01-02T03:04:05.678Z" + } + ] +} diff --git a/contracts/api/v2/fixtures/admin_network_access_status_unavailable_host.json b/contracts/api/v2/fixtures/admin_network_access_status_unavailable_host.json new file mode 100644 index 0000000000..583c85762b --- /dev/null +++ b/contracts/api/v2/fixtures/admin_network_access_status_unavailable_host.json @@ -0,0 +1,15 @@ +{ + "provider": "down", + "hosts": [ + { + "host": { + "id": "api", + "role": "api", + "name": "Living Room" + }, + "state": "unavailable", + "addresses": [], + "error": "plugin process is backoff: plugin process exited" + } + ] +} diff --git a/contracts/api/v2/fixtures/admin_network_access_unknown_host.json b/contracts/api/v2/fixtures/admin_network_access_unknown_host.json new file mode 100644 index 0000000000..d266a594d0 --- /dev/null +++ b/contracts/api/v2/fixtures/admin_network_access_unknown_host.json @@ -0,0 +1,14 @@ +{ + "type": "https://siloserver.org/docs/api/v2/problems/validation_failed", + "title": "Validation failed", + "status": 422, + "detail": "The request did not pass validation; see errors.", + "instance": "urn:silo:request:000000000000000000000288", + "errors": [ + { + "location": "body.hosts", + "code": "invalid", + "detail": "Unknown network access host." + } + ] +} diff --git a/contracts/api/v2/fixtures/admin_playback_sessions.json b/contracts/api/v2/fixtures/admin_playback_sessions.json index f3eb5de9c1..979972f6f5 100644 --- a/contracts/api/v2/fixtures/admin_playback_sessions.json +++ b/contracts/api/v2/fixtures/admin_playback_sessions.json @@ -50,6 +50,7 @@ "source_audio_channels": 8, "effective_play_method": "transcode", "is_jellyfin_client": true, + "routing_network_provider": "tailscale", "routing_execution_node_id": "9" } ], diff --git a/contracts/api/v2/fixtures/get_system_info_ok.json b/contracts/api/v2/fixtures/get_system_info_ok.json index 911a3a435f..3c197cc8c1 100644 --- a/contracts/api/v2/fixtures/get_system_info_ok.json +++ b/contracts/api/v2/fixtures/get_system_info_ok.json @@ -1,7 +1,7 @@ { "server_version": "unavailable", "api_major": 2, - "contract_digest": "e46579ccd64acf055bc60d644e5e8d93d3d03b73fee018b288dbff8fde3b0e58", + "contract_digest": "7866f390a42da02fa2ce39ffbd26d421c0778f32fc816935c9cf35f614d91598", "links": { "openapi": "/api/v2/openapi.json", "capabilities": "/api/v2/capabilities" diff --git a/contracts/api/v2/fixtures/index.json b/contracts/api/v2/fixtures/index.json index 664e3ab5b5..086935fd3d 100644 --- a/contracts/api/v2/fixtures/index.json +++ b/contracts/api/v2/fixtures/index.json @@ -1783,6 +1783,128 @@ "schema": "#/components/schemas/MetadataTranslationJob", "body_file": "admin_metadata_translation_queued.json" }, + { + "name": "admin_network_access_connect_accepted", + "operation_id": "connectNetworkAccess", + "scenario": "Connect on the named host is acknowledged with the state reached so far; enrollment continues and the auth URL is admin-only.", + "request": { + "method": "POST", + "path": "/api/v2/admin/network-access/stub/connect", + "headers": { + "Authorization": "Bearer tok-admin", + "X-Profile-Id": "p-primary" + }, + "body": "{\"hosts\":[\"api\"]}" + }, + "expected_status": 202, + "response_headers": { + "Content-Type": "application/json" + }, + "response_media_type": "application/json", + "schema": "#/components/schemas/NetworkAccessStatus", + "body_file": "admin_network_access_connect_accepted.json" + }, + { + "name": "admin_network_access_disconnect_accepted", + "operation_id": "disconnectNetworkAccess", + "scenario": "Disconnect without a body acts on every host and is acknowledged with the state reached.", + "request": { + "method": "POST", + "path": "/api/v2/admin/network-access/stub/disconnect", + "headers": { + "Authorization": "Bearer tok-admin", + "X-Profile-Id": "p-primary" + } + }, + "expected_status": 202, + "response_headers": { + "Content-Type": "application/json" + }, + "response_media_type": "application/json", + "schema": "#/components/schemas/NetworkAccessStatus", + "body_file": "admin_network_access_disconnect_accepted.json" + }, + { + "name": "admin_network_access_status_not_found", + "operation_id": "getAdminNetworkAccessStatus", + "scenario": "A provider slug no enabled installation declares.", + "request": { + "method": "GET", + "path": "/api/v2/admin/network-access/netbird/status", + "headers": { + "Authorization": "Bearer tok-admin", + "X-Profile-Id": "p-primary" + } + }, + "expected_status": 404, + "response_headers": { + "Content-Type": "application/problem+json" + }, + "response_media_type": "application/problem+json", + "schema": "#/components/schemas/Problem", + "body_file": "admin_network_access_status_not_found.json" + }, + { + "name": "admin_network_access_status_ok", + "operation_id": "getAdminNetworkAccessStatus", + "scenario": "The provider's live state on the API host, with overlay origin and addresses.", + "request": { + "method": "GET", + "path": "/api/v2/admin/network-access/stub/status", + "headers": { + "Authorization": "Bearer tok-admin", + "X-Profile-Id": "p-primary" + } + }, + "expected_status": 200, + "response_headers": { + "Content-Type": "application/json" + }, + "response_media_type": "application/json", + "schema": "#/components/schemas/NetworkAccessStatus", + "body_file": "admin_network_access_status_ok.json" + }, + { + "name": "admin_network_access_status_unavailable_host", + "operation_id": "getAdminNetworkAccessStatus", + "scenario": "A host whose provider plugin is not running answers state unavailable with the supervisor's reason and no updated_at.", + "request": { + "method": "GET", + "path": "/api/v2/admin/network-access/down/status", + "headers": { + "Authorization": "Bearer tok-admin", + "X-Profile-Id": "p-primary" + } + }, + "expected_status": 200, + "response_headers": { + "Content-Type": "application/json" + }, + "response_media_type": "application/json", + "schema": "#/components/schemas/NetworkAccessStatus", + "body_file": "admin_network_access_status_unavailable_host.json" + }, + { + "name": "admin_network_access_unknown_host", + "operation_id": "connectNetworkAccess", + "scenario": "A host id this deployment does not run the provider on is a validation failure at body.hosts; nothing is applied.", + "request": { + "method": "POST", + "path": "/api/v2/admin/network-access/stub/connect", + "headers": { + "Authorization": "Bearer tok-admin", + "X-Profile-Id": "p-primary" + }, + "body": "{\"hosts\":[\"node:99\"]}" + }, + "expected_status": 422, + "response_headers": { + "Content-Type": "application/problem+json" + }, + "response_media_type": "application/problem+json", + "schema": "#/components/schemas/Problem", + "body_file": "admin_network_access_unknown_host.json" + }, { "name": "admin_person_refreshed", "operation_id": "refreshAdminPerson", @@ -9534,6 +9656,27 @@ "schema": "#/components/schemas/Problem", "body_file": "mark_watched_profile_header_required.json" }, + { + "name": "network_access_capabilities_ok", + "operation_id": "getNetworkAccessCapabilities", + "scenario": "An authenticated account lists the installed overlay-network providers read from plugin manifests.", + "request": { + "method": "GET", + "path": "/api/v2/network-access/capabilities", + "headers": { + "Authorization": "Bearer tok-member" + } + }, + "expected_status": 200, + "response_headers": { + "Cache-Control": "private, no-cache", + "Content-Type": "application/json", + "ETag": "\"1400a291b22fc7156ba4727d7eafa8a40d6654b9ee567eb9bbd470de447fd03b\"" + }, + "response_media_type": "application/json", + "schema": "#/components/schemas/NetworkAccessCapabilities", + "body_file": "network_access_capabilities_ok.json" + }, { "name": "not_acceptable", "operation_id": "getSystemInfo", diff --git a/contracts/api/v2/fixtures/network_access_capabilities_ok.json b/contracts/api/v2/fixtures/network_access_capabilities_ok.json new file mode 100644 index 0000000000..c70da943f6 --- /dev/null +++ b/contracts/api/v2/fixtures/network_access_capabilities_ok.json @@ -0,0 +1,17 @@ +{ + "revision": "e7f28bde2cff62b41a60d6310d700206095c3c0a08d3db51bf0e0a543f47e906", + "state": "available", + "allowed": true, + "providers": [ + { + "provider": "stub", + "display_name": "Stub Overlay", + "installation_id": "7" + }, + { + "provider": "down", + "display_name": "Down Overlay", + "installation_id": "9" + } + ] +} diff --git a/contracts/api/v2/migration.json b/contracts/api/v2/migration.json index 6167e82e5f..fd3c9894d2 100644 --- a/contracts/api/v2/migration.json +++ b/contracts/api/v2/migration.json @@ -10,7 +10,7 @@ "consumer_method": "Client call sites were extracted by scripts/apiv2-ledger/extract_consumers.py from web/src (this repo) and the origin/main trees of silo-apple and silo-android at the commits pinned in source_trees (re-resolve a sibling site with git show : in that repo, never against a moving branch): path literals and templates passed to HTTP/WebSocket calls, plus paths built by helper functions that return a template, templates rooted at the API base URL, Kotlin buildString/const val builders, and Swift let/static let path bindings, with interpolations normalized to a wildcard and matched against inventory paths by method (scripts/apiv2-ledger/match_consumers.py). scripts/apiv2-ledger/sweep_uncredited.py then greps the last two static segments of every inventory path across all three trees and every hit not within four lines of a credited site was resolved by hand. Server-internal and compat consumers are Go URL builders and callers in this repo. Sites marked match=manual were resolved by hand where the path is assembled from variables the scripts cannot follow. Call-site files are repository-root-relative for their repo (web: src/... under web/; apple: iosApp/...; android: shared/, android-shared/, androidApp/, androidTvApp/; server: this repository root). Absence of a first-party consumer is not proof of disuse.", "field_reference": "internal/contractledger validates this file against migration.schema.json and reconciles it against the route inventory: listener, namespace, method, path, registration_index, handler, handler_kind, source_file, route_group, middleware_chain, auth_class, auth_traits, conditional, conditions, delegates_to (inventory empty string is null here), request_kind, response_media_kind, streams, and upgrades_websocket are copied from the inventory row and must match it, except that dynamic_plugin_proxy rows record both media kinds as dynamic_proxy; profile_required and admin_required are derived from auth_traits. Every other field is curated by hand and preserved when scripts/apiv2-ledger/build_ledger.py merges a regenerated inventory into this file.", "totals": { - "entries": 773 + "entries": 777 }, "entries": [ { @@ -34458,7 +34458,6 @@ "upgrades_websocket": false, "capability_endpoint": null, "consumers": [ - "compat", "external_unknown", "internal", "web" @@ -45785,6 +45784,212 @@ }, "notes": "External caller: Prometheus scraper. Inventory media kind is heuristic and unresolved for this row; confirm in the section PR." }, + { + "listener": "proxy", + "namespace": "legacy_unversioned", + "method": "GET", + "path": "/network-access/status", + "registration_index": 0, + "handler": "(*internal/proxy.Server).handleNetworkAccessStatus", + "handler_kind": "method", + "source_file": "internal/proxy/server.go", + "route_group": "/", + "middleware_chain": 26, + "auth_class": "node_bearer", + "auth_traits": [ + "cors", + "node_bearer" + ], + "profile_required": false, + "admin_required": false, + "conditional": false, + "conditions": [], + "delegates_to": null, + "request_kind": "none", + "response_media_kind": "json", + "streams": false, + "upgrades_websocket": false, + "capability_endpoint": null, + "consumers": [ + "internal" + ], + "consumer_call_sites": [ + { + "repo": "server", + "file": "internal/api/handlers/admin_node_network_access.go", + "line": 49, + "types": [], + "match": "manual" + } + ], + "section": "node-proxy", + "release_flow": "operational", + "tier": 2, + "disposition": "ported", + "disposition_rule": "listener_delegation", + "disposition_rationale": "No exclusion, removal, or ratified redesign applies. The node listeners have no /api/v2 namespace: this route is retained at its version-neutral path on the node listener and described through the typed worker protocol extension (`x-silo-worker-protocols`), with no /api/v2 alias (plan constraint 4; api-contract.md section 8, version-neutral legacy routes are retired individually and are not aliases into v2). Its section PR confirms retention or retires it individually.", + "owner": null, + "review_state": "ratified", + "v2": { + "method": null, + "path": null, + "operation_id": null + }, + "notes": "Retained on proxy listener with node_bearer; v2 unset by design; description: x-silo-worker-protocols proxy GET /network-access/status. v2 is null by design for node-listener rows: the route keeps its version-neutral path and is described by the worker protocol extension (`x-silo-worker-protocols`), not aliased under /api/v2. Internal consumer: internal/api/handlers/admin_node_network_access.go fans GET /api/v2/admin/network-access/{provider}/status out to every enabled proxy node through this route with the node bearer and a ten-second timeout." + }, + { + "listener": "proxy", + "namespace": "legacy_unversioned", + "method": "POST", + "path": "/network-access/{provider}/connect", + "registration_index": 0, + "handler": "(*internal/proxy.Server).handleNetworkAccessConnect", + "handler_kind": "method", + "source_file": "internal/proxy/server.go", + "route_group": "/", + "middleware_chain": 26, + "auth_class": "node_bearer", + "auth_traits": [ + "cors", + "node_bearer" + ], + "profile_required": false, + "admin_required": false, + "conditional": false, + "conditions": [], + "delegates_to": null, + "request_kind": "unknown", + "response_media_kind": "json", + "streams": false, + "upgrades_websocket": false, + "capability_endpoint": null, + "consumers": [ + "internal" + ], + "consumer_call_sites": [ + { + "repo": "server", + "file": "internal/api/handlers/admin_node_network_access.go", + "line": 63, + "types": [], + "match": "manual" + } + ], + "section": "node-proxy", + "release_flow": "operational", + "tier": 2, + "disposition": "ported", + "disposition_rule": "listener_delegation", + "disposition_rationale": "No exclusion, removal, or ratified redesign applies. The node listeners have no /api/v2 namespace: this route is retained at its version-neutral path on the node listener and described through the typed worker protocol extension (`x-silo-worker-protocols`), with no /api/v2 alias (plan constraint 4; api-contract.md section 8, version-neutral legacy routes are retired individually and are not aliases into v2). Its section PR confirms retention or retires it individually.", + "owner": null, + "review_state": "ratified", + "v2": { + "method": null, + "path": null, + "operation_id": null + }, + "notes": "Retained on proxy listener with node_bearer; v2 unset by design; description: x-silo-worker-protocols proxy POST /network-access/{provider}/connect. v2 is null by design for node-listener rows: the route keeps its version-neutral path and is described by the worker protocol extension (`x-silo-worker-protocols`), not aliased under /api/v2. Internal consumer: internal/api/handlers/admin_node_network_access.go fans POST /api/v2/admin/network-access/{provider}/connect out to every enabled proxy node through this route with the node bearer and a ten-second timeout. Inventory media kind is heuristic and unresolved for this row; confirm in the section PR.", + "retry_safety": "natural_idempotent" + }, + { + "listener": "proxy", + "namespace": "legacy_unversioned", + "method": "POST", + "path": "/network-access/{provider}/disconnect", + "registration_index": 0, + "handler": "(*internal/proxy.Server).handleNetworkAccessDisconnect", + "handler_kind": "method", + "source_file": "internal/proxy/server.go", + "route_group": "/", + "middleware_chain": 26, + "auth_class": "node_bearer", + "auth_traits": [ + "cors", + "node_bearer" + ], + "profile_required": false, + "admin_required": false, + "conditional": false, + "conditions": [], + "delegates_to": null, + "request_kind": "unknown", + "response_media_kind": "json", + "streams": false, + "upgrades_websocket": false, + "capability_endpoint": null, + "consumers": [ + "internal" + ], + "consumer_call_sites": [ + { + "repo": "server", + "file": "internal/api/handlers/admin_node_network_access.go", + "line": 70, + "types": [], + "match": "manual" + } + ], + "section": "node-proxy", + "release_flow": "operational", + "tier": 2, + "disposition": "ported", + "disposition_rule": "listener_delegation", + "disposition_rationale": "No exclusion, removal, or ratified redesign applies. The node listeners have no /api/v2 namespace: this route is retained at its version-neutral path on the node listener and described through the typed worker protocol extension (`x-silo-worker-protocols`), with no /api/v2 alias (plan constraint 4; api-contract.md section 8, version-neutral legacy routes are retired individually and are not aliases into v2). Its section PR confirms retention or retires it individually.", + "owner": null, + "review_state": "ratified", + "v2": { + "method": null, + "path": null, + "operation_id": null + }, + "notes": "Retained on proxy listener with node_bearer; v2 unset by design; description: x-silo-worker-protocols proxy POST /network-access/{provider}/disconnect. v2 is null by design for node-listener rows: the route keeps its version-neutral path and is described by the worker protocol extension (`x-silo-worker-protocols`), not aliased under /api/v2. Internal consumer: internal/api/handlers/admin_node_network_access.go fans POST /api/v2/admin/network-access/{provider}/disconnect out to every enabled proxy node through this route with the node bearer and a ten-second timeout. Inventory media kind is heuristic and unresolved for this row; confirm in the section PR.", + "retry_safety": "natural_idempotent" + }, + { + "listener": "proxy", + "namespace": "legacy_unversioned", + "method": "GET", + "path": "/network-access/{provider}/status", + "registration_index": 0, + "handler": "(*internal/proxy.Server).handleNetworkAccessProviderStatus", + "handler_kind": "method", + "source_file": "internal/proxy/server.go", + "route_group": "/", + "middleware_chain": 26, + "auth_class": "node_bearer", + "auth_traits": [ + "cors", + "node_bearer" + ], + "profile_required": false, + "admin_required": false, + "conditional": false, + "conditions": [], + "delegates_to": null, + "request_kind": "none", + "response_media_kind": "json", + "streams": false, + "upgrades_websocket": false, + "capability_endpoint": null, + "consumers": [ + "unused" + ], + "consumer_call_sites": [], + "section": "node-proxy", + "release_flow": "playback_lifecycle", + "tier": 1, + "disposition": "ported", + "disposition_rule": "default_ported", + "disposition_rationale": "No exclusion, removal, or ratified redesign applies. The node listeners have no /api/v2 namespace: this route is retained at its version-neutral path on the node listener and described through the typed worker protocol extension (`x-silo-worker-protocols`), with no /api/v2 alias (plan constraint 4; api-contract.md section 8, version-neutral legacy routes are retired individually and are not aliases into v2). Its section PR confirms retention or retires it individually.", + "owner": null, + "review_state": "proposed", + "v2": { + "method": null, + "path": null, + "operation_id": null + }, + "notes": "v2 is null by design for node-listener rows: the route keeps its version-neutral path and is described by the worker protocol extension (`x-silo-worker-protocols`), not aliased under /api/v2. No first-party or internal consumer found. Absence from first-party clients is not proof of disuse: third-party clients exist, and removal needs an affirmative product decision." + }, { "listener": "proxy", "namespace": "legacy_unversioned", diff --git a/contracts/api/v2/openapi.json b/contracts/api/v2/openapi.json index 011d9c5725..187854435f 100644 --- a/contracts/api/v2/openapi.json +++ b/contracts/api/v2/openapi.json @@ -7458,6 +7458,13 @@ "name": { "type": "string" }, + "network_access": { + "additionalProperties": { + "$ref": "#/components/schemas/AdminNodeNetworkAccess" + }, + "description": "Last network access provider status the node reported on its health check, keyed by provider slug. Omitted when the node reports no providers.", + "type": "object" + }, "physical_gpu_keys": { "items": { "type": "string" @@ -7579,6 +7586,34 @@ ], "type": "object" }, + "AdminNodeNetworkAccess": { + "additionalProperties": false, + "properties": { + "hostname": { + "description": "Overlay DNS name of the node.", + "type": "string" + }, + "origin": { + "description": "scheme://host[:port] clients on the provider's overlay use to reach this node. Only used while state is connected.", + "type": "string" + }, + "state": { + "description": "disconnected | awaiting_authorization | connecting | connected | error", + "type": "string" + }, + "updated_at": { + "description": "When the node last heard from the provider, on the node's clock.", + "format": "date-time", + "pattern": "^(?:[1-9][0-9]{3}|0[1-9][0-9]{2}|00[1-9][0-9]|000[2-9])-", + "patternDescription": "a non-zero RFC 3339 instant", + "type": "string" + } + }, + "required": [ + "state" + ], + "type": "object" + }, "AdminNodeReloadOutputBody": { "additionalProperties": false, "properties": { @@ -8544,6 +8579,10 @@ "routing_execution_node_name": { "type": "string" }, + "routing_network_provider": { + "description": "Access network selected for playback: empty means default; absent means unknown; otherwise the validated provider identifier.", + "type": "string" + }, "routing_workload": { "type": "string" }, @@ -8711,6 +8750,9 @@ "is_jellyfin_client": { "type": "boolean" }, + "network_access_route": { + "type": "boolean" + }, "node_observations": { "type": "boolean" }, @@ -8767,6 +8809,7 @@ "client_build", "client_channel", "target_audio_channels", + "network_access_route", "node_routing", "revision", "state", @@ -9634,6 +9677,9 @@ }, "type": "array" }, + "runtime": { + "$ref": "#/components/schemas/AdminPluginRuntime" + }, "source_kind": { "enum": [ "silo", @@ -9690,6 +9736,7 @@ "global_configs", "auth_bindings", "task_bindings", + "runtime", "created_at", "updated_at" ], @@ -9860,6 +9907,55 @@ }, "type": "object" }, + "AdminPluginRuntime": { + "additionalProperties": false, + "properties": { + "last_error": { + "description": "Why the process last stopped or failed to start", + "type": "string" + }, + "last_started_at": { + "description": "RFC 3339 instant in UTC with millisecond precision", + "format": "date-time", + "pattern": "^(?:[1-9][0-9]{3}|0[1-9][0-9]{2}|00[1-9][0-9]|000[2-9])-", + "patternDescription": "a non-zero RFC 3339 instant", + "type": "string" + }, + "next_restart_at": { + "description": "Scheduled automatic restart while in backoff", + "format": "date-time", + "pattern": "^(?:[1-9][0-9]{3}|0[1-9][0-9]{2}|00[1-9][0-9]|000[2-9])-", + "patternDescription": "a non-zero RFC 3339 instant", + "type": "string" + }, + "resident": { + "description": "True when the server supervises this plugin's process", + "type": "boolean" + }, + "restart_count": { + "description": "Automatic restarts since the plugin last ran stably or was restarted by an administrator", + "format": "int64", + "type": "integer" + }, + "state": { + "description": "Process state; backoff and failed occur only for resident plugins", + "enum": [ + "stopped", + "starting", + "running", + "backoff", + "failed" + ], + "type": "string" + } + }, + "required": [ + "resident", + "state", + "restart_count" + ], + "type": "object" + }, "AdminPluginTaskBinding": { "additionalProperties": false, "properties": { @@ -27161,6 +27257,189 @@ ], "type": "object" }, + "NetworkAccessCapabilities": { + "additionalProperties": false, + "properties": { + "allowed": { + "description": "Whether the current principal may use the capability", + "type": "boolean" + }, + "providers": { + "items": { + "$ref": "#/components/schemas/NetworkAccessProviderSummary" + }, + "type": "array" + }, + "revision": { + "description": "Opaque revision of this document", + "type": "string" + }, + "state": { + "description": "Support and configuration state, not health", + "enum": [ + "available", + "disabled", + "not_configured", + "unsupported" + ], + "type": "string" + } + }, + "required": [ + "providers", + "revision", + "state", + "allowed" + ], + "type": "object" + }, + "NetworkAccessCommand": { + "additionalProperties": false, + "properties": { + "hosts": { + "description": "Host ids to act on (api, node:\u003cid\u003e); omitted means every host", + "items": { + "type": "string" + }, + "maxItems": 256, + "type": "array" + } + }, + "type": "object" + }, + "NetworkAccessHostRef": { + "additionalProperties": false, + "properties": { + "id": { + "description": "api for the API server; node:\u003cid\u003e for a proxy node", + "type": "string" + }, + "name": { + "description": "Server name for the API host; node name for a proxy", + "type": "string" + }, + "role": { + "enum": [ + "api", + "proxy" + ], + "type": "string" + } + }, + "required": [ + "id", + "role", + "name" + ], + "type": "object" + }, + "NetworkAccessHostStatus": { + "additionalProperties": false, + "properties": { + "addresses": { + "description": "Overlay IP addresses", + "items": { + "type": "string" + }, + "type": "array" + }, + "auth_url": { + "description": "Interactive enrollment URL, only while awaiting_authorization; admin-only, never logged", + "type": "string" + }, + "error": { + "type": "string" + }, + "host": { + "$ref": "#/components/schemas/NetworkAccessHostRef" + }, + "hostname": { + "description": "Overlay DNS name", + "type": "string" + }, + "origin": { + "description": "scheme://host[:port] clients reach the API listener at over the overlay", + "type": "string" + }, + "provider_version": { + "type": "string" + }, + "raw_state": { + "description": "the provider's own state string when it is not one of the published states", + "type": "string" + }, + "state": { + "description": "Published host state; provider states outside this vocabulary map to error", + "enum": [ + "disconnected", + "awaiting_authorization", + "connecting", + "connected", + "error", + "unavailable" + ], + "type": "string" + }, + "updated_at": { + "description": "When the host last heard from the provider; absent while unavailable", + "format": "date-time", + "pattern": "^(?:[1-9][0-9]{3}|0[1-9][0-9]{2}|00[1-9][0-9]|000[2-9])-", + "patternDescription": "a non-zero RFC 3339 instant", + "type": "string" + } + }, + "required": [ + "host", + "state", + "addresses" + ], + "type": "object" + }, + "NetworkAccessProviderSummary": { + "additionalProperties": false, + "properties": { + "display_name": { + "type": "string" + }, + "installation_id": { + "description": "Plugin installation declaring network_access_provider.v1", + "examples": [ + "1" + ], + "minLength": 1, + "type": "string" + }, + "provider": { + "description": "Stable provider slug used in the admin routes, e.g. tailscale", + "type": "string" + } + }, + "required": [ + "provider", + "display_name", + "installation_id" + ], + "type": "object" + }, + "NetworkAccessStatus": { + "additionalProperties": false, + "properties": { + "hosts": { + "items": { + "$ref": "#/components/schemas/NetworkAccessHostStatus" + }, + "type": "array" + }, + "provider": { + "type": "string" + } + }, + "required": [ + "provider", + "hosts" + ], + "type": "object" + }, "NodeHWAccel": { "additionalProperties": false, "properties": { @@ -71734,9 +72013,9 @@ "x-silo-service-backed": true } }, - "/api/v2/admin/node-sessions": { - "get": { - "operationId": "listAdminNodeSessions", + "/api/v2/admin/network-access/{provider}/connect": { + "post": { + "operationId": "connectNetworkAccess", "parameters": [ { "description": "Optional. When present, it must name the authenticated account's primary profile; an absent header is accepted.", @@ -71761,56 +72040,36 @@ } }, { - "description": "Page size; default 50, maximum 200", - "explode": false, - "in": "query", - "name": "limit", - "schema": { - "default": 50, - "description": "Page size; default 50, maximum 200", - "examples": [ - 50 - ], - "format": "int64", - "maximum": 200, - "minimum": 1, - "type": "integer" - } - }, - { - "explode": false, - "in": "query", - "name": "cursor", - "schema": { - "maxLength": 8192, - "type": "string" - } - }, - { - "description": "Opaque identifier", - "explode": false, - "in": "query", - "name": "node_id", + "in": "path", + "name": "provider", + "required": true, "schema": { - "description": "Opaque identifier", - "examples": [ - "1" - ], + "maxLength": 64, "minLength": 1, + "pattern": "^[a-z0-9]+(?:[._-][a-z0-9]+)*$", "type": "string" } } ], + "requestBody": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/NetworkAccessCommand" + } + } + } + }, "responses": { - "200": { + "202": { "content": { "application/json": { "schema": { - "$ref": "#/components/schemas/AdminNodeSessionsOutputBody" + "$ref": "#/components/schemas/NetworkAccessStatus" } } }, - "description": "OK" + "description": "Accepted" }, "400": { "content": { @@ -71862,6 +72121,36 @@ }, "description": "Not Acceptable" }, + "408": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Request Timeout" + }, + "413": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Request Entity Too Large" + }, + "415": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Unsupported Media Type" + }, "422": { "content": { "application/problem+json": { @@ -71908,17 +72197,19 @@ "bearerAuth": [] } ], - "summary": "Read best-effort Redis observations, not authoritative playback sessions. Each page enumerates current records; expired or unreadable values may be absent.", + "summary": "Ask the provider on the named hosts (every host when hosts is omitted) to bring its overlay identity up and start proxying. Answers 202 with the state each host reached within ten seconds; enrollment may continue in the background (awaiting_authorization carries the auth_url), so poll status for the final state. Repeating the request converges on one connected instance per host. Hosts not named answer their current status; an unknown host id is 422.", "tags": [ - "admin" + "network-access" ], "x-silo-class": "acting_admin", + "x-silo-demo-restricted": true, + "x-silo-retry-safety": "natural_idempotent", "x-silo-service-backed": true } }, - "/api/v2/admin/nodes": { - "get": { - "operationId": "listAdminNodes", + "/api/v2/admin/network-access/{provider}/disconnect": { + "post": { + "operationId": "disconnectNetworkAccess", "parameters": [ { "description": "Optional. When present, it must name the authenticated account's primary profile; an absent header is accepted.", @@ -71943,42 +72234,36 @@ } }, { - "description": "Page size; default 50, maximum 200", - "explode": false, - "in": "query", - "name": "limit", - "schema": { - "default": 50, - "description": "Page size; default 50, maximum 200", - "examples": [ - 50 - ], - "format": "int64", - "maximum": 200, - "minimum": 1, - "type": "integer" - } - }, - { - "explode": false, - "in": "query", - "name": "cursor", + "in": "path", + "name": "provider", + "required": true, "schema": { - "maxLength": 8192, + "maxLength": 64, + "minLength": 1, + "pattern": "^[a-z0-9]+(?:[._-][a-z0-9]+)*$", "type": "string" } } ], + "requestBody": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/NetworkAccessCommand" + } + } + } + }, "responses": { - "200": { + "202": { "content": { "application/json": { "schema": { - "$ref": "#/components/schemas/CollectionAdminNode" + "$ref": "#/components/schemas/NetworkAccessStatus" } } }, - "description": "OK" + "description": "Accepted" }, "400": { "content": { @@ -72030,6 +72315,36 @@ }, "description": "Not Acceptable" }, + "408": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Request Timeout" + }, + "413": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Request Entity Too Large" + }, + "415": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Unsupported Media Type" + }, "422": { "content": { "application/problem+json": { @@ -72076,15 +72391,19 @@ "bearerAuth": [] } ], - "summary": "Page configured nodes and their last stored observations; no worker probe is performed.", + "summary": "Ask the provider on the named hosts (every host when hosts is omitted) to tear its overlay listener down and clear its desired-connected intent. Answers 202 with the state each host reached within ten seconds. Repeating the request converges on disconnected. Hosts not named answer their current status; an unknown host id is 422.", "tags": [ - "admin-nodes" + "network-access" ], "x-silo-class": "acting_admin", + "x-silo-demo-restricted": true, + "x-silo-retry-safety": "natural_idempotent", "x-silo-service-backed": true - }, - "post": { - "operationId": "createAdminNode", + } + }, + "/api/v2/admin/network-access/{provider}/status": { + "get": { + "operationId": "getAdminNetworkAccessStatus", "parameters": [ { "description": "Optional. When present, it must name the authenticated account's primary profile; an absent header is accepted.", @@ -72107,35 +72426,29 @@ ], "type": "string" } + }, + { + "in": "path", + "name": "provider", + "required": true, + "schema": { + "maxLength": 64, + "minLength": 1, + "pattern": "^[a-z0-9]+(?:[._-][a-z0-9]+)*$", + "type": "string" + } } ], - "requestBody": { - "content": { - "application/json": { - "schema": { - "$ref": "#/components/schemas/AdminNodeCreateBody" - } - } - }, - "required": true - }, "responses": { - "201": { + "200": { "content": { "application/json": { "schema": { - "$ref": "#/components/schemas/AdminNode" + "$ref": "#/components/schemas/NetworkAccessStatus" } } }, - "description": "Created", - "headers": { - "ETag": { - "schema": { - "type": "string" - } - } - } + "description": "OK" }, "400": { "content": { @@ -72187,46 +72500,6 @@ }, "description": "Not Acceptable" }, - "408": { - "content": { - "application/problem+json": { - "schema": { - "$ref": "#/components/schemas/Problem" - } - } - }, - "description": "Request Timeout" - }, - "409": { - "content": { - "application/problem+json": { - "schema": { - "$ref": "#/components/schemas/Problem" - } - } - }, - "description": "Conflict" - }, - "413": { - "content": { - "application/problem+json": { - "schema": { - "$ref": "#/components/schemas/Problem" - } - } - }, - "description": "Request Entity Too Large" - }, - "415": { - "content": { - "application/problem+json": { - "schema": { - "$ref": "#/components/schemas/Problem" - } - } - }, - "description": "Unsupported Media Type" - }, "422": { "content": { "application/problem+json": { @@ -72273,19 +72546,17 @@ "bearerAuth": [] } ], - "summary": "Persist node configuration and durable pool invalidation. An identical natural URL configuration resolves a repeated create; conflicting configuration returns 409. No worker probe or replica completion acknowledgement.", + "summary": "Read the provider's live state on every host that runs it, asking each plugin instance directly with a ten-second timeout. A host whose plugin process is not running answers state unavailable with the supervisor's last error. An unknown provider slug is 404.", "tags": [ - "admin-nodes" + "network-access" ], "x-silo-class": "acting_admin", - "x-silo-demo-restricted": true, - "x-silo-retry-safety": "unique_constraint", "x-silo-service-backed": true } }, - "/api/v2/admin/nodes/force-reload": { - "post": { - "operationId": "forceReloadAdminNodes", + "/api/v2/admin/node-sessions": { + "get": { + "operationId": "listAdminNodeSessions", "parameters": [ { "description": "Optional. When present, it must name the authenticated account's primary profile; an absent header is accepted.", @@ -72308,6 +72579,46 @@ ], "type": "string" } + }, + { + "description": "Page size; default 50, maximum 200", + "explode": false, + "in": "query", + "name": "limit", + "schema": { + "default": 50, + "description": "Page size; default 50, maximum 200", + "examples": [ + 50 + ], + "format": "int64", + "maximum": 200, + "minimum": 1, + "type": "integer" + } + }, + { + "explode": false, + "in": "query", + "name": "cursor", + "schema": { + "maxLength": 8192, + "type": "string" + } + }, + { + "description": "Opaque identifier", + "explode": false, + "in": "query", + "name": "node_id", + "schema": { + "description": "Opaque identifier", + "examples": [ + "1" + ], + "minLength": 1, + "type": "string" + } } ], "responses": { @@ -72315,7 +72626,7 @@ "content": { "application/json": { "schema": { - "$ref": "#/components/schemas/AdminNodeReloadOutputBody" + "$ref": "#/components/schemas/AdminNodeSessionsOutputBody" } } }, @@ -72417,37 +72728,18 @@ "bearerAuth": [] } ], - "summary": "Synchronously request force reload from every currently enumerated enabled node in parallel, with ten-second per-node timeout and no redirects or replay. May tear down sessions. Results are individual acknowledgements, not an atomic or durable completion receipt.", + "summary": "Read best-effort Redis observations, not authoritative playback sessions. Each page enumerates current records; expired or unreadable values may be absent.", "tags": [ - "admin-nodes" + "admin" ], "x-silo-class": "acting_admin", - "x-silo-demo-restricted": true, - "x-silo-retry-safety": "non_retryable", "x-silo-service-backed": true } }, - "/api/v2/admin/nodes/{id}": { - "delete": { - "operationId": "deleteAdminNode", + "/api/v2/admin/nodes": { + "get": { + "operationId": "listAdminNodes", "parameters": [ - { - "description": "The resource's current ETag, or \"*\" to overwrite deliberately. A missing field is 428 precondition_required; a stale tag is 412 precondition_failed with the current ETag.", - "in": "header", - "name": "If-Match", - "required": true, - "schema": { - "type": "string" - } - }, - { - "description": "Optional second precondition, evaluated after If-Match succeeds: \"*\" or any tag matching the current representation is 412 precondition_failed with the current ETag.", - "in": "header", - "name": "If-None-Match", - "schema": { - "type": "string" - } - }, { "description": "Optional. When present, it must name the authenticated account's primary profile; an absent header is accepted.", "in": "header", @@ -72471,19 +72763,42 @@ } }, { - "in": "path", - "name": "id", - "required": true, + "description": "Page size; default 50, maximum 200", + "explode": false, + "in": "query", + "name": "limit", "schema": { - "maxLength": 10, - "pattern": "^[1-9][0-9]*$", + "default": 50, + "description": "Page size; default 50, maximum 200", + "examples": [ + 50 + ], + "format": "int64", + "maximum": 200, + "minimum": 1, + "type": "integer" + } + }, + { + "explode": false, + "in": "query", + "name": "cursor", + "schema": { + "maxLength": 8192, "type": "string" } } ], "responses": { - "204": { - "description": "No Content" + "200": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/CollectionAdminNode" + } + } + }, + "description": "OK" }, "400": { "content": { @@ -72535,24 +72850,6 @@ }, "description": "Not Acceptable" }, - "412": { - "content": { - "application/problem+json": { - "schema": { - "$ref": "#/components/schemas/Problem" - } - } - }, - "description": "Precondition Failed", - "headers": { - "ETag": { - "description": "The strong, opaque validator of the representation; send it back in If-Match on a guarded mutation or If-None-Match on a conditional read.", - "schema": { - "type": "string" - } - } - } - }, "422": { "content": { "application/problem+json": { @@ -72563,16 +72860,6 @@ }, "description": "Unprocessable Entity" }, - "428": { - "content": { - "application/problem+json": { - "schema": { - "$ref": "#/components/schemas/Problem" - } - } - }, - "description": "Precondition Required" - }, "429": { "content": { "application/problem+json": { @@ -72609,36 +72896,16 @@ "bearerAuth": [] } ], - "summary": "Delete the original If-Match configuration and atomically persist pool invalidation for replica reconciliation. No worker teardown, session completion or all-replica acknowledgement. A subsequent 404 is not proof of this caller's outcome.", + "summary": "Page configured nodes and their last stored observations; no worker probe is performed.", "tags": [ "admin-nodes" ], "x-silo-class": "acting_admin", - "x-silo-demo-restricted": true, - "x-silo-guarded": true, - "x-silo-retry-safety": "durable_dispatch", "x-silo-service-backed": true }, - "put": { - "operationId": "updateAdminNode", + "post": { + "operationId": "createAdminNode", "parameters": [ - { - "description": "The resource's current ETag, or \"*\" to overwrite deliberately. A missing field is 428 precondition_required; a stale tag is 412 precondition_failed with the current ETag.", - "in": "header", - "name": "If-Match", - "required": true, - "schema": { - "type": "string" - } - }, - { - "description": "Optional second precondition, evaluated after If-Match succeeds: \"*\" or any tag matching the current representation is 412 precondition_failed with the current ETag.", - "in": "header", - "name": "If-None-Match", - "schema": { - "type": "string" - } - }, { "description": "Optional. When present, it must name the authenticated account's primary profile; an absent header is accepted.", "in": "header", @@ -72660,30 +72927,20 @@ ], "type": "string" } - }, - { - "in": "path", - "name": "id", - "required": true, - "schema": { - "maxLength": 10, - "pattern": "^[1-9][0-9]*$", - "type": "string" - } } ], "requestBody": { "content": { "application/json": { "schema": { - "$ref": "#/components/schemas/AdminNodeUpdateBody" + "$ref": "#/components/schemas/AdminNodeCreateBody" } } }, "required": true }, "responses": { - "200": { + "201": { "content": { "application/json": { "schema": { @@ -72691,10 +72948,9 @@ } } }, - "description": "OK", + "description": "Created", "headers": { "ETag": { - "description": "The strong, opaque validator of the representation; send it back in If-Match on a guarded mutation or If-None-Match on a conditional read.", "schema": { "type": "string" } @@ -72771,24 +73027,6 @@ }, "description": "Conflict" }, - "412": { - "content": { - "application/problem+json": { - "schema": { - "$ref": "#/components/schemas/Problem" - } - } - }, - "description": "Precondition Failed", - "headers": { - "ETag": { - "description": "The strong, opaque validator of the representation; send it back in If-Match on a guarded mutation or If-None-Match on a conditional read.", - "schema": { - "type": "string" - } - } - } - }, "413": { "content": { "application/problem+json": { @@ -72819,16 +73057,6 @@ }, "description": "Unprocessable Entity" }, - "428": { - "content": { - "application/problem+json": { - "schema": { - "$ref": "#/components/schemas/Problem" - } - } - }, - "description": "Precondition Required" - }, "429": { "content": { "application/problem+json": { @@ -72865,20 +73093,19 @@ "bearerAuth": [] } ], - "summary": "Update stored node configuration under the original If-Match revision. Configuration changes advance ETag and persist pool invalidation; a no-change PUT retains ETag. Disabling alone removes new placement and routine health sampling after reconciliation, preserves the last health sample, leaves existing streams serving and does not contact the worker. The response acknowledges stored configuration, not worker reload, replica completion or session teardown.", + "summary": "Persist node configuration and durable pool invalidation. An identical natural URL configuration resolves a repeated create; conflicting configuration returns 409. No worker probe or replica completion acknowledgement.", "tags": [ "admin-nodes" ], "x-silo-class": "acting_admin", "x-silo-demo-restricted": true, - "x-silo-guarded": true, - "x-silo-retry-safety": "natural_idempotent", + "x-silo-retry-safety": "unique_constraint", "x-silo-service-backed": true } }, - "/api/v2/admin/nodes/{id}/check": { + "/api/v2/admin/nodes/force-reload": { "post": { - "operationId": "checkAdminNode", + "operationId": "forceReloadAdminNodes", "parameters": [ { "description": "Optional. When present, it must name the authenticated account's primary profile; an absent header is accepted.", @@ -72901,16 +73128,6 @@ ], "type": "string" } - }, - { - "in": "path", - "name": "id", - "required": true, - "schema": { - "maxLength": 10, - "pattern": "^[1-9][0-9]*$", - "type": "string" - } } ], "responses": { @@ -72918,7 +73135,7 @@ "content": { "application/json": { "schema": { - "$ref": "#/components/schemas/AdminNodeHealth" + "$ref": "#/components/schemas/AdminNodeReloadOutputBody" } } }, @@ -73020,20 +73237,37 @@ "bearerAuth": [] } ], - "summary": "Synchronously observe node health and apply URL-fenced persistence and process pool updates. Unhealthy is a successful observation; no durable job or hardware support guarantee.", + "summary": "Synchronously request force reload from every currently enumerated enabled node in parallel, with ten-second per-node timeout and no redirects or replay. May tear down sessions. Results are individual acknowledgements, not an atomic or durable completion receipt.", "tags": [ "admin-nodes" ], "x-silo-class": "acting_admin", "x-silo-demo-restricted": true, - "x-silo-retry-safety": "natural_idempotent", + "x-silo-retry-safety": "non_retryable", "x-silo-service-backed": true } }, - "/api/v2/admin/nodes/{id}/force-reload": { - "post": { - "operationId": "forceReloadAdminNode", + "/api/v2/admin/nodes/{id}": { + "delete": { + "operationId": "deleteAdminNode", "parameters": [ + { + "description": "The resource's current ETag, or \"*\" to overwrite deliberately. A missing field is 428 precondition_required; a stale tag is 412 precondition_failed with the current ETag.", + "in": "header", + "name": "If-Match", + "required": true, + "schema": { + "type": "string" + } + }, + { + "description": "Optional second precondition, evaluated after If-Match succeeds: \"*\" or any tag matching the current representation is 412 precondition_failed with the current ETag.", + "in": "header", + "name": "If-None-Match", + "schema": { + "type": "string" + } + }, { "description": "Optional. When present, it must name the authenticated account's primary profile; an absent header is accepted.", "in": "header", @@ -73068,15 +73302,8 @@ } ], "responses": { - "200": { - "content": { - "application/json": { - "schema": { - "$ref": "#/components/schemas/AdminNodeReloadOutputBody" - } - } - }, - "description": "OK" + "204": { + "description": "No Content" }, "400": { "content": { @@ -73128,6 +73355,24 @@ }, "description": "Not Acceptable" }, + "412": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Precondition Failed", + "headers": { + "ETag": { + "description": "The strong, opaque validator of the representation; send it back in If-Match on a guarded mutation or If-None-Match on a conditional read.", + "schema": { + "type": "string" + } + } + } + }, "422": { "content": { "application/problem+json": { @@ -73138,6 +73383,16 @@ }, "description": "Unprocessable Entity" }, + "428": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Precondition Required" + }, "429": { "content": { "application/problem+json": { @@ -73174,20 +73429,36 @@ "bearerAuth": [] } ], - "summary": "Synchronously request force reload from one stored node, including a disabled node. May tear down sessions. Ten-second timeout, no redirects or automatic replay; per-node status is acknowledgement or uncertainty, not a durable receipt.", + "summary": "Delete the original If-Match configuration and atomically persist pool invalidation for replica reconciliation. No worker teardown, session completion or all-replica acknowledgement. A subsequent 404 is not proof of this caller's outcome.", "tags": [ "admin-nodes" ], "x-silo-class": "acting_admin", "x-silo-demo-restricted": true, - "x-silo-retry-safety": "non_retryable", + "x-silo-guarded": true, + "x-silo-retry-safety": "durable_dispatch", "x-silo-service-backed": true - } - }, - "/api/v2/admin/nodes/{id}/reprobe": { - "post": { - "operationId": "reprobeAdminNode", + }, + "put": { + "operationId": "updateAdminNode", "parameters": [ + { + "description": "The resource's current ETag, or \"*\" to overwrite deliberately. A missing field is 428 precondition_required; a stale tag is 412 precondition_failed with the current ETag.", + "in": "header", + "name": "If-Match", + "required": true, + "schema": { + "type": "string" + } + }, + { + "description": "Optional second precondition, evaluated after If-Match succeeds: \"*\" or any tag matching the current representation is 412 precondition_failed with the current ETag.", + "in": "header", + "name": "If-None-Match", + "schema": { + "type": "string" + } + }, { "description": "Optional. When present, it must name the authenticated account's primary profile; an absent header is accepted.", "in": "header", @@ -73221,16 +73492,34 @@ } } ], + "requestBody": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/AdminNodeUpdateBody" + } + } + }, + "required": true + }, "responses": { "200": { "content": { "application/json": { "schema": { - "$ref": "#/components/schemas/AdminNodeReprobe" + "$ref": "#/components/schemas/AdminNode" } } }, - "description": "OK" + "description": "OK", + "headers": { + "ETag": { + "description": "The strong, opaque validator of the representation; send it back in If-Match on a guarded mutation or If-None-Match on a conditional read.", + "schema": { + "type": "string" + } + } + } }, "400": { "content": { @@ -73282,7 +73571,7 @@ }, "description": "Not Acceptable" }, - "422": { + "408": { "content": { "application/problem+json": { "schema": { @@ -73290,9 +73579,9 @@ } } }, - "description": "Unprocessable Entity" + "description": "Request Timeout" }, - "429": { + "409": { "content": { "application/problem+json": { "schema": { @@ -73300,9 +73589,9 @@ } } }, - "description": "Too Many Requests" + "description": "Conflict" }, - "500": { + "412": { "content": { "application/problem+json": { "schema": { @@ -73310,93 +73599,17 @@ } } }, - "description": "Internal Server Error" - }, - "503": { - "content": { - "application/problem+json": { + "description": "Precondition Failed", + "headers": { + "ETag": { + "description": "The strong, opaque validator of the representation; send it back in If-Match on a guarded mutation or If-None-Match on a conditional read.", "schema": { - "$ref": "#/components/schemas/Problem" + "type": "string" } } - }, - "description": "Service Unavailable" - } - }, - "security": [ - { - "bearerAuth": [] - } - ], - "summary": "Synchronously request a node hardware reprobe with existing policy-derived deadlines. The body distinguishes node refusal or uncertainty from success and reports inventory refresh separately. No automatic replay or durable completion receipt.", - "tags": [ - "admin-nodes" - ], - "x-silo-class": "acting_admin", - "x-silo-demo-restricted": true, - "x-silo-retry-safety": "non_retryable", - "x-silo-service-backed": true - } - }, - "/api/v2/admin/notifications/discord/test": { - "post": { - "operationId": "testAdminDiscordNotification", - "parameters": [ - { - "description": "Optional. When present, it must name the authenticated account's primary profile; an absent header is accepted.", - "in": "header", - "name": "X-Profile-Id", - "schema": { - "examples": [ - "1" - ], - "type": "string" - } - }, - { - "description": "Verification proof for a PIN-locked profile, issued by POST /api/v2/profiles/{id}/verify-pin; required only when the declared profile is locked", - "in": "header", - "name": "X-Profile-Token", - "schema": { - "examples": [ - "pvt_5f3a9c1e7b2d4e8fa0c6" - ], - "type": "string" } - } - ], - "responses": { - "200": { - "content": { - "application/json": { - "schema": { - "$ref": "#/components/schemas/AdminNotificationDiscordTestResult" - } - } - }, - "description": "OK" - }, - "400": { - "content": { - "application/problem+json": { - "schema": { - "$ref": "#/components/schemas/Problem" - } - } - }, - "description": "Bad Request" }, - "401": { - "content": { - "application/problem+json": { - "schema": { - "$ref": "#/components/schemas/Problem" - } - } - }, - "description": "Unauthorized" - }, - "403": { + "413": { "content": { "application/problem+json": { "schema": { @@ -73404,9 +73617,9 @@ } } }, - "description": "Forbidden" + "description": "Request Entity Too Large" }, - "404": { + "415": { "content": { "application/problem+json": { "schema": { @@ -73414,9 +73627,9 @@ } } }, - "description": "Not Found" + "description": "Unsupported Media Type" }, - "406": { + "422": { "content": { "application/problem+json": { "schema": { @@ -73424,9 +73637,9 @@ } } }, - "description": "Not Acceptable" + "description": "Unprocessable Entity" }, - "422": { + "428": { "content": { "application/problem+json": { "schema": { @@ -73434,7 +73647,7 @@ } } }, - "description": "Unprocessable Entity" + "description": "Precondition Required" }, "429": { "content": { @@ -73472,19 +73685,20 @@ "bearerAuth": [] } ], - "summary": "Verify the stored Discord bot credential without sending a message.", + "summary": "Update stored node configuration under the original If-Match revision. Configuration changes advance ETag and persist pool invalidation; a no-change PUT retains ETag. Disabling alone removes new placement and routine health sampling after reconciliation, preserves the last health sample, leaves existing streams serving and does not contact the worker. The response acknowledges stored configuration, not worker reload, replica completion or session teardown.", "tags": [ - "admin" + "admin-nodes" ], "x-silo-class": "acting_admin", "x-silo-demo-restricted": true, - "x-silo-retry-safety": "non_retryable", + "x-silo-guarded": true, + "x-silo-retry-safety": "natural_idempotent", "x-silo-service-backed": true } }, - "/api/v2/admin/notifications/push/apple/test": { + "/api/v2/admin/nodes/{id}/check": { "post": { - "operationId": "testAdminApplePushNotification", + "operationId": "checkAdminNode", "parameters": [ { "description": "Optional. When present, it must name the authenticated account's primary profile; an absent header is accepted.", @@ -73507,24 +73721,24 @@ ], "type": "string" } + }, + { + "in": "path", + "name": "id", + "required": true, + "schema": { + "maxLength": 10, + "pattern": "^[1-9][0-9]*$", + "type": "string" + } } ], - "requestBody": { - "content": { - "application/json": { - "schema": { - "$ref": "#/components/schemas/AdminNotificationPushTestInputBody" - } - } - }, - "required": true - }, "responses": { "200": { "content": { "application/json": { "schema": { - "$ref": "#/components/schemas/AdminNotificationPushTestResult" + "$ref": "#/components/schemas/AdminNodeHealth" } } }, @@ -73580,36 +73794,6 @@ }, "description": "Not Acceptable" }, - "408": { - "content": { - "application/problem+json": { - "schema": { - "$ref": "#/components/schemas/Problem" - } - } - }, - "description": "Request Timeout" - }, - "413": { - "content": { - "application/problem+json": { - "schema": { - "$ref": "#/components/schemas/Problem" - } - } - }, - "description": "Request Entity Too Large" - }, - "415": { - "content": { - "application/problem+json": { - "schema": { - "$ref": "#/components/schemas/Problem" - } - } - }, - "description": "Unsupported Media Type" - }, "422": { "content": { "application/problem+json": { @@ -73656,19 +73840,19 @@ "bearerAuth": [] } ], - "summary": "Dispatch one test push notification and report its current delivery outcome.", + "summary": "Synchronously observe node health and apply URL-fenced persistence and process pool updates. Unhealthy is a successful observation; no durable job or hardware support guarantee.", "tags": [ - "admin" + "admin-nodes" ], "x-silo-class": "acting_admin", "x-silo-demo-restricted": true, - "x-silo-retry-safety": "non_retryable", + "x-silo-retry-safety": "natural_idempotent", "x-silo-service-backed": true } }, - "/api/v2/admin/notifications/push/fcm/test": { + "/api/v2/admin/nodes/{id}/force-reload": { "post": { - "operationId": "testAdminAndroidPushNotification", + "operationId": "forceReloadAdminNode", "parameters": [ { "description": "Optional. When present, it must name the authenticated account's primary profile; an absent header is accepted.", @@ -73691,24 +73875,24 @@ ], "type": "string" } + }, + { + "in": "path", + "name": "id", + "required": true, + "schema": { + "maxLength": 10, + "pattern": "^[1-9][0-9]*$", + "type": "string" + } } ], - "requestBody": { - "content": { - "application/json": { - "schema": { - "$ref": "#/components/schemas/AdminNotificationPushTestInputBody" - } - } - }, - "required": true - }, "responses": { "200": { "content": { "application/json": { "schema": { - "$ref": "#/components/schemas/AdminNotificationPushTestResult" + "$ref": "#/components/schemas/AdminNodeReloadOutputBody" } } }, @@ -73764,36 +73948,6 @@ }, "description": "Not Acceptable" }, - "408": { - "content": { - "application/problem+json": { - "schema": { - "$ref": "#/components/schemas/Problem" - } - } - }, - "description": "Request Timeout" - }, - "413": { - "content": { - "application/problem+json": { - "schema": { - "$ref": "#/components/schemas/Problem" - } - } - }, - "description": "Request Entity Too Large" - }, - "415": { - "content": { - "application/problem+json": { - "schema": { - "$ref": "#/components/schemas/Problem" - } - } - }, - "description": "Unsupported Media Type" - }, "422": { "content": { "application/problem+json": { @@ -73840,9 +73994,9 @@ "bearerAuth": [] } ], - "summary": "Dispatch one test push notification and report its current delivery outcome.", + "summary": "Synchronously request force reload from one stored node, including a disabled node. May tear down sessions. Ten-second timeout, no redirects or automatic replay; per-node status is acknowledgement or uncertainty, not a durable receipt.", "tags": [ - "admin" + "admin-nodes" ], "x-silo-class": "acting_admin", "x-silo-demo-restricted": true, @@ -73850,9 +74004,9 @@ "x-silo-service-backed": true } }, - "/api/v2/admin/notifications/push/relay": { - "delete": { - "operationId": "clearAdminNotificationRelay", + "/api/v2/admin/nodes/{id}/reprobe": { + "post": { + "operationId": "reprobeAdminNode", "parameters": [ { "description": "Optional. When present, it must name the authenticated account's primary profile; an absent header is accepted.", @@ -73875,11 +74029,28 @@ ], "type": "string" } + }, + { + "in": "path", + "name": "id", + "required": true, + "schema": { + "maxLength": 10, + "pattern": "^[1-9][0-9]*$", + "type": "string" + } } ], "responses": { - "204": { - "description": "No Content" + "200": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/AdminNodeReprobe" + } + } + }, + "description": "OK" }, "400": { "content": { @@ -73977,9 +74148,9 @@ "bearerAuth": [] } ], - "summary": "Clear the local push relay credential.", + "summary": "Synchronously request a node hardware reprobe with existing policy-derived deadlines. The body distinguishes node refusal or uncertainty from success and reports inventory refresh separately. No automatic replay or durable completion receipt.", "tags": [ - "admin" + "admin-nodes" ], "x-silo-class": "acting_admin", "x-silo-demo-restricted": true, @@ -73987,9 +74158,9 @@ "x-silo-service-backed": true } }, - "/api/v2/admin/notifications/push/relay/register": { + "/api/v2/admin/notifications/discord/test": { "post": { - "operationId": "registerAdminNotificationRelay", + "operationId": "testAdminDiscordNotification", "parameters": [ { "description": "Optional. When present, it must name the authenticated account's primary profile; an absent header is accepted.", @@ -74014,22 +74185,12 @@ } } ], - "requestBody": { - "content": { - "application/json": { - "schema": { - "$ref": "#/components/schemas/NotificationRelayRegisterInputBody" - } - } - }, - "required": true - }, "responses": { "200": { "content": { "application/json": { "schema": { - "$ref": "#/components/schemas/NotificationRelayRegistration" + "$ref": "#/components/schemas/AdminNotificationDiscordTestResult" } } }, @@ -74085,36 +74246,6 @@ }, "description": "Not Acceptable" }, - "408": { - "content": { - "application/problem+json": { - "schema": { - "$ref": "#/components/schemas/Problem" - } - } - }, - "description": "Request Timeout" - }, - "413": { - "content": { - "application/problem+json": { - "schema": { - "$ref": "#/components/schemas/Problem" - } - } - }, - "description": "Request Entity Too Large" - }, - "415": { - "content": { - "application/problem+json": { - "schema": { - "$ref": "#/components/schemas/Problem" - } - } - }, - "description": "Unsupported Media Type" - }, "422": { "content": { "application/problem+json": { @@ -74161,7 +74292,7 @@ "bearerAuth": [] } ], - "summary": "Register or rotate the configured push relay credential.", + "summary": "Verify the stored Discord bot credential without sending a message.", "tags": [ "admin" ], @@ -74171,9 +74302,9 @@ "x-silo-service-backed": true } }, - "/api/v2/admin/notifications/server-channels": { - "get": { - "operationId": "listAdminNotificationServerChannels", + "/api/v2/admin/notifications/push/apple/test": { + "post": { + "operationId": "testAdminApplePushNotification", "parameters": [ { "description": "Optional. When present, it must name the authenticated account's primary profile; an absent header is accepted.", @@ -74196,39 +74327,24 @@ ], "type": "string" } - }, - { - "description": "Page size; default 50, maximum 200", - "explode": false, - "in": "query", - "name": "limit", - "schema": { - "default": 50, - "description": "Page size; default 50, maximum 200", - "examples": [ - 50 - ], - "format": "int64", - "maximum": 200, - "minimum": 1, - "type": "integer" - } - }, - { - "explode": false, - "in": "query", - "name": "cursor", - "schema": { - "type": "string" - } } ], + "requestBody": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/AdminNotificationPushTestInputBody" + } + } + }, + "required": true + }, "responses": { "200": { "content": { "application/json": { "schema": { - "$ref": "#/components/schemas/CollectionNotificationServerChannel" + "$ref": "#/components/schemas/AdminNotificationPushTestResult" } } }, @@ -74284,6 +74400,36 @@ }, "description": "Not Acceptable" }, + "408": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Request Timeout" + }, + "413": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Request Entity Too Large" + }, + "415": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Unsupported Media Type" + }, "422": { "content": { "application/problem+json": { @@ -74330,15 +74476,19 @@ "bearerAuth": [] } ], - "summary": "List server notification channels.", + "summary": "Dispatch one test push notification and report its current delivery outcome.", "tags": [ "admin" ], "x-silo-class": "acting_admin", + "x-silo-demo-restricted": true, + "x-silo-retry-safety": "non_retryable", "x-silo-service-backed": true - }, + } + }, + "/api/v2/admin/notifications/push/fcm/test": { "post": { - "operationId": "createAdminNotificationServerChannel", + "operationId": "testAdminAndroidPushNotification", "parameters": [ { "description": "Optional. When present, it must name the authenticated account's primary profile; an absent header is accepted.", @@ -74367,22 +74517,22 @@ "content": { "application/json": { "schema": { - "$ref": "#/components/schemas/NotificationServerChannelCreateInputBody" + "$ref": "#/components/schemas/AdminNotificationPushTestInputBody" } } }, "required": true }, "responses": { - "201": { + "200": { "content": { "application/json": { "schema": { - "$ref": "#/components/schemas/NotificationDestinationCreated" + "$ref": "#/components/schemas/AdminNotificationPushTestResult" } } }, - "description": "Created" + "description": "OK" }, "400": { "content": { @@ -74510,7 +74660,7 @@ "bearerAuth": [] } ], - "summary": "Create a server notification channel and reveal its signing secret once.", + "summary": "Dispatch one test push notification and report its current delivery outcome.", "tags": [ "admin" ], @@ -74520,9 +74670,9 @@ "x-silo-service-backed": true } }, - "/api/v2/admin/notifications/server-channels/{id}": { + "/api/v2/admin/notifications/push/relay": { "delete": { - "operationId": "deleteAdminNotificationServerChannel", + "operationId": "clearAdminNotificationRelay", "parameters": [ { "description": "Optional. When present, it must name the authenticated account's primary profile; an absent header is accepted.", @@ -74545,14 +74695,6 @@ ], "type": "string" } - }, - { - "in": "path", - "name": "id", - "required": true, - "schema": { - "type": "string" - } } ], "responses": { @@ -74655,17 +74797,19 @@ "bearerAuth": [] } ], - "summary": "Delete the exact server notification channel and its row-local delivery bookkeeping. Does not recall already-dispatched provider work.", + "summary": "Clear the local push relay credential.", "tags": [ "admin" ], "x-silo-class": "acting_admin", "x-silo-demo-restricted": true, - "x-silo-retry-safety": "natural_idempotent", + "x-silo-retry-safety": "non_retryable", "x-silo-service-backed": true - }, - "put": { - "operationId": "updateAdminNotificationServerChannel", + } + }, + "/api/v2/admin/notifications/push/relay/register": { + "post": { + "operationId": "registerAdminNotificationRelay", "parameters": [ { "description": "Optional. When present, it must name the authenticated account's primary profile; an absent header is accepted.", @@ -74688,21 +74832,13 @@ ], "type": "string" } - }, - { - "in": "path", - "name": "id", - "required": true, - "schema": { - "type": "string" - } } ], "requestBody": { "content": { "application/json": { "schema": { - "$ref": "#/components/schemas/NotificationServerChannelUpdateInputBody" + "$ref": "#/components/schemas/NotificationRelayRegisterInputBody" } } }, @@ -74713,7 +74849,7 @@ "content": { "application/json": { "schema": { - "$ref": "#/components/schemas/NotificationServerChannel" + "$ref": "#/components/schemas/NotificationRelayRegistration" } } }, @@ -74845,7 +74981,7 @@ "bearerAuth": [] } ], - "summary": "Update channel configuration and atomically reset dispatch state on URL replacement or re-enabling. Never replay an uncertain update.", + "summary": "Register or rotate the configured push relay credential.", "tags": [ "admin" ], @@ -74855,9 +74991,9 @@ "x-silo-service-backed": true } }, - "/api/v2/admin/notifications/server-channels/{id}/rotate-secret": { - "post": { - "operationId": "rotateAdminNotificationServerChannelSecret", + "/api/v2/admin/notifications/server-channels": { + "get": { + "operationId": "listAdminNotificationServerChannels", "parameters": [ { "description": "Optional. When present, it must name the authenticated account's primary profile; an absent header is accepted.", @@ -74882,9 +75018,26 @@ } }, { - "in": "path", - "name": "id", - "required": true, + "description": "Page size; default 50, maximum 200", + "explode": false, + "in": "query", + "name": "limit", + "schema": { + "default": 50, + "description": "Page size; default 50, maximum 200", + "examples": [ + 50 + ], + "format": "int64", + "maximum": 200, + "minimum": 1, + "type": "integer" + } + }, + { + "explode": false, + "in": "query", + "name": "cursor", "schema": { "type": "string" } @@ -74895,18 +75048,11 @@ "content": { "application/json": { "schema": { - "$ref": "#/components/schemas/NotificationWebhookSecretOutputBody" + "$ref": "#/components/schemas/CollectionNotificationServerChannel" } } }, - "description": "OK", - "headers": { - "Cache-Control": { - "schema": { - "type": "string" - } - } - } + "description": "OK" }, "400": { "content": { @@ -75004,19 +75150,15 @@ "bearerAuth": [] } ], - "summary": "Replace the generic server channel signing secret and reveal it once. Never replay an uncertain rotation.", + "summary": "List server notification channels.", "tags": [ "admin" ], "x-silo-class": "acting_admin", - "x-silo-demo-restricted": true, - "x-silo-retry-safety": "non_retryable", "x-silo-service-backed": true - } - }, - "/api/v2/admin/notifications/server-channels/{id}/test": { + }, "post": { - "operationId": "testAdminNotificationServerChannel", + "operationId": "createAdminNotificationServerChannel", "parameters": [ { "description": "Optional. When present, it must name the authenticated account's primary profile; an absent header is accepted.", @@ -75039,32 +75181,28 @@ ], "type": "string" } - }, - { - "description": "Opaque identifier", - "in": "path", - "name": "id", - "required": true, - "schema": { - "description": "Opaque identifier", - "examples": [ - "1" - ], - "minLength": 1, - "type": "string" - } } ], + "requestBody": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/NotificationServerChannelCreateInputBody" + } + } + }, + "required": true + }, "responses": { - "200": { + "201": { "content": { "application/json": { "schema": { - "$ref": "#/components/schemas/NotificationDestinationTestResult" + "$ref": "#/components/schemas/NotificationDestinationCreated" } } }, - "description": "OK" + "description": "Created" }, "400": { "content": { @@ -75116,6 +75254,36 @@ }, "description": "Not Acceptable" }, + "408": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Request Timeout" + }, + "413": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Request Entity Too Large" + }, + "415": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Unsupported Media Type" + }, "422": { "content": { "application/problem+json": { @@ -75162,7 +75330,7 @@ "bearerAuth": [] } ], - "summary": "Send one synchronous sample to a server notification channel.", + "summary": "Create a server notification channel and reveal its signing secret once.", "tags": [ "admin" ], @@ -75172,9 +75340,9 @@ "x-silo-service-backed": true } }, - "/api/v2/admin/people/{id}": { - "patch": { - "operationId": "updateAdminPerson", + "/api/v2/admin/notifications/server-channels/{id}": { + "delete": { + "operationId": "deleteAdminNotificationServerChannel", "parameters": [ { "description": "Optional. When present, it must name the authenticated account's primary profile; an absent header is accepted.", @@ -75199,40 +75367,17 @@ } }, { - "description": "Opaque identifier", "in": "path", "name": "id", "required": true, "schema": { - "description": "Opaque identifier", - "examples": [ - "1" - ], - "minLength": 1, "type": "string" } } ], - "requestBody": { - "content": { - "application/json": { - "schema": { - "$ref": "#/components/schemas/AdminPersonUpdate" - } - } - }, - "required": true - }, "responses": { - "200": { - "content": { - "application/json": { - "schema": { - "$ref": "#/components/schemas/Person" - } - } - }, - "description": "OK" + "204": { + "description": "No Content" }, "400": { "content": { @@ -75284,36 +75429,6 @@ }, "description": "Not Acceptable" }, - "408": { - "content": { - "application/problem+json": { - "schema": { - "$ref": "#/components/schemas/Problem" - } - } - }, - "description": "Request Timeout" - }, - "413": { - "content": { - "application/problem+json": { - "schema": { - "$ref": "#/components/schemas/Problem" - } - } - }, - "description": "Request Entity Too Large" - }, - "415": { - "content": { - "application/problem+json": { - "schema": { - "$ref": "#/components/schemas/Problem" - } - } - }, - "description": "Unsupported Media Type" - }, "422": { "content": { "application/problem+json": { @@ -75360,19 +75475,17 @@ "bearerAuth": [] } ], - "summary": "Apply a partial person metadata update.", + "summary": "Delete the exact server notification channel and its row-local delivery bookkeeping. Does not recall already-dispatched provider work.", "tags": [ - "admin-catalog" + "admin" ], "x-silo-class": "acting_admin", "x-silo-demo-restricted": true, - "x-silo-retry-safety": "non_retryable", + "x-silo-retry-safety": "natural_idempotent", "x-silo-service-backed": true - } - }, - "/api/v2/admin/people/{id}/refresh": { - "post": { - "operationId": "refreshAdminPerson", + }, + "put": { + "operationId": "updateAdminNotificationServerChannel", "parameters": [ { "description": "Optional. When present, it must name the authenticated account's primary profile; an absent header is accepted.", @@ -75397,26 +75510,30 @@ } }, { - "description": "Person identifier", "in": "path", "name": "id", "required": true, "schema": { - "description": "Person identifier", - "examples": [ - "7" - ], - "minLength": 1, "type": "string" } } ], + "requestBody": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/NotificationServerChannelUpdateInputBody" + } + } + }, + "required": true + }, "responses": { "200": { "content": { "application/json": { "schema": { - "$ref": "#/components/schemas/Person" + "$ref": "#/components/schemas/NotificationServerChannel" } } }, @@ -75472,6 +75589,36 @@ }, "description": "Not Acceptable" }, + "408": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Request Timeout" + }, + "413": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Request Entity Too Large" + }, + "415": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Unsupported Media Type" + }, "422": { "content": { "application/problem+json": { @@ -75518,9 +75665,9 @@ "bearerAuth": [] } ], - "summary": "Wait for a provider refresh and return the updated person.", + "summary": "Update channel configuration and atomically reset dispatch state on URL replacement or re-enabling. Never replay an uncertain update.", "tags": [ - "admin-catalog" + "admin" ], "x-silo-class": "acting_admin", "x-silo-demo-restricted": true, @@ -75528,9 +75675,9 @@ "x-silo-service-backed": true } }, - "/api/v2/admin/playback-history": { - "get": { - "operationId": "listAdminPlaybackHistory", + "/api/v2/admin/notifications/server-channels/{id}/rotate-secret": { + "post": { + "operationId": "rotateAdminNotificationServerChannelSecret", "parameters": [ { "description": "Optional. When present, it must name the authenticated account's primary profile; an absent header is accepted.", @@ -75555,82 +75702,10 @@ } }, { - "description": "Page size; default 50, maximum 200", - "explode": false, - "in": "query", - "name": "limit", - "schema": { - "default": 50, - "description": "Page size; default 50, maximum 200", - "examples": [ - 50 - ], - "format": "int64", - "maximum": 200, - "minimum": 1, - "type": "integer" - } - }, - { - "description": "Opaque cursor from page.next_cursor", - "explode": false, - "in": "query", - "name": "cursor", - "schema": { - "description": "Opaque cursor from page.next_cursor", - "maxLength": 8192, - "type": "string" - } - }, - { - "description": "Only attempts by this login account", - "explode": false, - "in": "query", - "name": "user_id", - "schema": { - "description": "Only attempts by this login account", - "examples": [ - "1" - ], - "minLength": 1, - "type": "string" - } - }, - { - "description": "Only attempts by this household profile", - "explode": false, - "in": "query", - "name": "profile_id", - "schema": { - "description": "Only attempts by this household profile", - "maxLength": 1024, - "type": "string" - } - }, - { - "description": "Only attempts of this catalog item", - "explode": false, - "in": "query", - "name": "media_item_id", - "schema": { - "description": "Only attempts of this catalog item", - "maxLength": 1024, - "type": "string" - } - }, - { - "description": "Completion filter; all returns every finalized attempt", - "explode": false, - "in": "query", - "name": "completed", + "in": "path", + "name": "id", + "required": true, "schema": { - "default": "all", - "description": "Completion filter; all returns every finalized attempt", - "enum": [ - "all", - "true", - "false" - ], "type": "string" } } @@ -75640,11 +75715,18 @@ "content": { "application/json": { "schema": { - "$ref": "#/components/schemas/AdminPlaybackHistoryCollection" + "$ref": "#/components/schemas/NotificationWebhookSecretOutputBody" } } }, - "description": "OK" + "description": "OK", + "headers": { + "Cache-Control": { + "schema": { + "type": "string" + } + } + } }, "400": { "content": { @@ -75742,26 +75824,764 @@ "bearerAuth": [] } ], - "summary": "List finalized playback attempts across every account and profile, newest ended first. Each page is one consistent read; later pages read the live log.", + "summary": "Replace the generic server channel signing secret and reveal it once. Never replay an uncertain rotation.", "tags": [ "admin" ], "x-silo-class": "acting_admin", + "x-silo-demo-restricted": true, + "x-silo-retry-safety": "non_retryable", "x-silo-service-backed": true } }, - "/api/v2/admin/playback-routing/capabilities": { - "get": { - "operationId": "getAdminPlaybackRoutingCapabilities", + "/api/v2/admin/notifications/server-channels/{id}/test": { + "post": { + "operationId": "testAdminNotificationServerChannel", "parameters": [ - { - "description": "Optional first precondition, evaluated before If-None-Match: a tag that does not match the current representation is 412 precondition_failed.", - "in": "header", - "name": "If-Match", - "schema": { - "type": "string" - } - }, + { + "description": "Optional. When present, it must name the authenticated account's primary profile; an absent header is accepted.", + "in": "header", + "name": "X-Profile-Id", + "schema": { + "examples": [ + "1" + ], + "type": "string" + } + }, + { + "description": "Verification proof for a PIN-locked profile, issued by POST /api/v2/profiles/{id}/verify-pin; required only when the declared profile is locked", + "in": "header", + "name": "X-Profile-Token", + "schema": { + "examples": [ + "pvt_5f3a9c1e7b2d4e8fa0c6" + ], + "type": "string" + } + }, + { + "description": "Opaque identifier", + "in": "path", + "name": "id", + "required": true, + "schema": { + "description": "Opaque identifier", + "examples": [ + "1" + ], + "minLength": 1, + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/NotificationDestinationTestResult" + } + } + }, + "description": "OK" + }, + "400": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Bad Request" + }, + "401": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Unauthorized" + }, + "403": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Forbidden" + }, + "404": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Not Found" + }, + "406": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Not Acceptable" + }, + "422": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Unprocessable Entity" + }, + "429": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Too Many Requests" + }, + "500": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Internal Server Error" + }, + "503": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Service Unavailable" + } + }, + "security": [ + { + "bearerAuth": [] + } + ], + "summary": "Send one synchronous sample to a server notification channel.", + "tags": [ + "admin" + ], + "x-silo-class": "acting_admin", + "x-silo-demo-restricted": true, + "x-silo-retry-safety": "non_retryable", + "x-silo-service-backed": true + } + }, + "/api/v2/admin/people/{id}": { + "patch": { + "operationId": "updateAdminPerson", + "parameters": [ + { + "description": "Optional. When present, it must name the authenticated account's primary profile; an absent header is accepted.", + "in": "header", + "name": "X-Profile-Id", + "schema": { + "examples": [ + "1" + ], + "type": "string" + } + }, + { + "description": "Verification proof for a PIN-locked profile, issued by POST /api/v2/profiles/{id}/verify-pin; required only when the declared profile is locked", + "in": "header", + "name": "X-Profile-Token", + "schema": { + "examples": [ + "pvt_5f3a9c1e7b2d4e8fa0c6" + ], + "type": "string" + } + }, + { + "description": "Opaque identifier", + "in": "path", + "name": "id", + "required": true, + "schema": { + "description": "Opaque identifier", + "examples": [ + "1" + ], + "minLength": 1, + "type": "string" + } + } + ], + "requestBody": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/AdminPersonUpdate" + } + } + }, + "required": true + }, + "responses": { + "200": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/Person" + } + } + }, + "description": "OK" + }, + "400": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Bad Request" + }, + "401": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Unauthorized" + }, + "403": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Forbidden" + }, + "404": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Not Found" + }, + "406": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Not Acceptable" + }, + "408": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Request Timeout" + }, + "413": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Request Entity Too Large" + }, + "415": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Unsupported Media Type" + }, + "422": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Unprocessable Entity" + }, + "429": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Too Many Requests" + }, + "500": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Internal Server Error" + }, + "503": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Service Unavailable" + } + }, + "security": [ + { + "bearerAuth": [] + } + ], + "summary": "Apply a partial person metadata update.", + "tags": [ + "admin-catalog" + ], + "x-silo-class": "acting_admin", + "x-silo-demo-restricted": true, + "x-silo-retry-safety": "non_retryable", + "x-silo-service-backed": true + } + }, + "/api/v2/admin/people/{id}/refresh": { + "post": { + "operationId": "refreshAdminPerson", + "parameters": [ + { + "description": "Optional. When present, it must name the authenticated account's primary profile; an absent header is accepted.", + "in": "header", + "name": "X-Profile-Id", + "schema": { + "examples": [ + "1" + ], + "type": "string" + } + }, + { + "description": "Verification proof for a PIN-locked profile, issued by POST /api/v2/profiles/{id}/verify-pin; required only when the declared profile is locked", + "in": "header", + "name": "X-Profile-Token", + "schema": { + "examples": [ + "pvt_5f3a9c1e7b2d4e8fa0c6" + ], + "type": "string" + } + }, + { + "description": "Person identifier", + "in": "path", + "name": "id", + "required": true, + "schema": { + "description": "Person identifier", + "examples": [ + "7" + ], + "minLength": 1, + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/Person" + } + } + }, + "description": "OK" + }, + "400": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Bad Request" + }, + "401": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Unauthorized" + }, + "403": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Forbidden" + }, + "404": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Not Found" + }, + "406": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Not Acceptable" + }, + "422": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Unprocessable Entity" + }, + "429": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Too Many Requests" + }, + "500": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Internal Server Error" + }, + "503": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Service Unavailable" + } + }, + "security": [ + { + "bearerAuth": [] + } + ], + "summary": "Wait for a provider refresh and return the updated person.", + "tags": [ + "admin-catalog" + ], + "x-silo-class": "acting_admin", + "x-silo-demo-restricted": true, + "x-silo-retry-safety": "non_retryable", + "x-silo-service-backed": true + } + }, + "/api/v2/admin/playback-history": { + "get": { + "operationId": "listAdminPlaybackHistory", + "parameters": [ + { + "description": "Optional. When present, it must name the authenticated account's primary profile; an absent header is accepted.", + "in": "header", + "name": "X-Profile-Id", + "schema": { + "examples": [ + "1" + ], + "type": "string" + } + }, + { + "description": "Verification proof for a PIN-locked profile, issued by POST /api/v2/profiles/{id}/verify-pin; required only when the declared profile is locked", + "in": "header", + "name": "X-Profile-Token", + "schema": { + "examples": [ + "pvt_5f3a9c1e7b2d4e8fa0c6" + ], + "type": "string" + } + }, + { + "description": "Page size; default 50, maximum 200", + "explode": false, + "in": "query", + "name": "limit", + "schema": { + "default": 50, + "description": "Page size; default 50, maximum 200", + "examples": [ + 50 + ], + "format": "int64", + "maximum": 200, + "minimum": 1, + "type": "integer" + } + }, + { + "description": "Opaque cursor from page.next_cursor", + "explode": false, + "in": "query", + "name": "cursor", + "schema": { + "description": "Opaque cursor from page.next_cursor", + "maxLength": 8192, + "type": "string" + } + }, + { + "description": "Only attempts by this login account", + "explode": false, + "in": "query", + "name": "user_id", + "schema": { + "description": "Only attempts by this login account", + "examples": [ + "1" + ], + "minLength": 1, + "type": "string" + } + }, + { + "description": "Only attempts by this household profile", + "explode": false, + "in": "query", + "name": "profile_id", + "schema": { + "description": "Only attempts by this household profile", + "maxLength": 1024, + "type": "string" + } + }, + { + "description": "Only attempts of this catalog item", + "explode": false, + "in": "query", + "name": "media_item_id", + "schema": { + "description": "Only attempts of this catalog item", + "maxLength": 1024, + "type": "string" + } + }, + { + "description": "Completion filter; all returns every finalized attempt", + "explode": false, + "in": "query", + "name": "completed", + "schema": { + "default": "all", + "description": "Completion filter; all returns every finalized attempt", + "enum": [ + "all", + "true", + "false" + ], + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/AdminPlaybackHistoryCollection" + } + } + }, + "description": "OK" + }, + "400": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Bad Request" + }, + "401": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Unauthorized" + }, + "403": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Forbidden" + }, + "404": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Not Found" + }, + "406": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Not Acceptable" + }, + "422": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Unprocessable Entity" + }, + "429": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Too Many Requests" + }, + "500": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Internal Server Error" + }, + "503": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Service Unavailable" + } + }, + "security": [ + { + "bearerAuth": [] + } + ], + "summary": "List finalized playback attempts across every account and profile, newest ended first. Each page is one consistent read; later pages read the live log.", + "tags": [ + "admin" + ], + "x-silo-class": "acting_admin", + "x-silo-service-backed": true + } + }, + "/api/v2/admin/playback-routing/capabilities": { + "get": { + "operationId": "getAdminPlaybackRoutingCapabilities", + "parameters": [ + { + "description": "Optional first precondition, evaluated before If-None-Match: a tag that does not match the current representation is 412 precondition_failed.", + "in": "header", + "name": "If-Match", + "schema": { + "type": "string" + } + }, { "description": "Optional. When present, it must name the authenticated account's primary profile; an absent header is accepted.", "in": "header", @@ -76433,54 +77253,209 @@ }, "description": "Not Acceptable" }, - "408": { - "content": { - "application/problem+json": { - "schema": { - "$ref": "#/components/schemas/Problem" - } - } - }, - "description": "Request Timeout" - }, - "412": { - "content": { - "application/problem+json": { - "schema": { - "$ref": "#/components/schemas/Problem" - } - } - }, - "description": "Precondition Failed", - "headers": { - "ETag": { - "description": "The strong, opaque validator of the representation; send it back in If-Match on a guarded mutation or If-None-Match on a conditional read.", - "schema": { - "type": "string" - } - } - } - }, - "413": { - "content": { - "application/problem+json": { - "schema": { - "$ref": "#/components/schemas/Problem" - } - } - }, - "description": "Request Entity Too Large" - }, - "415": { - "content": { - "application/problem+json": { - "schema": { - "$ref": "#/components/schemas/Problem" - } - } - }, - "description": "Unsupported Media Type" - }, + "408": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Request Timeout" + }, + "412": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Precondition Failed", + "headers": { + "ETag": { + "description": "The strong, opaque validator of the representation; send it back in If-Match on a guarded mutation or If-None-Match on a conditional read.", + "schema": { + "type": "string" + } + } + } + }, + "413": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Request Entity Too Large" + }, + "415": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Unsupported Media Type" + }, + "422": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Unprocessable Entity" + }, + "428": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Precondition Required" + }, + "429": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Too Many Requests" + }, + "500": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Internal Server Error" + }, + "503": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Service Unavailable" + } + }, + "security": [ + { + "bearerAuth": [] + } + ], + "summary": "Atomically replace captured catalog configuration and reconcile managed repositories.", + "tags": [ + "admin-plugins" + ], + "x-silo-class": "acting_admin", + "x-silo-demo-restricted": true, + "x-silo-guarded": true, + "x-silo-retry-safety": "natural_idempotent", + "x-silo-service-backed": true + } + }, + "/api/v2/admin/plugins/catalog-status": { + "get": { + "operationId": "getAdminPluginCatalogStatus", + "parameters": [ + { + "description": "Optional. When present, it must name the authenticated account's primary profile; an absent header is accepted.", + "in": "header", + "name": "X-Profile-Id", + "schema": { + "examples": [ + "1" + ], + "type": "string" + } + }, + { + "description": "Verification proof for a PIN-locked profile, issued by POST /api/v2/profiles/{id}/verify-pin; required only when the declared profile is locked", + "in": "header", + "name": "X-Profile-Token", + "schema": { + "examples": [ + "pvt_5f3a9c1e7b2d4e8fa0c6" + ], + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/AdminPluginCatalogStatus" + } + } + }, + "description": "OK" + }, + "400": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Bad Request" + }, + "401": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Unauthorized" + }, + "403": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Forbidden" + }, + "404": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Not Found" + }, + "406": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Not Acceptable" + }, "422": { "content": { "application/problem+json": { @@ -76491,16 +77466,6 @@ }, "description": "Unprocessable Entity" }, - "428": { - "content": { - "application/problem+json": { - "schema": { - "$ref": "#/components/schemas/Problem" - } - } - }, - "description": "Precondition Required" - }, "429": { "content": { "application/problem+json": { @@ -76537,20 +77502,17 @@ "bearerAuth": [] } ], - "summary": "Atomically replace captured catalog configuration and reconcile managed repositories.", + "summary": "Read plugin catalog counts and update availability separately from editable configuration.", "tags": [ "admin-plugins" ], "x-silo-class": "acting_admin", - "x-silo-demo-restricted": true, - "x-silo-guarded": true, - "x-silo-retry-safety": "natural_idempotent", "x-silo-service-backed": true } }, - "/api/v2/admin/plugins/catalog-status": { + "/api/v2/admin/plugins/installations": { "get": { - "operationId": "getAdminPluginCatalogStatus", + "operationId": "listAdminPluginInstallations", "parameters": [ { "description": "Optional. When present, it must name the authenticated account's primary profile; an absent header is accepted.", @@ -76573,6 +77535,32 @@ ], "type": "string" } + }, + { + "description": "Page size; default 50, maximum 200", + "explode": false, + "in": "query", + "name": "limit", + "schema": { + "default": 50, + "description": "Page size; default 50, maximum 200", + "examples": [ + 50 + ], + "format": "int64", + "maximum": 200, + "minimum": 1, + "type": "integer" + } + }, + { + "explode": false, + "in": "query", + "name": "cursor", + "schema": { + "maxLength": 8192, + "type": "string" + } } ], "responses": { @@ -76580,7 +77568,7 @@ "content": { "application/json": { "schema": { - "$ref": "#/components/schemas/AdminPluginCatalogStatus" + "$ref": "#/components/schemas/CollectionAdminPluginInstallation" } } }, @@ -76682,17 +77670,15 @@ "bearerAuth": [] } ], - "summary": "Read plugin catalog counts and update availability separately from editable configuration.", + "summary": "Read manageable installations with manifest surface, redacted global configuration and bindings. The reserved builtin row is excluded. Every page enumerates the full stored list; continuation is live, not a snapshot.", "tags": [ "admin-plugins" ], "x-silo-class": "acting_admin", "x-silo-service-backed": true - } - }, - "/api/v2/admin/plugins/installations": { - "get": { - "operationId": "listAdminPluginInstallations", + }, + "post": { + "operationId": "createAdminPluginInstallation", "parameters": [ { "description": "Optional. When present, it must name the authenticated account's primary profile; an absent header is accepted.", @@ -76715,44 +77701,28 @@ ], "type": "string" } - }, - { - "description": "Page size; default 50, maximum 200", - "explode": false, - "in": "query", - "name": "limit", - "schema": { - "default": 50, - "description": "Page size; default 50, maximum 200", - "examples": [ - 50 - ], - "format": "int64", - "maximum": 200, - "minimum": 1, - "type": "integer" - } - }, - { - "explode": false, - "in": "query", - "name": "cursor", - "schema": { - "maxLength": 8192, - "type": "string" - } } ], + "requestBody": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/AdminPluginInstallCreate" + } + } + }, + "required": true + }, "responses": { - "200": { + "201": { "content": { "application/json": { "schema": { - "$ref": "#/components/schemas/CollectionAdminPluginInstallation" + "$ref": "#/components/schemas/AdminPluginInstallation" } } }, - "description": "OK" + "description": "Created" }, "400": { "content": { @@ -76804,6 +77774,36 @@ }, "description": "Not Acceptable" }, + "408": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Request Timeout" + }, + "413": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Request Entity Too Large" + }, + "415": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Unsupported Media Type" + }, "422": { "content": { "application/problem+json": { @@ -76850,15 +77850,19 @@ "bearerAuth": [] } ], - "summary": "Read manageable installations with manifest surface, redacted global configuration and bindings. The reserved builtin row is excluded. Every page enumerates the full stored list; continuation is live, not a snapshot.", + "summary": "Install a plugin from a catalog repository (repository_id, plugin_id, version) or a direct archive URL, fetching over the network. An installation with the same plugin_id is stopped and replaced rather than duplicated. There is no replay identity: a lost response may follow a committed install; reconcile from the installation list and never automatically retry.", "tags": [ "admin-plugins" ], "x-silo-class": "acting_admin", + "x-silo-demo-restricted": true, + "x-silo-retry-safety": "non_retryable", "x-silo-service-backed": true - }, - "post": { - "operationId": "createAdminPluginInstallation", + } + }, + "/api/v2/admin/plugins/installations/{id}": { + "delete": { + "operationId": "deleteAdminPluginInstallation", "parameters": [ { "description": "Optional. When present, it must name the authenticated account's primary profile; an absent header is accepted.", @@ -76881,28 +77885,26 @@ ], "type": "string" } + }, + { + "description": "Opaque identifier", + "in": "path", + "name": "id", + "required": true, + "schema": { + "description": "Opaque identifier", + "examples": [ + "1" + ], + "minLength": 1, + "pattern": "^[1-9][0-9]*$", + "type": "string" + } } ], - "requestBody": { - "content": { - "application/json": { - "schema": { - "$ref": "#/components/schemas/AdminPluginInstallCreate" - } - } - }, - "required": true - }, "responses": { - "201": { - "content": { - "application/json": { - "schema": { - "$ref": "#/components/schemas/AdminPluginInstallation" - } - } - }, - "description": "Created" + "204": { + "description": "No Content" }, "400": { "content": { @@ -76954,27 +77956,7 @@ }, "description": "Not Acceptable" }, - "408": { - "content": { - "application/problem+json": { - "schema": { - "$ref": "#/components/schemas/Problem" - } - } - }, - "description": "Request Timeout" - }, - "413": { - "content": { - "application/problem+json": { - "schema": { - "$ref": "#/components/schemas/Problem" - } - } - }, - "description": "Request Entity Too Large" - }, - "415": { + "409": { "content": { "application/problem+json": { "schema": { @@ -76982,7 +77964,7 @@ } } }, - "description": "Unsupported Media Type" + "description": "Conflict" }, "422": { "content": { @@ -77030,7 +78012,7 @@ "bearerAuth": [] } ], - "summary": "Install a plugin from a catalog repository (repository_id, plugin_id, version) or a direct archive URL, fetching over the network. An installation with the same plugin_id is stopped and replaced rather than duplicated. There is no replay identity: a lost response may follow a committed install; reconcile from the installation list and never automatically retry.", + "summary": "Stop the plugin, delete the installation row (configuration, bindings and archives cascade) and remove its files. Row delete and file removal are not one transaction and a repeat finds no row: a later 404 is not this caller's receipt; never automatically retry.", "tags": [ "admin-plugins" ], @@ -77038,11 +78020,9 @@ "x-silo-demo-restricted": true, "x-silo-retry-safety": "non_retryable", "x-silo-service-backed": true - } - }, - "/api/v2/admin/plugins/installations/{id}": { - "delete": { - "operationId": "deleteAdminPluginInstallation", + }, + "put": { + "operationId": "updateAdminPluginInstallation", "parameters": [ { "description": "Optional. When present, it must name the authenticated account's primary profile; an absent header is accepted.", @@ -77082,9 +78062,26 @@ } } ], + "requestBody": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/AdminPluginInstallationUpdate" + } + } + }, + "required": true + }, "responses": { - "204": { - "description": "No Content" + "200": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/AdminPluginInstallation" + } + } + }, + "description": "OK" }, "400": { "content": { @@ -77136,6 +78133,16 @@ }, "description": "Not Acceptable" }, + "408": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Request Timeout" + }, "409": { "content": { "application/problem+json": { @@ -77146,6 +78153,26 @@ }, "description": "Conflict" }, + "413": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Request Entity Too Large" + }, + "415": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Unsupported Media Type" + }, "422": { "content": { "application/problem+json": { @@ -77192,17 +78219,19 @@ "bearerAuth": [] } ], - "summary": "Stop the plugin, delete the installation row (configuration, bindings and archives cascade) and remove its files. Row delete and file removal are not one transaction and a repeat finds no row: a later 404 is not this caller's receipt; never automatically retry.", + "summary": "Assign enabled and/or update_policy on one installation. Disabling stops the running plugin first. Repeating the same assignment converges on one stored row.", "tags": [ "admin-plugins" ], "x-silo-class": "acting_admin", "x-silo-demo-restricted": true, - "x-silo-retry-safety": "non_retryable", + "x-silo-retry-safety": "natural_idempotent", "x-silo-service-backed": true - }, + } + }, + "/api/v2/admin/plugins/installations/{id}/auth-binding": { "put": { - "operationId": "updateAdminPluginInstallation", + "operationId": "updateAdminPluginAuthBinding", "parameters": [ { "description": "Optional. When present, it must name the authenticated account's primary profile; an absent header is accepted.", @@ -77246,22 +78275,23 @@ "content": { "application/json": { "schema": { - "$ref": "#/components/schemas/AdminPluginInstallationUpdate" + "$ref": "#/components/schemas/AdminPluginAuthBindingWrite" } } }, "required": true }, "responses": { - "200": { - "content": { - "application/json": { + "204": { + "description": "No Content", + "headers": { + "X-Silo-Restart-Required": { "schema": { - "$ref": "#/components/schemas/AdminPluginInstallation" + "description": "Always true: bindings load at server start", + "type": "string" } } - }, - "description": "OK" + } }, "400": { "content": { @@ -77399,7 +78429,7 @@ "bearerAuth": [] } ], - "summary": "Assign enabled and/or update_policy on one installation. Disabling stops the running plugin first. Repeating the same assignment converges on one stored row.", + "summary": "Replace the auth provider binding for one capability and mark a server restart required. The whole row is assigned, so repeating the request converges on one stored binding.", "tags": [ "admin-plugins" ], @@ -77409,9 +78439,9 @@ "x-silo-service-backed": true } }, - "/api/v2/admin/plugins/installations/{id}/auth-binding": { + "/api/v2/admin/plugins/installations/{id}/config": { "put": { - "operationId": "updateAdminPluginAuthBinding", + "operationId": "updateAdminPluginInstallationConfig", "parameters": [ { "description": "Optional. When present, it must name the authenticated account's primary profile; an absent header is accepted.", @@ -77455,7 +78485,7 @@ "content": { "application/json": { "schema": { - "$ref": "#/components/schemas/AdminPluginAuthBindingWrite" + "$ref": "#/components/schemas/AdminPluginConfigWrite" } } }, @@ -77463,15 +78493,7 @@ }, "responses": { "204": { - "description": "No Content", - "headers": { - "X-Silo-Restart-Required": { - "schema": { - "description": "Always true: bindings load at server start", - "type": "string" - } - } - } + "description": "No Content" }, "400": { "content": { @@ -77609,7 +78631,7 @@ "bearerAuth": [] } ], - "summary": "Replace the auth provider binding for one capability and mark a server restart required. The whole row is assigned, so repeating the request converges on one stored binding.", + "summary": "Replace one global configuration entry after validating it against the plugin manifest, then stop the running plugin so it rebinds. Blank secret fields keep stored secrets; clear_secrets removes them. Repeating the same request converges on one stored entry.", "tags": [ "admin-plugins" ], @@ -77619,9 +78641,9 @@ "x-silo-service-backed": true } }, - "/api/v2/admin/plugins/installations/{id}/config": { - "put": { - "operationId": "updateAdminPluginInstallationConfig", + "/api/v2/admin/plugins/installations/{id}/config/test": { + "post": { + "operationId": "testAdminPluginInstallationConfig", "parameters": [ { "description": "Optional. When present, it must name the authenticated account's primary profile; an absent header is accepted.", @@ -77672,8 +78694,15 @@ "required": true }, "responses": { - "204": { - "description": "No Content" + "200": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/AdminPluginConnectionCheck" + } + } + }, + "description": "OK" }, "400": { "content": { @@ -77811,19 +78840,19 @@ "bearerAuth": [] } ], - "summary": "Replace one global configuration entry after validating it against the plugin manifest, then stop the running plugin so it rebinds. Blank secret fields keep stored secrets; clear_secrets removes them. Repeating the same request converges on one stored entry.", + "summary": "Probe one prospective configuration by starting a temporary plugin instance and running its connection check; nothing is stored. The check calls the plugin's provider and is bounded by a server timeout; never automatically retry an uncertain result. A failed check is a 200 result with success false.", "tags": [ "admin-plugins" ], "x-silo-class": "acting_admin", "x-silo-demo-restricted": true, - "x-silo-retry-safety": "natural_idempotent", + "x-silo-retry-safety": "non_retryable", "x-silo-service-backed": true } }, - "/api/v2/admin/plugins/installations/{id}/config/test": { + "/api/v2/admin/plugins/installations/{id}/restart": { "post": { - "operationId": "testAdminPluginInstallationConfig", + "operationId": "restartAdminPluginInstallation", "parameters": [ { "description": "Optional. When present, it must name the authenticated account's primary profile; an absent header is accepted.", @@ -77863,22 +78892,12 @@ } } ], - "requestBody": { - "content": { - "application/json": { - "schema": { - "$ref": "#/components/schemas/AdminPluginConfigWrite" - } - } - }, - "required": true - }, "responses": { "200": { "content": { "application/json": { "schema": { - "$ref": "#/components/schemas/AdminPluginConnectionCheck" + "$ref": "#/components/schemas/AdminPluginInstallation" } } }, @@ -77934,16 +78953,6 @@ }, "description": "Not Acceptable" }, - "408": { - "content": { - "application/problem+json": { - "schema": { - "$ref": "#/components/schemas/Problem" - } - } - }, - "description": "Request Timeout" - }, "409": { "content": { "application/problem+json": { @@ -77954,26 +78963,6 @@ }, "description": "Conflict" }, - "413": { - "content": { - "application/problem+json": { - "schema": { - "$ref": "#/components/schemas/Problem" - } - } - }, - "description": "Request Entity Too Large" - }, - "415": { - "content": { - "application/problem+json": { - "schema": { - "$ref": "#/components/schemas/Problem" - } - } - }, - "description": "Unsupported Media Type" - }, "422": { "content": { "application/problem+json": { @@ -78020,13 +79009,13 @@ "bearerAuth": [] } ], - "summary": "Probe one prospective configuration by starting a temporary plugin instance and running its connection check; nothing is stored. The check calls the plugin's provider and is bounded by a server timeout; never automatically retry an uncertain result. A failed check is a 200 result with success false.", + "summary": "Stop the installation's process and, for a resident plugin (one the server supervises, such as a network access provider), start it again with a fresh failure budget; the response's runtime reports the outcome, including a launch that failed. A non-resident plugin is only stopped and launches on its next use. A disabled installation is 409. Repeating the request converges on one running process.", "tags": [ "admin-plugins" ], "x-silo-class": "acting_admin", "x-silo-demo-restricted": true, - "x-silo-retry-safety": "non_retryable", + "x-silo-retry-safety": "natural_idempotent", "x-silo-service-backed": true } }, @@ -138974,6 +139963,164 @@ "x-silo-service-backed": true } }, + "/api/v2/network-access/capabilities": { + "get": { + "operationId": "getNetworkAccessCapabilities", + "parameters": [ + { + "description": "Optional first precondition, evaluated before If-None-Match: a tag that does not match the current representation is 412 precondition_failed.", + "in": "header", + "name": "If-Match", + "schema": { + "type": "string" + } + }, + { + "in": "header", + "name": "If-None-Match", + "schema": { + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/NetworkAccessCapabilities" + } + } + }, + "description": "OK", + "headers": { + "Cache-Control": { + "schema": { + "type": "string" + } + }, + "ETag": { + "description": "The strong, opaque validator of the representation; send it back in If-Match on a guarded mutation or If-None-Match on a conditional read.", + "schema": { + "type": "string" + } + } + } + }, + "304": { + "description": "The representation named by If-None-Match is current; no body.", + "headers": { + "ETag": { + "description": "The strong, opaque validator of the representation; send it back in If-Match on a guarded mutation or If-None-Match on a conditional read.", + "schema": { + "type": "string" + } + } + } + }, + "400": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Bad Request" + }, + "401": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Unauthorized" + }, + "406": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Not Acceptable" + }, + "412": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Precondition Failed", + "headers": { + "ETag": { + "description": "The strong, opaque validator of the representation; send it back in If-Match on a guarded mutation or If-None-Match on a conditional read.", + "schema": { + "type": "string" + } + } + } + }, + "422": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Unprocessable Entity" + }, + "429": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Too Many Requests" + }, + "500": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Internal Server Error" + }, + "503": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/Problem" + } + } + }, + "description": "Service Unavailable" + } + }, + "security": [ + { + "bearerAuth": [] + } + ], + "summary": "List the installed network access providers (overlay networks such as Tailscale that reach this server without port forwarding). available when an enabled plugin declares one; not_configured otherwise. Read from plugin manifests; no plugin is launched and no health is implied.", + "tags": [ + "network-access" + ], + "x-silo-class": "authenticated", + "x-silo-conditional": true, + "x-silo-service-backed": true + } + }, "/api/v2/notifications": { "get": { "operationId": "listNotifications", @@ -180224,6 +181371,271 @@ } } } + }, + { + "description": "Live status of every network access provider plugin instance running on this proxy, read from the plugin with a ten-second timeout each. Carries auth_url and error because the caller holds the node bearer. 503 when the proxy hosts no plugins.", + "listener": "proxy", + "method": "GET", + "path": "/network-access/status", + "handler": "(*internal/proxy.Server).handleNetworkAccessStatus", + "auth_class": "node_bearer", + "responses": { + "200": { + "content": { + "application/json": { + "schema": { + "$ref": "#/x-silo-worker-protocols/schemas/netaccess_HostStatusReport" + } + } + }, + "description": "OK" + }, + "401": { + "content": { + "text/plain": { + "schema": { + "type": "string" + } + } + }, + "description": "Unauthorized" + }, + "500": { + "content": { + "text/plain": { + "schema": { + "type": "string" + } + } + }, + "description": "Internal Server Error" + }, + "503": { + "content": { + "text/plain": { + "schema": { + "type": "string" + } + } + }, + "description": "Service Unavailable" + } + } + }, + { + "parameters": [ + { + "description": "Provider slug declared by the plugin manifest, e.g. tailscale.", + "in": "path", + "name": "provider", + "required": true, + "schema": { + "type": "string" + } + } + ], + "description": "Live status of the named network access provider on this proxy, with a ten-second timeout. Reads only this provider, so another provider's timeout does not hide its status. 404 for a provider no enabled installation declares; 503 when the proxy hosts no plugins.", + "listener": "proxy", + "method": "GET", + "path": "/network-access/{provider}/status", + "handler": "(*internal/proxy.Server).handleNetworkAccessProviderStatus", + "auth_class": "node_bearer", + "responses": { + "200": { + "content": { + "application/json": { + "schema": { + "$ref": "#/x-silo-worker-protocols/schemas/netaccess_Status" + } + } + }, + "description": "OK" + }, + "401": { + "content": { + "text/plain": { + "schema": { + "type": "string" + } + } + }, + "description": "Unauthorized" + }, + "404": { + "content": { + "text/plain": { + "schema": { + "type": "string" + } + } + }, + "description": "Not Found" + }, + "500": { + "content": { + "text/plain": { + "schema": { + "type": "string" + } + } + }, + "description": "Internal Server Error" + }, + "503": { + "content": { + "text/plain": { + "schema": { + "type": "string" + } + } + }, + "description": "Service Unavailable" + } + } + }, + { + "parameters": [ + { + "description": "Provider slug declared by the plugin manifest, e.g. tailscale.", + "in": "path", + "name": "provider", + "required": true, + "schema": { + "type": "string" + } + } + ], + "description": "Ask the provider on this proxy to bring its overlay identity up; answers the state reached within ten seconds. Repeating converges on one connected instance. 404 for a provider no enabled installation declares.", + "retry_safety": "natural_idempotent", + "listener": "proxy", + "method": "POST", + "path": "/network-access/{provider}/connect", + "handler": "(*internal/proxy.Server).handleNetworkAccessConnect", + "auth_class": "node_bearer", + "responses": { + "200": { + "content": { + "application/json": { + "schema": { + "$ref": "#/x-silo-worker-protocols/schemas/netaccess_Status" + } + } + }, + "description": "OK" + }, + "401": { + "content": { + "text/plain": { + "schema": { + "type": "string" + } + } + }, + "description": "Unauthorized" + }, + "404": { + "content": { + "text/plain": { + "schema": { + "type": "string" + } + } + }, + "description": "Not Found" + }, + "500": { + "content": { + "text/plain": { + "schema": { + "type": "string" + } + } + }, + "description": "Internal Server Error" + }, + "503": { + "content": { + "text/plain": { + "schema": { + "type": "string" + } + } + }, + "description": "Service Unavailable" + } + } + }, + { + "parameters": [ + { + "description": "Provider slug declared by the plugin manifest, e.g. tailscale.", + "in": "path", + "name": "provider", + "required": true, + "schema": { + "type": "string" + } + } + ], + "description": "Ask the provider on this proxy to tear its overlay listener down; answers the state reached within ten seconds. Repeating converges on disconnected. 404 for a provider no enabled installation declares.", + "retry_safety": "natural_idempotent", + "listener": "proxy", + "method": "POST", + "path": "/network-access/{provider}/disconnect", + "handler": "(*internal/proxy.Server).handleNetworkAccessDisconnect", + "auth_class": "node_bearer", + "responses": { + "200": { + "content": { + "application/json": { + "schema": { + "$ref": "#/x-silo-worker-protocols/schemas/netaccess_Status" + } + } + }, + "description": "OK" + }, + "401": { + "content": { + "text/plain": { + "schema": { + "type": "string" + } + } + }, + "description": "Unauthorized" + }, + "404": { + "content": { + "text/plain": { + "schema": { + "type": "string" + } + } + }, + "description": "Not Found" + }, + "500": { + "content": { + "text/plain": { + "schema": { + "type": "string" + } + } + }, + "description": "Internal Server Error" + }, + "503": { + "content": { + "text/plain": { + "schema": { + "type": "string" + } + } + }, + "description": "Service Unavailable" + } + } } ], "schemas": { @@ -180389,6 +181801,92 @@ ], "type": "object" }, + "netaccess_HostStatusReport": { + "additionalProperties": false, + "properties": { + "providers": { + "items": { + "$ref": "#/x-silo-worker-protocols/schemas/netaccess_Status" + }, + "type": "array" + } + }, + "required": [ + "providers" + ], + "type": "object" + }, + "netaccess_Listener": { + "additionalProperties": false, + "properties": { + "name": { + "type": "string" + }, + "origin": { + "type": "string" + } + }, + "required": [ + "name", + "origin" + ], + "type": "object" + }, + "netaccess_Status": { + "additionalProperties": false, + "properties": { + "addresses": { + "items": { + "type": "string" + }, + "type": "array" + }, + "auth_url": { + "type": "string" + }, + "desired_connected": { + "type": "boolean" + }, + "error": { + "type": "string" + }, + "hostname": { + "type": "string" + }, + "installation_id": { + "format": "int64", + "type": "integer" + }, + "listeners": { + "items": { + "$ref": "#/x-silo-worker-protocols/schemas/netaccess_Listener" + }, + "type": "array" + }, + "origin": { + "type": "string" + }, + "provider": { + "type": "string" + }, + "provider_version": { + "type": "string" + }, + "state": { + "type": "string" + }, + "updated_at": { + "format": "date-time", + "type": "string" + } + }, + "required": [ + "installation_id", + "provider", + "state" + ], + "type": "object" + }, "nodemetrics_CgroupCPUStats": { "additionalProperties": false, "properties": { diff --git a/contracts/api/v2/route-inventory.json b/contracts/api/v2/route-inventory.json index 5ee5b919c7..1e5503e555 100644 --- a/contracts/api/v2/route-inventory.json +++ b/contracts/api/v2/route-inventory.json @@ -50,7 +50,7 @@ "id": "proxy", "entrypoint": "internal/proxy.(*Server).Handler", "description": "Proxy node listener: media relay, node control, and its own health/metrics probes.", - "route_count": 25 + "route_count": 29 }, { "id": "transcode_node", @@ -60,7 +60,7 @@ } ], "totals": { - "routes": 863, + "routes": 867, "conditional_routes": 691, "streaming_routes": 42, "websocket_routes": 4 @@ -75,6 +75,7 @@ "middleware": [ "apimw.RequestID", "clientip.Middleware(deps.ClientIPResolver) [when deps.ClientIPResolver != nil]", + "netaccess.Middleware(deps.NetworkAccess.Registry) [when deps.NetworkAccess != nil]", "apimw.RequestLogger(deps.NodeID)", "middleware.Recoverer", "apimw.Metrics", @@ -87,6 +88,7 @@ "middleware": [ "apimw.RequestID", "clientip.Middleware(deps.ClientIPResolver) [when deps.ClientIPResolver != nil]", + "netaccess.Middleware(deps.NetworkAccess.Registry) [when deps.NetworkAccess != nil]", "apimw.RequestLogger(deps.NodeID)", "middleware.Recoverer", "apimw.Metrics", @@ -100,6 +102,7 @@ "middleware": [ "apimw.RequestID", "clientip.Middleware(deps.ClientIPResolver) [when deps.ClientIPResolver != nil]", + "netaccess.Middleware(deps.NetworkAccess.Registry) [when deps.NetworkAccess != nil]", "apimw.RequestLogger(deps.NodeID)", "middleware.Recoverer", "apimw.Metrics", @@ -114,6 +117,7 @@ "middleware": [ "apimw.RequestID", "clientip.Middleware(deps.ClientIPResolver) [when deps.ClientIPResolver != nil]", + "netaccess.Middleware(deps.NetworkAccess.Registry) [when deps.NetworkAccess != nil]", "apimw.RequestLogger(deps.NodeID)", "middleware.Recoverer", "apimw.Metrics", @@ -129,6 +133,7 @@ "middleware": [ "apimw.RequestID", "clientip.Middleware(deps.ClientIPResolver) [when deps.ClientIPResolver != nil]", + "netaccess.Middleware(deps.NetworkAccess.Registry) [when deps.NetworkAccess != nil]", "apimw.RequestLogger(deps.NodeID)", "middleware.Recoverer", "apimw.Metrics", @@ -145,6 +150,7 @@ "middleware": [ "apimw.RequestID", "clientip.Middleware(deps.ClientIPResolver) [when deps.ClientIPResolver != nil]", + "netaccess.Middleware(deps.NetworkAccess.Registry) [when deps.NetworkAccess != nil]", "apimw.RequestLogger(deps.NodeID)", "middleware.Recoverer", "apimw.Metrics", @@ -162,6 +168,7 @@ "middleware": [ "apimw.RequestID", "clientip.Middleware(deps.ClientIPResolver) [when deps.ClientIPResolver != nil]", + "netaccess.Middleware(deps.NetworkAccess.Registry) [when deps.NetworkAccess != nil]", "apimw.RequestLogger(deps.NodeID)", "middleware.Recoverer", "apimw.Metrics", @@ -179,6 +186,7 @@ "middleware": [ "apimw.RequestID", "clientip.Middleware(deps.ClientIPResolver) [when deps.ClientIPResolver != nil]", + "netaccess.Middleware(deps.NetworkAccess.Registry) [when deps.NetworkAccess != nil]", "apimw.RequestLogger(deps.NodeID)", "middleware.Recoverer", "apimw.Metrics", @@ -196,6 +204,7 @@ "middleware": [ "apimw.RequestID", "clientip.Middleware(deps.ClientIPResolver) [when deps.ClientIPResolver != nil]", + "netaccess.Middleware(deps.NetworkAccess.Registry) [when deps.NetworkAccess != nil]", "apimw.RequestLogger(deps.NodeID)", "middleware.Recoverer", "apimw.Metrics", @@ -213,6 +222,7 @@ "middleware": [ "apimw.RequestID", "clientip.Middleware(deps.ClientIPResolver) [when deps.ClientIPResolver != nil]", + "netaccess.Middleware(deps.NetworkAccess.Registry) [when deps.NetworkAccess != nil]", "apimw.RequestLogger(deps.NodeID)", "middleware.Recoverer", "apimw.Metrics", @@ -227,6 +237,7 @@ "middleware": [ "apimw.RequestID", "clientip.Middleware(deps.ClientIPResolver) [when deps.ClientIPResolver != nil]", + "netaccess.Middleware(deps.NetworkAccess.Registry) [when deps.NetworkAccess != nil]", "apimw.RequestLogger(deps.NodeID)", "middleware.Recoverer", "apimw.Metrics", @@ -241,6 +252,7 @@ "middleware": [ "apimw.RequestID", "clientip.Middleware(deps.ClientIPResolver) [when deps.ClientIPResolver != nil]", + "netaccess.Middleware(deps.NetworkAccess.Registry) [when deps.NetworkAccess != nil]", "apimw.RequestLogger(deps.NodeID)", "middleware.Recoverer", "apimw.Metrics", @@ -255,6 +267,7 @@ "middleware": [ "apimw.RequestID", "clientip.Middleware(deps.ClientIPResolver) [when deps.ClientIPResolver != nil]", + "netaccess.Middleware(deps.NetworkAccess.Registry) [when deps.NetworkAccess != nil]", "apimw.RequestLogger(deps.NodeID)", "middleware.Recoverer", "apimw.Metrics", @@ -269,6 +282,7 @@ "middleware": [ "apimw.RequestID", "clientip.Middleware(deps.ClientIPResolver) [when deps.ClientIPResolver != nil]", + "netaccess.Middleware(deps.NetworkAccess.Registry) [when deps.NetworkAccess != nil]", "apimw.RequestLogger(deps.NodeID)", "middleware.Recoverer", "apimw.Metrics", @@ -283,6 +297,7 @@ "middleware": [ "apimw.RequestID", "clientip.Middleware(deps.ClientIPResolver) [when deps.ClientIPResolver != nil]", + "netaccess.Middleware(deps.NetworkAccess.Registry) [when deps.NetworkAccess != nil]", "apimw.RequestLogger(deps.NodeID)", "middleware.Recoverer", "apimw.Metrics", @@ -296,6 +311,7 @@ "middleware": [ "apimw.RequestID", "clientip.Middleware(deps.ClientIPResolver) [when deps.ClientIPResolver != nil]", + "netaccess.Middleware(deps.NetworkAccess.Registry) [when deps.NetworkAccess != nil]", "apimw.RequestLogger(deps.NodeID)", "middleware.Recoverer", "apimw.Metrics", @@ -309,6 +325,7 @@ "middleware": [ "apimw.RequestID", "clientip.Middleware(deps.ClientIPResolver) [when deps.ClientIPResolver != nil]", + "netaccess.Middleware(deps.NetworkAccess.Registry) [when deps.NetworkAccess != nil]", "apimw.RequestLogger(deps.NodeID)", "middleware.Recoverer", "apimw.Metrics", @@ -322,6 +339,7 @@ "middleware": [ "apimw.RequestID", "clientip.Middleware(deps.ClientIPResolver) [when deps.ClientIPResolver != nil]", + "netaccess.Middleware(deps.NetworkAccess.Registry) [when deps.NetworkAccess != nil]", "apimw.RequestLogger(deps.NodeID)", "middleware.Recoverer", "apimw.Metrics", @@ -335,6 +353,7 @@ "middleware": [ "apimw.RequestID", "clientip.Middleware(deps.ClientIPResolver) [when deps.ClientIPResolver != nil]", + "netaccess.Middleware(deps.NetworkAccess.Registry) [when deps.NetworkAccess != nil]", "apimw.RequestLogger(deps.NodeID)", "middleware.Recoverer", "apimw.Metrics", @@ -348,6 +367,7 @@ "middleware": [ "apimw.RequestID", "clientip.Middleware(deps.ClientIPResolver) [when deps.ClientIPResolver != nil]", + "netaccess.Middleware(deps.NetworkAccess.Registry) [when deps.NetworkAccess != nil]", "apimw.RequestLogger(deps.NodeID)", "middleware.Recoverer", "apimw.Metrics", @@ -361,6 +381,7 @@ "middleware": [ "apimw.RequestID", "clientip.Middleware(deps.ClientIPResolver) [when deps.ClientIPResolver != nil]", + "netaccess.Middleware(deps.NetworkAccess.Registry) [when deps.NetworkAccess != nil]", "apimw.RequestLogger(deps.NodeID)", "middleware.Recoverer", "apimw.Metrics", @@ -374,6 +395,7 @@ "middleware": [ "apimw.RequestID", "clientip.Middleware(deps.ClientIPResolver) [when deps.ClientIPResolver != nil]", + "netaccess.Middleware(deps.NetworkAccess.Registry) [when deps.NetworkAccess != nil]", "apimw.RequestLogger(deps.NodeID)", "middleware.Recoverer", "apimw.Metrics", @@ -387,6 +409,7 @@ "middleware": [ "apimw.RequestID", "clientip.Middleware(deps.ClientIPResolver) [when deps.ClientIPResolver != nil]", + "netaccess.Middleware(deps.NetworkAccess.Registry) [when deps.NetworkAccess != nil]", "apimw.RequestLogger(deps.NodeID)", "middleware.Recoverer", "apimw.Metrics", @@ -399,6 +422,7 @@ "id": 24, "middleware": [ "clientip.Middleware(s.clientIP) [when s.clientIP != nil]", + "netaccess.Middleware(s.ingressTokens) [when s.ingressTokens != nil]", "cors.Handler(cors.Options{ AllowedOrigins: []string{\"*\"}, AllowedMethods: []string{\"GET\", \"HEAD\", \"OPTIONS\"}, AllowedHeaders: []string{ \"Accept\", \"Authorization\", \"Content-Type\", proxyRangeHeader, \"If-Match\", \"If-Modified-Since\", \"If-None-Match\", \"If-Range\", \"If-Unmodified-Since\", }, ExposedHeaders: []string{ \"Accept-Ranges\", \"Content-Encoding\", \"Content-Length\", \"Content-Range\", \"ETag\", \"Last-Modified\", }, MaxAge: 86400, })" ] }, @@ -406,6 +430,7 @@ "id": 25, "middleware": [ "clientip.Middleware(s.clientIP) [when s.clientIP != nil]", + "netaccess.Middleware(s.ingressTokens) [when s.ingressTokens != nil]", "cors.Handler(cors.Options{ AllowedOrigins: []string{\"*\"}, AllowedMethods: []string{\"GET\", \"HEAD\", \"OPTIONS\"}, AllowedHeaders: []string{ \"Accept\", \"Authorization\", \"Content-Type\", proxyRangeHeader, \"If-Match\", \"If-Modified-Since\", \"If-None-Match\", \"If-Range\", \"If-Unmodified-Since\", }, ExposedHeaders: []string{ \"Accept-Ranges\", \"Content-Encoding\", \"Content-Length\", \"Content-Range\", \"ETag\", \"Last-Modified\", }, MaxAge: 86400, })", "s.meterEgress" ] @@ -414,6 +439,7 @@ "id": 26, "middleware": [ "clientip.Middleware(s.clientIP) [when s.clientIP != nil]", + "netaccess.Middleware(s.ingressTokens) [when s.ingressTokens != nil]", "cors.Handler(cors.Options{ AllowedOrigins: []string{\"*\"}, AllowedMethods: []string{\"GET\", \"HEAD\", \"OPTIONS\"}, AllowedHeaders: []string{ \"Accept\", \"Authorization\", \"Content-Type\", proxyRangeHeader, \"If-Match\", \"If-Modified-Since\", \"If-None-Match\", \"If-Range\", \"If-Unmodified-Since\", }, ExposedHeaders: []string{ \"Accept-Ranges\", \"Content-Encoding\", \"Content-Length\", \"Content-Range\", \"ETag\", \"Last-Modified\", }, MaxAge: 86400, })", "s.requireBearer" ] @@ -27155,6 +27181,118 @@ "error_codes": null, "source_file": "internal/proxy/server.go" }, + { + "listener": "proxy", + "namespace": "legacy_unversioned", + "method": "GET", + "path": "/network-access/status", + "route_group": "/", + "handler": "(*internal/proxy.Server).handleNetworkAccessStatus", + "handler_expr": "s.handleNetworkAccessStatus", + "handler_kind": "method", + "handler_resolved": true, + "middleware_chain": 26, + "auth_class": "node_bearer", + "auth_traits": [ + "cors", + "node_bearer" + ], + "conditions": [], + "conditional": false, + "request_kind": "none", + "response_media_kind": "json", + "delegates_to": "", + "streams": false, + "upgrades_websocket": false, + "method_origin": "explicit", + "success_statuses": null, + "error_codes": null, + "source_file": "internal/proxy/server.go" + }, + { + "listener": "proxy", + "namespace": "legacy_unversioned", + "method": "POST", + "path": "/network-access/{provider}/connect", + "route_group": "/", + "handler": "(*internal/proxy.Server).handleNetworkAccessConnect", + "handler_expr": "s.handleNetworkAccessConnect", + "handler_kind": "method", + "handler_resolved": true, + "middleware_chain": 26, + "auth_class": "node_bearer", + "auth_traits": [ + "cors", + "node_bearer" + ], + "conditions": [], + "conditional": false, + "request_kind": "unknown", + "response_media_kind": "json", + "delegates_to": "", + "streams": false, + "upgrades_websocket": false, + "method_origin": "explicit", + "success_statuses": null, + "error_codes": null, + "source_file": "internal/proxy/server.go" + }, + { + "listener": "proxy", + "namespace": "legacy_unversioned", + "method": "POST", + "path": "/network-access/{provider}/disconnect", + "route_group": "/", + "handler": "(*internal/proxy.Server).handleNetworkAccessDisconnect", + "handler_expr": "s.handleNetworkAccessDisconnect", + "handler_kind": "method", + "handler_resolved": true, + "middleware_chain": 26, + "auth_class": "node_bearer", + "auth_traits": [ + "cors", + "node_bearer" + ], + "conditions": [], + "conditional": false, + "request_kind": "unknown", + "response_media_kind": "json", + "delegates_to": "", + "streams": false, + "upgrades_websocket": false, + "method_origin": "explicit", + "success_statuses": null, + "error_codes": null, + "source_file": "internal/proxy/server.go" + }, + { + "listener": "proxy", + "namespace": "legacy_unversioned", + "method": "GET", + "path": "/network-access/{provider}/status", + "route_group": "/", + "handler": "(*internal/proxy.Server).handleNetworkAccessProviderStatus", + "handler_expr": "s.handleNetworkAccessProviderStatus", + "handler_kind": "method", + "handler_resolved": true, + "middleware_chain": 26, + "auth_class": "node_bearer", + "auth_traits": [ + "cors", + "node_bearer" + ], + "conditions": [], + "conditional": false, + "request_kind": "none", + "response_media_kind": "json", + "delegates_to": "", + "streams": false, + "upgrades_websocket": false, + "method_origin": "explicit", + "success_statuses": null, + "error_codes": null, + "source_file": "internal/proxy/server.go" + }, { "listener": "proxy", "namespace": "legacy_unversioned", diff --git a/docker-compose.yml b/docker-compose.yml index acb071cb4a..db54c3fcba 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -111,11 +111,16 @@ services: # SECRET_KEY: ${SECRET_KEY:?Set SECRET_KEY in .env — generate one with openssl rand -base64 48} # DATABASE_URL: postgres://${POSTGRES_USER:-silo}:${POSTGRES_PASSWORD:-silo}@postgres:5432/${POSTGRES_DB:-silo}?sslmode=disable # REDIS_URL: redis://redis:6379 + # # Network access provider plugins run on proxies too. The proxy keeps + # # its own copies, rehydrated from the database, under this dir; it + # # does not need the primary node's plugin volume. + # SILO_PLUGIN_CACHE_DIR: /var/lib/silo/plugins # PORT: "8080" # ports: # - "${PROXY_PORT:-8083}:8080" # volumes: # - ${MEDIA_ROOT:?Set MEDIA_ROOT in .env to the host media path}:${MEDIA_CONTAINER_ROOT:-/mnt/media}:ro + # - ${SILO_DATA_ROOT:-/opt/silo}/proxy-plugins:/var/lib/silo/plugins # # See the note on the integrated service above: a node reports its own # # CPU, load, and memory rather than the physical machine's only with # # these mounted, and the Nodes page reads exactly what the node reports. diff --git a/docs/admin-api.md b/docs/admin-api.md index b61bc26940..a3591e5a64 100644 --- a/docs/admin-api.md +++ b/docs/admin-api.md @@ -153,6 +153,21 @@ Rows from `GET /api/v2/admin/sessions` may then include `routing_egress_node_name`. Node fields are absent for the integrated API process and for direct play's `none` executor. +`network_access_route: true` on the same capability response advertises +`routing_network_provider` on v2 session rows. A nonempty value is the validated +network access provider identifier selected when preparing playback (for example, +`tailscale`); an empty string means the default network, and an absent field means +the session predates this telemetry. Default does not distinguish LAN, public URL, +or reverse proxy access. This records the prepared route, not a live measurement +of every media request or an inference from the client's IP address. Provider +display names come from `/api/v2/network-access/capabilities`. + +The web activity views show that network alongside the named execution and egress +nodes. API egress is labeled "API server"; its reporting identity remains in the +tooltip. Native and Jellyfin-compatible playback both populate the route, including +session recovery. This additive admin observation does not change Apple, Android, +or Jellyfin playback contracts; those clients need no changes to report it. + `silo_playback_routing_decisions_total` counts routing outcomes with bounded `workload`, `execution`, `egress`, `outcome`, and `reason` labels. It never labels observations with playback-session or node identity. @@ -200,6 +215,7 @@ configuration, last health result, and last stored hardware inventory. See | `hw_accel_override`, `hw_device_override` | string | This node's own acceleration policy (see below). Omitted when the node inherits the cluster-wide settings, which is the normal case. | | `capability_drift` | string | Human-readable note describing how the node's hardware got worse at the last capability refetch. Omitted when the last refetch found no regression (see below). | | `capability_drift_baseline` | object | What that note is waiting on — `{"backends": ["nvenc"], "devices": [{"uuid": "GPU-8a7b…", "aliases": ["GPU-8a7b…", "0000:03:00.0", "/dev/dri/renderD128"]}]}`. Never present without `capability_drift`; absent with it only for a note written before this field existed (see below). Each device carries every stable name it answered to, so it is recognized if it returns renumbered; `uuid` is held apart because it is the only name that can prove a *different* card, a replacement in the same slot inheriting both the slot and the render path. Either key is omitted when empty. | +| `network_access` | object | The node's last report about the network access provider plugins running beside it, keyed by provider slug — `{"tailscale": {"state": "connected", "origin": "https://proxy-1.tail1234.ts.net", "hostname": "proxy-1.tail1234.ts.net", "updated_at": "…"}}`. `state` is one of `disconnected`, `awaiting_authorization`, `connecting`, `connected`, `error`; `origin`, `hostname` and `updated_at` are omitted when the provider did not report them. Written by the same health check that writes `last_stats`, so it is exactly as fresh as `last_health_check`, and a check that carries no report clears it. Omitted when the node reports no providers. Only proxy nodes report it: clients never talk to transcode nodes. See [proxy origins by access path](#proxy-origins-by-access-path). | ### Acceleration overrides @@ -216,10 +232,39 @@ denominator. A homogeneous deployment should leave both unset and configure Repointing a node's `url` to a different machine clears the identity-bound state on that row — `capabilities`, `capabilities_hash`, -`capabilities_refreshed_at`, `last_stats`, and the drift note with its baseline -— because all of it describes the worker the old address reached, and the pools -are reloaded from the row immediately. The replacement is treated as newly -registered until its first health check and capability fetch. +`capabilities_refreshed_at`, `last_stats`, `network_access`, and the drift note +with its baseline — because all of it describes the worker the old address +reached, and the pools are reloaded from the row immediately. The replacement +is treated as newly registered until its first health check and capability +fetch. + +### Proxy origins by access path + +A request reaches Silo on an *access path*: the default path (LAN, `public_url`, +a reverse proxy) or the overlay of a network access provider plugin, which +stamps the requests it forwards with a per-process ingress token. Stream and +download URLs that name a proxy node are built for the path the request came +in on: + +- Default path: the proxy's `public_url` when set, otherwise its `url` — the + behavior described under `public_url` above. +- Provider path (for example a tailnet client): the `origin` that the same + provider reports on that proxy in `network_access`, and only while its + `state` is `connected`. A proxy without a connected origin for the client's + provider is excluded from proxy egress for that request *before* a route is + reserved, so the existing fallbacks apply unchanged: an API-relative stream + URL, and the API relaying the transcode node. A client on an overlay is never + handed a LAN origin it cannot open. + +Routing policies interact with this the way they interact with any pool +shortage. Under `prefer_proxy` a provider-path request with no reachable proxy +falls back to API egress; under `proxy_only` it fails with the existing +`route_capacity_unavailable` outcome until a proxy enrolls with that provider. +Downloads served through `/downloads/{id}/file-proxy` and +`/direct-download-proxy` follow the same rule: the `Location` names the +provider origin, or the file is served from the API server when the planned +proxy has none. The Jellyfin-compatible playback redirects use the same +accessor. A node finds its own row by URL first: `NODE_URL` on the node is matched against `stream_nodes.url`, ignoring a trailing slash on either side. Set @@ -2589,6 +2634,20 @@ installation without a recorded update or without a repository is 409 (v1 answer Success is 200. Non-retryable: no replay identity, and a lost response may follow a committed update. +`POST /api/v2/admin/plugins/installations/{id}/restart` stops the installation's process +and, for a resident plugin, starts it again with a fresh failure budget; a non-resident plugin +is only stopped and launches on its next use. A disabled installation is 409. Success is 200 +with the installation, whose `runtime` reports the outcome, including a launch that failed. +Repeating the request converges on one running process, so it is naturally idempotent. + +Every installation carries `runtime`: `resident` (true when the server supervises the +process: it starts at boot once the API listener is bound, restarts after a crash with +exponential backoff from 1 s to 60 s, and is parked as `failed` after ten consecutive +failures until restarted or reconfigured), `state` (`stopped`, `starting`, `running`, +`backoff`, `failed`), `restart_count`, `last_error`, `last_started_at` and `next_restart_at`. +Plugins declaring `network_access_provider.v1` are resident; every other plugin starts on +first use and reports only `running` or `stopped`. + `DELETE /api/v2/admin/plugins/installations/{id}` stops the plugin, deletes the row (configuration, bindings and archives cascade) and removes its files; on a failed row delete an enabled plugin is restarted. Row delete and file removal are not one transaction and a diff --git a/docs/architecture/network-access.md b/docs/architecture/network-access.md new file mode 100644 index 0000000000..b6aacffdcc --- /dev/null +++ b/docs/architecture/network-access.md @@ -0,0 +1,225 @@ +# Network access providers + +Network access providers let an installed plugin give a Silo deployment an +overlay-network identity (Tailscale via tsnet first, NetBird later) so clients +reach it without port forwarding or a public reverse proxy. Phase one covers +one API server plus any number of proxy nodes; transcode nodes are never +exposed because clients never talk to them. The plugin owns the overlay client +and reverse-proxies to the local Silo listeners; the server owns supervision, +per-instance state, the access path, status, and the admin API +([docs/network-access-api.md](../network-access-api.md)). Tracking issue: +Silo-Server/silo-server#1001. + +``` + overlay LAN / backend + client ──► [plugin listener] ──► 127.0.0.1:8080 API server + X-Silo-Ingress-Token │ relay ▲ health pull + ▼ │ (network_access) + client ──► [plugin listener] ──► 127.0.0.1:PORT proxy node ── transcode node + on the proxy host +``` + +## Why a plugin + +The overlay client (tsnet) updates on its own cadence; shipping it as a plugin +lets it release without a Silo release. The capability +`network_access_provider.v1` is provider-neutral so a second provider is a +second plugin, not a new server surface. Providers are discovered from the +manifest's typed descriptor (`network_access_provider.provider`, the stable +slug, and `display_name`) without launching the plugin. + +## Resident contract + +Plugins declaring `network_access_provider.v1` are *resident*: ingress must be +up before anyone can use it, so they do not start lazily on first RPC like +other plugins. The resident supervisor (`internal/plugins/resident.go`) owns +them: + +- Started once the API listener is bound (`Service.StartResidents`), never + from the boot-time lifecycle hooks, so a plugin cannot proxy to a listener + that does not exist yet. +- Reconciled on every `OnLifecycleChange` (install, enable, disable, config + save, auto-update, uninstall). A config save or auto-update stops the + process; the next reconcile starts it again. +- Restarted after a crash with exponential backoff from 1 s to 60 s plus + jitter, reset after five minutes of stable running; parked as `failed` + after ten consecutive failures until an admin restarts or reconfigures it. + The plugin host's exit watcher reports the exit; the gRPC health probe + still catches a hung process. +- Stopped before the HTTP servers drain (`Service.StopResidents`) so overlay + ingress goes away first. +- Visible to admins as `runtime` on each installation and restartable through + `POST /api/v2/admin/plugins/installations/{id}/restart`. + +The predicate is a list of capability types +(`IsResidentCapabilityType`), so a later resident kind is one more entry. + +The admin network access reads never launch a plugin either: a host whose +process is not running answers `unavailable` with the supervisor's last error, +and the admin fixes it from the Plugins page. + +## Ingress token and access path + +The server must know which requests arrived through a provider so stream URLs +and WebSocket origin checks can act on it, and it must learn that from +something a client cannot forge. Trusted-proxy CIDRs would not do: the +plugin's loopback source is indistinguishable from any other local proxy. + +- On every start of a provider instance the host mints 32 random bytes, the + *ingress token*, and hands it to the plugin through `RuntimeHost.GetHostInfo` + over the plugin's private gRPC broker. Stopping the process revokes it. The + token is in memory only (`internal/netaccess.Registry`) and so outlives + neither the plugin process nor the server process; a stale token after a + plugin restart is rejected until the plugin re-reads `GetHostInfo`. +- The plugin stamps `X-Silo-Ingress-Token` on every request it proxies, keeps + `Host`, overwrites `X-Forwarded-Proto` with `https`, and sets + `X-Forwarded-For` to the overlay peer. +- `netaccess.Middleware` runs on all three listeners (API, Jellyfin, ABS) + before logging or any handler. It validates the token in constant time, + strips the header, and records `netaccess.Path{Provider}` on the request + context. An unknown token is `403`; a request without the header is on the + default path. The route inventory classifies it as infrastructure + middleware: it never grants or changes authorization. +- `clientip` already trusts loopback, so the forwarded headers are honoured. + Operators who narrow `clientip.trusted_proxies` must keep loopback. +- `socketOriginAllowed` accepts the overlay origins of providers currently + connected on this host (`StatusCache.ConnectedOrigins`) next to + `server.public_url`, so a browser reaching Silo over the overlay can open + WebSockets. + +Downstream, `Path.Provider` selects the proxy origin a client is handed: a +tailnet client cannot reach a LAN proxy origin, so a provider path returns +that provider's connected origin on the proxy or falls back to API relay. + +A status push is accepted only from the process holding the installation's +current ingress token. A push that was in flight when its process was +stopped, crashed, or replaced lands after the revoke and is dropped, so a +dead instance can never write a stale origin over the replacement's. + +## Status + +Plugins push status on every change through +`RuntimeHost.ReportNetworkAccessStatus`; the host keeps the latest per +installation in `netaccess.StatusCache` and logs state transitions only, +never `auth_url`. The admin API still asks the plugin directly (`GetStatus`, +`Connect`, `Disconnect`, ten-second timeout each) and writes the answer back +into the cache so the origin allow list does not wait for the next push. Each +push and each read go through one converter +(`pluginhost.NetworkAccessStatusFromProto`) so the two views cannot drift. + +`unavailable` is a host-side state a plugin never reports: it means the +process is not running or did not answer. + +## Instance state and scopes + +tsnet needs a persistent node key per overlay identity, and each proxy needs +its own. State lives in `plugin_instance_state`, keyed by installation and +*host scope*, encrypted with GCM and row AAD bound to `::` +(see [secret-encryption.md](secret-encryption.md)). The scope is derived by +the host, never supplied by the plugin: `api` on the API server, `node:` +(`stream_nodes.id`) on a proxy. The same scope string is the `host.id` the +admin API reports, so the two never disagree. Limits: key ≤ 256 bytes, value +≤ 256 KiB, ≤ 256 keys per scope. Uninstall cascades; config test-runs +(negative installation ids) get no state. The API never returns instance +state. Empty values also use an encrypted, row-bound envelope; a plaintext +empty marker is rejected. + +## Proxy nodes + +Every enabled proxy node runs the same provider installations as the API +server, each instance with its own overlay identity under its own +`node:` scope. Transcode nodes run none: clients never talk +to them. + +- The proxy process builds a reduced plugin host (`cmd/silo/proxy_plugins.go`, + `plugins.NewNodeService`): the resident supervisor, an archive cache, the + instance-state store scoped to the node row, and the netaccess broker. No + admin routes, no catalog or installer, no metadata or other capability + dispatch. `GetHostInfo` reports role `proxy`, the node name and row id, and + a single `api` listener: the proxy's own address. +- A proxy never reads the install path the API server recorded on the row: + that names a directory on the API server's machine. The archive cache + (`plugins.NewArchiveCacheAt`) rehydrates each release from + `plugin_archives` into the proxy's own cache root, `SILO_PLUGIN_CACHE_DIR` + (default `/silo-plugins`), under + `////plugin`, where `` is the + unique install directory name the API's installer chose. Only that root + has to exist and be writable on the proxy; a shared plugin volume is not + required. When a new release lands the previous one is removed from the + cache. The archive was resolved for the API server's platform at install + time, so every proxy must run the same OS and architecture as the API + server in this release; a proxy on another platform refuses the binary + with a clear error at rehydration instead of failing every launch. +- A replaced or auto-updated binary changes the row's version and install + path. The API host stopped its own process before installing; the proxy's + process is still alive and healthy, so the supervisor stops it on the next + reconcile (`Reconcile` treats a version or install-path change as a + replacement, not a restart-worthy failure) and starts the new release from + its rehydrated archive. +- The node row id comes from the config watcher, which matches `NODE_URL` / + `NODE_NAME` to `stream_nodes`. Until it resolves there is no state scope + for the node key, so the supervisor's resident gate keeps every provider + stopped, logs why once, and the admin status reports the reason as that + host's `unavailable` error. The gate also requires the row to be an + enabled proxy node: disabling the node or changing its type stops its + providers on the next reconcile, since the API stops listing it as a + network-access host and could no longer disconnect them. The gate is + re-checked on every reconcile. The node scope is also the host identity + residents run under: if the row is deleted and re-registered under a new + id, the next reconcile replaces every provider process, since a process + keeps the identity it started with (its overlay node key, the node id it + reports) in memory. +- Lifecycle changes happen on the API server. It publishes + `cache.EventPluginsChanged` on `ChannelAdmin` after every + `OnLifecycleChange`. Config saves and admin restarts advance the persisted + `runtime_generation`; the event asks proxies to reconcile against that + generation. Proxies also reconcile on a 60 s poll + (`Service.FollowLifecycleChanges`), so a missed publish costs at most one + minute. A caller that cancels an accepted restart stops waiting while the + supervisor continues the restart and records the result. +- The API reaches proxies over their backend URL with the node bearer: + `GET /network-access/status`, `POST /network-access/{provider}/connect` + and `.../disconnect` on the proxy listener, ten seconds each, in parallel, + like force-reload. Those routes carry `auth_url` and error text because + they take the bearer; the public `/health` report carries only state, + origin and hostname. The proxy listener also runs `netaccess.Middleware`, + so its own provider's stamped requests are validated there too. +- The proxy's `/health` `network_access` block is what the API stores on the + node row every sweep and what `ClientURLFor` hands overlay clients. The + admin status is live and can run ahead of it by up to one sweep. + +A provider slug names one provider per deployment. If two enabled +installations declare the same slug, the one with the lowest installation +id owns it: commands, status, and the node health report address that one, +and the duplicate is neither started as a resident nor listed; it is +logged and skipped on every host. +Other capabilities on a resident plugin cannot launch it lazily. This also +applies to excluded duplicates and hosts whose resident gate is closed; +capability RPCs may only reuse the supervisor's running process. + +## Single-API constraint + +Two API replicas would both load the `api` scope and share one node key, +which the overlay treats as one node flapping between two machines. Phase one +therefore supports one API server; in `server.mode = api` each replica keeps +a unique process presence marker in Redis (`cache.APIReplicaPresence`, 90 s +TTL), independent of shared logical node names, and logs a +warning at start when providers are installed and more than one live replica +is seen (best effort, no hard refusal). The census runs once per start, so +the warning appears in the log of whichever replica starts second, not the +one already running. Proxy nodes scale freely because each has its own scope. + +## Security notes + +- `auth_url` is admin-only and never logged. +- The ingress token is per process start, delivered over the private broker, + compared in constant time, and revoked on stop. +- Instance state is encrypted with row AAD and never leaves the server. +- Enrollment supports an auth key in plugin config so N proxies enroll + without N admin clicks, with the interactive `auth_url` as fallback. + +## Out of scope for phase one + +Funnel or any public exposure, login via overlay identity, multiple API +replicas, NetBird. The non-goals in [docs/non-goals.md](../non-goals.md) +apply unchanged: a provider proxies Silo's own listeners and nothing else. diff --git a/docs/architecture/worker-http-protocol.md b/docs/architecture/worker-http-protocol.md index 624a73db32..295c17bbbb 100644 --- a/docs/architecture/worker-http-protocol.md +++ b/docs/architecture/worker-http-protocol.md @@ -37,6 +37,18 @@ worker. They consume no request DTO and retain the existing paths: | `/admin/force-reload` | Empty 204 | 401, 500 | Proxy reloads configuration; transcode node also tears down sessions and delivery authority. | | `/admin/reprobe-capabilities` | JSON 200: `resolved`, `capability_hash` | 401, 409, 503 | Rebuild the capability snapshot; an incomplete rebuild retains the prior published hash. | +The proxy listener additionally carries the network access provider routes the +API server fans its `/api/v2/admin/network-access/{provider}/...` operations +out to, under the same node bearer: `GET /network-access/{provider}/status` +(JSON, only the named provider, independent of other providers' response times), and +`POST /network-access/{provider}/connect` / `.../disconnect` (JSON, the state +the instance reached; 404 for a provider no enabled installation declares). +The status responses carry `auth_url` and error text. The original +`GET /network-access/status` also remains available to read every provider +instance on that proxy. Connect and disconnect are `natural_idempotent`: repeating converges on connected or +disconnected. A proxy that hosts no plugins answers plain-text 503. See +[network-access.md](network-access.md). + An unconfigured transcode listener also returns plain-text 503 from its bearer middleware, including for both reload commands. All six require the existing node bearer token. Reprobe refuses active probes; the transcode node also refuses active jobs while holding its GPU admission gate. diff --git a/docs/network-access-api.md b/docs/network-access-api.md new file mode 100644 index 0000000000..cfe58616a7 --- /dev/null +++ b/docs/network-access-api.md @@ -0,0 +1,134 @@ +# Network Access API + +Network access providers are plugins that give a Silo deployment an identity on +an overlay network (Tailscale first; the contract is provider-neutral) so +clients reach it without port forwarding or a public reverse proxy. The plugin +owns the overlay client and reverse-proxies to the local Silo listeners; the +server owns supervision, state storage, status and this admin surface. The +design is in [docs/architecture/network-access.md](architecture/network-access.md). + +Provider slugs (`tailscale`, `netbird`, …) come from the plugin's manifest +descriptor and key every route below. A slug is a lowercase path-safe token +matching `^[a-z0-9]+(?:[._-][a-z0-9]+)*$`; anything else is rejected as +`422 validation_failed` before lookup. Clients need no changes to use a +deployment reached this way. + +## `GET /api/v2/network-access/capabilities` + +Authenticated account; no profile required. Lists the installed providers, +read from plugin manifests without launching any plugin: + +```json +{ + "revision": "…", + "state": "available", + "allowed": true, + "providers": [ + { "provider": "tailscale", "display_name": "Tailscale", "installation_id": "7" } + ] +} +``` + +`state` is `available` when at least one enabled installation declares +`network_access_provider.v1`, `not_configured` otherwise (including worker +modes where the plugin service is not wired). Provider health is not state; +read it from the admin status below. `providers` is always an array. The +document carries an `ETag` and answers `If-None-Match` with 304. + +## `GET /api/v2/admin/network-access/{provider}/status` + +Acting admin. Asks the provider's plugin instance on every host that runs it +for its live status, with a ten-second timeout per host: + +```json +{ + "provider": "tailscale", + "hosts": [ + { + "host": { "id": "api", "role": "api", "name": "Living Room" }, + "state": "connected", + "hostname": "silo.overlay.example.test", + "origin": "https://silo.overlay.example.test", + "addresses": ["127.0.0.1"], + "provider_version": "tsnet 1.102.4", + "updated_at": "2026-09-14T09:00:00.000Z" + } + ] +} +``` + +- `host.id` is `api` for the API server and `node:` for a proxy node; it + is also the instance-state scope the plugin's node keys are stored under. + The API host comes first, then every enabled proxy node in id order. Proxy + rows are read over the node's backend URL with the node bearer + (`GET /network-access/{provider}/status` on the proxy listener) with the same + ten-second timeout; a proxy that cannot be reached, refuses the bearer, + runs a build without the routes, or has not resolved its `stream_nodes` + row yet answers `unavailable` with the reason in `error`. Transcode nodes + are never listed. Each request reads only the named provider, so another + provider's timeout cannot hide its status. +- `state` is one of the plugin's states `disconnected`, + `awaiting_authorization`, `connecting`, `connected`, `error`, or the + server-side `unavailable`: the plugin process is not running on that host + (stopped, in restart backoff, failed, or unhealthy) or did not answer in + time. Provider states outside this published set map to `error`; the + provider's original value is preserved in `raw_state`. `error` then says + why, and `updated_at` is absent. +- `auth_url` is present only while `awaiting_authorization`. It is + admin-only and the server never logs it. +- `origin` is the `scheme://host[:port]` clients on the overlay use for the + API listener. `addresses` is always an array. +- Fields the provider did not report are omitted. + +An unknown provider slug is `404 not_found`. The status is read live and is +not cacheable; the web page polls it every 15 seconds while in front. + +## `POST /api/v2/admin/network-access/{provider}/connect` + +## `POST /api/v2/admin/network-access/{provider}/disconnect` + +Acting admin; refused to non-admins in demo mode. Body is optional: + +```json +{ "hosts": ["api"] } +``` + +`hosts` names the host ids to act on; omitting the body, sending `{}`, or +omitting `hosts` acts on every host. An empty `hosts` array or an id this +deployment does not run the provider on is `422 validation_failed` at +`body.hosts`, and nothing is applied. + +Both answer `202 Accepted` with the same body as the status read: the state +each host reached within ten seconds. Connect starts enrollment; the plugin +may keep working in the background (`awaiting_authorization` carries the +`auth_url`; `connecting` becomes `connected` later), so poll status for the +final state. Disconnect tears the overlay listener down and clears the +plugin's desired-connected intent. Hosts not named in `hosts` answer their +current status. A host whose plugin is not running is still acknowledged: its +row says `unavailable` and the command was not applied there; restart the +plugin from `POST /api/v2/admin/plugins/installations/{id}/restart` first. + +Both are naturally idempotent: repeating connect converges on one connected +instance per host, repeating disconnect on disconnected. An unknown provider +is `404 not_found`. + +Proxy hosts are commanded over the node bearer routes +`POST /network-access/{provider}/connect` and +`POST /network-access/{provider}/disconnect` on the proxy listener, each +bounded by the same ten seconds; the proxy answers the state its own +instance reached. A proxy's overlay origin then reaches stream URL selection +through the node's `/health` report on the next sweep (see +`network_access` in [docs/admin-api.md](admin-api.md)). + +## Related surfaces + +- Plugin process state (resident supervisor, restart counts, last error) is + on each installation's `runtime` in the + [plugin installation API](admin-api.md) and on the Plugins page. +- Proxy nodes report their provider status through the node health check; + `GET /api/v2/admin/nodes` carries it as `network_access` (see + [admin-api.md](admin-api.md)). +- The web client's Settings → Network Access page is built on these three + operations. jellycompat has no equivalent: Jellyfin clients reach the + Jellyfin listener through the same overlay origin the provider exposes for + it, and need no provider awareness. diff --git a/go.mod b/go.mod index 35552915d6..58363cf79a 100644 --- a/go.mod +++ b/go.mod @@ -128,7 +128,7 @@ require ( ) require ( - github.com/Silo-Server/silo-plugin-sdk v0.13.2 + github.com/Silo-Server/silo-plugin-sdk v0.16.1 github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.14 // indirect github.com/aws/aws-sdk-go-v2/internal/configsources v1.4.30 // indirect github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.7.30 // indirect diff --git a/go.sum b/go.sum index ec671a414c..0b6f4fa7f4 100644 --- a/go.sum +++ b/go.sum @@ -6,8 +6,8 @@ github.com/PuerkitoBio/goquery v1.8.0 h1:PJTF7AmFCFKk1N6V6jmKfrNH9tV5pNE6lZMkG0g github.com/PuerkitoBio/goquery v1.8.0/go.mod h1:ypIiRMtY7COPGk+I/YbZLbxsxn9g5ejnI2HSMtkjZvI= github.com/SherClockHolmes/webpush-go v1.4.0 h1:ocnzNKWN23T9nvHi6IfyrQjkIc0oJWv1B1pULsf9i3s= github.com/SherClockHolmes/webpush-go v1.4.0/go.mod h1:XSq8pKX11vNV8MJEMwjrlTkxhAj1zKfxmyhdV7Pd6UA= -github.com/Silo-Server/silo-plugin-sdk v0.13.2 h1:w7U0mmljVPauKfzRLNKusuiYFpuoEhuCSX4Hp3s9eRw= -github.com/Silo-Server/silo-plugin-sdk v0.13.2/go.mod h1:etqmxLTwjxpFH9goAjBDfNDoqHMv2/sqUXu8yx3hNfA= +github.com/Silo-Server/silo-plugin-sdk v0.16.1 h1:bctPlzlsr75Z9ARQ86VnYY6UhyQ9YPwgoP+mqSRESPA= +github.com/Silo-Server/silo-plugin-sdk v0.16.1/go.mod h1:abwsCEKuPAAgeAqpNGbwoaut2eQlC/Kj97u89Vvg9qM= github.com/TwiN/go-color v1.4.1 h1:mqG0P/KBgHKVqmtL5ye7K0/Gr4l6hTksPgTgMk3mUzc= github.com/TwiN/go-color v1.4.1/go.mod h1:WcPf/jtiW95WBIsEeY1Lc/b8aaWoiqQpu5cf8WFxu+s= github.com/abadojack/whatlanggo v1.0.1 h1:19N6YogDnf71CTHm3Mp2qhYfkRdyvbgwWdd2EPxJRG4= diff --git a/internal/api/handlers/admin_logs_socket_v2.go b/internal/api/handlers/admin_logs_socket_v2.go index c8d22b4ef5..44be3955aa 100644 --- a/internal/api/handlers/admin_logs_socket_v2.go +++ b/internal/api/handlers/admin_logs_socket_v2.go @@ -38,9 +38,10 @@ type AdminLogsSocketV2 struct { Tickets *evt.SocketTicketStore Validate EventsSocketValidator // PublicOrigin is the configured external origin, never a forwarded header. - PublicOrigin string - publicOrigin atomic.Pointer[string] - checkInterval time.Duration + PublicOrigin string + publicOrigin atomic.Pointer[string] + overlayOrigins atomic.Pointer[OverlayOriginSource] + checkInterval time.Duration } // NewAdminLogsSocketV2 reuses the events socket's ticket store shape and the @@ -90,7 +91,7 @@ func (h *AdminLogsSocketV2) ServeHTTP(w http.ResponseWriter, r *http.Request) { http.Error(w, "invalid handshake", http.StatusBadRequest) return } - if !socketOriginAllowed(r, h.currentPublicOrigin()) { + if !socketOriginAllowed(r, h.currentPublicOrigin(), overlayOriginsFrom(h.overlayOrigins.Load())) { http.Error(w, "origin refused", http.StatusForbidden) return } @@ -155,7 +156,9 @@ func (h *AdminLogsSocketV2) ServeHTTP(w http.ResponseWriter, r *http.Request) { } } }() - h.Logs.serveLogStream(w, r.WithContext(ctx), websocket.Upgrader{Subprotocols: []string{AdminLogsSocketProtocol}, CheckOrigin: func(r *http.Request) bool { return socketOriginAllowed(r, h.currentPublicOrigin()) }}) + h.Logs.serveLogStream(w, r.WithContext(ctx), websocket.Upgrader{Subprotocols: []string{AdminLogsSocketProtocol}, CheckOrigin: func(r *http.Request) bool { + return socketOriginAllowed(r, h.currentPublicOrigin(), overlayOriginsFrom(h.overlayOrigins.Load())) + }}) } func (h *AdminLogsSocketV2) currentPublicOrigin() string { @@ -169,3 +172,9 @@ func (h *AdminLogsSocketV2) SetPublicOrigin(origin string) { normalized := strings.TrimRight(origin, "/") h.publicOrigin.Store(&normalized) } + +// SetOverlayOrigins installs the source of overlay origins accepted next to +// the public origin. +func (h *AdminLogsSocketV2) SetOverlayOrigins(source OverlayOriginSource) { + h.overlayOrigins.Store(&source) +} diff --git a/internal/api/handlers/admin_node_commands.go b/internal/api/handlers/admin_node_commands.go index 621b924bb0..1711cb4183 100644 --- a/internal/api/handlers/admin_node_commands.go +++ b/internal/api/handlers/admin_node_commands.go @@ -29,12 +29,12 @@ func (h *NodeHandler) CheckAdminNode(ctx context.Context, id int) (AdminNodeChec return h.checkNodeView(ctx, node), nil } func (h *NodeHandler) checkNodeView(ctx context.Context, node *nodepool.Node) AdminNodeCheckView { - healthy, jobs, egress, hash, stats := nodepool.CheckNode(ctx, node) - err := h.repo.UpdateHealth(ctx, node.ID, node.URL, healthy, jobs, egress, stats) + healthy, jobs, egress, hash, stats, networkAccess := nodepool.CheckNode(ctx, node) + err := h.repo.UpdateHealth(ctx, node.ID, node.URL, healthy, jobs, egress, stats, networkAccess) if err != nil { slog.ErrorContext(ctx, "persisting health check result", "component", "api", "node_id", node.ID, "error", err) } - h.applyHealthToPools(node, healthy, jobs, egress, hash, stats) + h.applyHealthToPools(node, healthy, jobs, egress, hash, stats, networkAccess) return AdminNodeCheckView{Healthy: healthy, ActiveJobs: jobs, EgressKbps: egress, CapabilitiesHash: hash, HealthPersisted: err == nil} } func (h *NodeHandler) ReprobeAdminNode(w http.ResponseWriter, r *http.Request, id int) (ReprobeNodeResult, error) { diff --git a/internal/api/handlers/admin_node_commands_test.go b/internal/api/handlers/admin_node_commands_test.go index ae082dc3d9..8ccfc57f3d 100644 --- a/internal/api/handlers/admin_node_commands_test.go +++ b/internal/api/handlers/admin_node_commands_test.go @@ -8,6 +8,7 @@ import ( "testing" "time" + "github.com/Silo-Server/silo-server/internal/netaccess" "github.com/Silo-Server/silo-server/internal/nodepool" ) @@ -17,7 +18,7 @@ type commandNodeRepo struct { persistErr error } -func (r *commandNodeRepo) UpdateHealth(_ context.Context, _ int, url string, _ bool, _, _ int, _ []byte) error { +func (r *commandNodeRepo) UpdateHealth(_ context.Context, _ int, url string, _ bool, _, _ int, _ []byte, _ netaccess.NodeNetworkAccess) error { r.persistedURL = url return r.persistErr } diff --git a/internal/api/handlers/admin_node_network_access.go b/internal/api/handlers/admin_node_network_access.go new file mode 100644 index 0000000000..93d4b7a76b --- /dev/null +++ b/internal/api/handlers/admin_node_network_access.go @@ -0,0 +1,128 @@ +package handlers + +import ( + "context" + "encoding/json" + "fmt" + "io" + "net/http" + "net/url" + "time" + + "github.com/Silo-Server/silo-server/internal/netaccess" + "github.com/Silo-Server/silo-server/internal/nodepool" + "github.com/Silo-Server/silo-server/internal/plugins" + "github.com/Silo-Server/silo-server/internal/telemetry" +) + +// nodeNetworkAccessTimeout bounds one proxy's answer to a network access +// status read or command, like the node force-reload. +const nodeNetworkAccessTimeout = 10 * time.Second + +// maxNodeNetworkAccessBodyBytes bounds a proxy's answer; an honest status is +// under a kilobyte. +const maxNodeNetworkAccessBodyBytes = 64 << 10 + +// ListNetworkAccessNodes returns every enabled proxy node. Transcode nodes +// never run network access providers: clients never talk to them. +func (h *NodeHandler) ListNetworkAccessNodes(ctx context.Context) ([]plugins.NetworkAccessNode, error) { + if h == nil || h.repo == nil { + return nil, ErrAdminNodesUnavailable + } + all, err := h.repo.List(ctx) + if err != nil { + return nil, err + } + nodes := make([]plugins.NetworkAccessNode, 0, len(all)) + for _, node := range all { + if node == nil || !node.Enabled || node.Type != nodepool.NodeTypeProxy { + continue + } + nodes = append(nodes, plugins.NetworkAccessNode{ID: node.ID, Name: node.Name, URL: node.URL}) + } + return nodes, nil +} + +// NodeNetworkAccessStatus reads the provider's status from one proxy. +func (h *NodeHandler) NodeNetworkAccessStatus(ctx context.Context, node plugins.NetworkAccessNode, provider string) (netaccess.Status, error) { + var status netaccess.Status + err := h.nodeNetworkAccessCall(ctx, node, http.MethodGet, "/network-access/"+url.PathEscape(provider)+"/status", "status", &status) + return status, err +} + +// NodeNetworkAccessConnect brings the provider up on one proxy. +func (h *NodeHandler) NodeNetworkAccessConnect(ctx context.Context, node plugins.NetworkAccessNode, provider string) (netaccess.Status, error) { + var status netaccess.Status + err := h.nodeNetworkAccessCall(ctx, node, http.MethodPost, "/network-access/"+url.PathEscape(provider)+"/connect", "status", &status) + return status, err +} + +// NodeNetworkAccessDisconnect tears the provider down on one proxy. +func (h *NodeHandler) NodeNetworkAccessDisconnect(ctx context.Context, node plugins.NetworkAccessNode, provider string) (netaccess.Status, error) { + var status netaccess.Status + err := h.nodeNetworkAccessCall(ctx, node, http.MethodPost, "/network-access/"+url.PathEscape(provider)+"/disconnect", "status", &status) + return status, err +} + +// nodeNetworkAccessCall performs one bearer-authenticated request against a +// proxy's network-access routes and decodes the JSON answer into out. A 404 +// is the proxy saying the provider is not installed there and maps to +// plugins.ErrNetworkAccessProviderNotFound; every other failure carries the +// proxy's plain-text reason. +func (h *NodeHandler) nodeNetworkAccessCall(ctx context.Context, node plugins.NetworkAccessNode, method, path, operation string, out any) error { + if h == nil { + return ErrAdminNodesUnavailable + } + ctx, cancel := context.WithTimeout(ctx, nodeNetworkAccessTimeout) + defer cancel() + req, err := http.NewRequestWithContext(ctx, method, nodepool.NodeEndpoint(node.URL, path), nil) + if err != nil { + return err + } + req.Header.Set("Authorization", "Bearer "+h.jwtSecret) + req.Header.Set("Accept", "application/json") + client := &http.Client{Timeout: nodeNetworkAccessTimeout} + resp, err := telemetry.DoTrustedNode(client, req, operation) + if err != nil { + return fmt.Errorf("proxy node unreachable: %w", err) + } + defer func() { _ = resp.Body.Close() }() + body, err := io.ReadAll(io.LimitReader(resp.Body, maxNodeNetworkAccessBodyBytes+1)) + if err != nil { + return fmt.Errorf("read proxy node response: %w", err) + } + if len(body) > maxNodeNetworkAccessBodyBytes { + return fmt.Errorf("proxy node response exceeds %d bytes", maxNodeNetworkAccessBodyBytes) + } + switch resp.StatusCode { + case http.StatusOK: + case http.StatusNotFound: + return plugins.ErrNetworkAccessProviderNotFound + case http.StatusUnauthorized: + return fmt.Errorf("proxy node refused the node bearer") + case http.StatusServiceUnavailable: + return fmt.Errorf("proxy node does not host network access providers: %s", trimNodeReason(body)) + default: + return fmt.Errorf("proxy node answered %d: %s", resp.StatusCode, trimNodeReason(body)) + } + if err := json.Unmarshal(body, out); err != nil { + return fmt.Errorf("decode proxy node response: %w", err) + } + return nil +} + +// trimNodeReason keeps a proxy's plain-text failure short enough for a status +// row. +func trimNodeReason(body []byte) string { + const limit = 200 + reason := string(body) + for len(reason) > 0 && (reason[len(reason)-1] == '\n' || reason[len(reason)-1] == '\r') { + reason = reason[:len(reason)-1] + } + if len(reason) > limit { + reason = reason[:limit] + "…" + } + return reason +} + +var _ plugins.NetworkAccessNodes = (*NodeHandler)(nil) diff --git a/internal/api/handlers/admin_node_network_access_test.go b/internal/api/handlers/admin_node_network_access_test.go new file mode 100644 index 0000000000..858bb6f9da --- /dev/null +++ b/internal/api/handlers/admin_node_network_access_test.go @@ -0,0 +1,146 @@ +package handlers + +import ( + "context" + "encoding/json" + "errors" + "net/http" + "net/http/httptest" + "strings" + "sync" + "testing" + + "github.com/Silo-Server/silo-server/internal/netaccess" + "github.com/Silo-Server/silo-server/internal/nodepool" + "github.com/Silo-Server/silo-server/internal/plugins" +) + +// fakeProxyNode is an httptest stand-in for a proxy's bearer network-access +// routes: it records what it was asked, refuses a wrong bearer, and answers +// the status the API stores. +type fakeProxyNode struct { + secret string + mu sync.Mutex + calls []string + // connected flips on connect and off on disconnect. + connected bool +} + +func (f *fakeProxyNode) status() netaccess.Status { + status := netaccess.Status{InstallationID: 5, Provider: "stub", State: netaccess.StateDisconnected} + if f.connected { + status.State = netaccess.StateConnected + status.Origin = "https://proxy-1.stub.test" + status.AuthURL = "" + } + return status +} + +func (f *fakeProxyNode) ServeHTTP(w http.ResponseWriter, r *http.Request) { + f.mu.Lock() + defer f.mu.Unlock() + f.calls = append(f.calls, r.Method+" "+r.URL.Path) + if r.Header.Get("Authorization") != "Bearer "+f.secret { + http.Error(w, "unauthorized", http.StatusUnauthorized) + return + } + w.Header().Set("Content-Type", "application/json") + switch { + case r.Method == http.MethodGet && r.URL.Path == "/network-access/stub/status": + _ = json.NewEncoder(w).Encode(f.status()) + case r.Method == http.MethodGet && r.URL.Path == "/network-access/down/status": + _ = json.NewEncoder(w).Encode(netaccess.Status{InstallationID: 9, Provider: "down", State: netaccess.StateUnavailable, Error: "plugin process is failed"}) + case r.Method == http.MethodGet && r.URL.Path == "/network-access/status": + // Reading every provider could wait for an unrelated hung provider. + // API requests must use the provider-specific route instead. + http.Error(w, "unrelated provider did not answer", http.StatusGatewayTimeout) + case r.Method == http.MethodPost && r.URL.Path == "/network-access/stub/connect": + f.connected = true + _ = json.NewEncoder(w).Encode(f.status()) + case r.Method == http.MethodPost && r.URL.Path == "/network-access/stub/disconnect": + f.connected = false + _ = json.NewEncoder(w).Encode(f.status()) + case strings.HasPrefix(r.URL.Path, "/network-access/"): + http.Error(w, "network access provider not found", http.StatusNotFound) + default: + http.NotFound(w, r) + } +} + +// The node handler is the API's reach into proxy nodes: it lists only enabled +// proxies, calls the node's bearer routes with the node secret, decodes what +// the proxy answers, and maps a 404 to "provider not installed there". +func TestNodeHandlerFansNetworkAccessOutToProxies(t *testing.T) { + const secret = "node-bearer-secret" + proxy := &fakeProxyNode{secret: secret} + server := httptest.NewServer(proxy) + defer server.Close() + + repo := &stubNodeRepository{nodes: []*nodepool.Node{ + {ID: 3, Name: "proxy-1", Type: nodepool.NodeTypeProxy, URL: server.URL + "/", Enabled: true}, + {ID: 4, Name: "proxy-off", Type: nodepool.NodeTypeProxy, URL: "http://proxy-off", Enabled: false}, + {ID: 5, Name: "gpu-1", Type: nodepool.NodeTypeTranscode, URL: "http://gpu-1", Enabled: true}, + }} + handler := NewNodeHandler(repo, nil, nil, nil, nil, nil, secret) + ctx := context.Background() + + nodes, err := handler.ListNetworkAccessNodes(ctx) + if err != nil { + t.Fatal(err) + } + if len(nodes) != 1 || nodes[0].ID != 3 || nodes[0].Name != "proxy-1" || nodes[0].URL != server.URL+"/" { + t.Fatalf("nodes = %+v", nodes) + } + if host := nodes[0].Host(); host.ID != "node:3" || host.Role != "proxy" || host.Name != "proxy-1" { + t.Fatalf("host = %+v", host) + } + node := nodes[0] + + status, err := handler.NodeNetworkAccessStatus(ctx, node, "stub") + if err != nil || status.State != netaccess.StateDisconnected || status.InstallationID != 5 { + t.Fatalf("status = %+v, %v", status, err) + } + down, err := handler.NodeNetworkAccessStatus(ctx, node, "down") + if err != nil || down.State != netaccess.StateUnavailable || down.Error == "" { + t.Fatalf("down status = %+v, %v", down, err) + } + if _, err := handler.NodeNetworkAccessStatus(ctx, node, "netbird"); !errors.Is(err, plugins.ErrNetworkAccessProviderNotFound) { + t.Fatalf("unknown provider status err = %v", err) + } + + connected, err := handler.NodeNetworkAccessConnect(ctx, node, "stub") + if err != nil || connected.State != netaccess.StateConnected || connected.Origin != "https://proxy-1.stub.test" { + t.Fatalf("connect = %+v, %v", connected, err) + } + disconnected, err := handler.NodeNetworkAccessDisconnect(ctx, node, "stub") + if err != nil || disconnected.State != netaccess.StateDisconnected { + t.Fatalf("disconnect = %+v, %v", disconnected, err) + } + if _, err := handler.NodeNetworkAccessConnect(ctx, node, "netbird"); !errors.Is(err, plugins.ErrNetworkAccessProviderNotFound) { + t.Fatalf("unknown provider connect err = %v", err) + } + + want := []string{ + "GET /network-access/stub/status", "GET /network-access/down/status", "GET /network-access/netbird/status", + "POST /network-access/stub/connect", "POST /network-access/stub/disconnect", "POST /network-access/netbird/connect", + } + proxy.mu.Lock() + calls := append([]string(nil), proxy.calls...) + proxy.mu.Unlock() + if strings.Join(calls, ",") != strings.Join(want, ",") { + t.Fatalf("proxy calls = %v, want %v", calls, want) + } + + // A wrong node secret is reported as the proxy refusing the bearer, not as + // an unknown provider. + wrong := NewNodeHandler(repo, nil, nil, nil, nil, nil, "other-secret") + if _, err := wrong.NodeNetworkAccessStatus(ctx, node, "stub"); err == nil || errors.Is(err, plugins.ErrNetworkAccessProviderNotFound) || !strings.Contains(err.Error(), "bearer") { + t.Fatalf("wrong bearer err = %v", err) + } + + // An unreachable proxy is an error the service turns into unavailable. + server.Close() + if _, err := handler.NodeNetworkAccessStatus(ctx, node, "stub"); err == nil || !strings.Contains(err.Error(), "unreachable") { + t.Fatalf("unreachable err = %v", err) + } +} diff --git a/internal/api/handlers/admin_plugin_lifecycle.go b/internal/api/handlers/admin_plugin_lifecycle.go index 971b4758f9..f92abc1b6b 100644 --- a/internal/api/handlers/admin_plugin_lifecycle.go +++ b/internal/api/handlers/admin_plugin_lifecycle.go @@ -145,6 +145,36 @@ func (h *PluginHandler) ApplyAdminPluginUpdate(ctx context.Context, id int) (Plu return h.buildInstallationResponse(ctx, installation, nil) } +// RestartAdminPluginInstallation stops the installation's process and, for a +// resident plugin, starts it again with a fresh failure budget, so an +// administrator can bring back a resident the supervisor parked as failed. A +// non-resident plugin is only stopped; its next call launches it. A launch +// failure is reported through the returned view's runtime state rather than +// as an error. Errors: plugins.ErrInstallationNotFound, +// plugins.ErrInstallationDisabled, ErrPluginBuiltinInstallation. +func (h *PluginHandler) RestartAdminPluginInstallation(ctx context.Context, id int) (PluginInstallationView, error) { + if err := h.pluginLifecycleReady(); err != nil { + return PluginInstallationView{}, err + } + current, err := h.installations.GetByID(ctx, id) + if err != nil { + return PluginInstallationView{}, err + } + if current.IsBuiltin() { + return PluginInstallationView{}, ErrPluginBuiltinInstallation + } + if err := h.service.RestartInstallation(ctx, id); err != nil { + return PluginInstallationView{}, err + } + // The restart advanced runtime_generation and updated_at on the row; the + // receipt reflects the row as it is now, not the pre-restart snapshot. + restarted, err := h.installations.GetByID(ctx, id) + if err != nil { + return PluginInstallationView{}, err + } + return h.buildInstallationResponse(ctx, restarted, nil) +} + // DeleteAdminPluginInstallation stops the plugin, deletes its row (dependent // rows cascade) and removes its files. The row delete and the file removal are // not one transaction, and a repeat finds no row: a lost response is not a diff --git a/internal/api/handlers/admin_plugin_lifecycle_test.go b/internal/api/handlers/admin_plugin_lifecycle_test.go index a99d95ff50..5e8e887afd 100644 --- a/internal/api/handlers/admin_plugin_lifecycle_test.go +++ b/internal/api/handlers/admin_plugin_lifecycle_test.go @@ -23,6 +23,9 @@ func TestPluginLifecycleSeamRefusals(t *testing.T) { if _, err := unwired.ApplyAdminPluginUpdate(t.Context(), 1); !errors.As(err, &apiErr) || apiErr.Status != http.StatusServiceUnavailable { t.Fatalf("apply err = %v", err) } + if _, err := unwired.RestartAdminPluginInstallation(t.Context(), 1); !errors.As(err, &apiErr) || apiErr.Status != http.StatusServiceUnavailable { + t.Fatalf("restart err = %v", err) + } if err := unwired.DeleteAdminPluginInstallation(t.Context(), 1); !errors.As(err, &apiErr) || apiErr.Status != http.StatusServiceUnavailable { t.Fatalf("delete err = %v", err) } diff --git a/internal/api/handlers/downloads.go b/internal/api/handlers/downloads.go index 0a6029c32e..3c35a415f4 100644 --- a/internal/api/handlers/downloads.go +++ b/internal/api/handlers/downloads.go @@ -20,6 +20,7 @@ import ( "github.com/Silo-Server/silo-server/internal/catalog" "github.com/Silo-Server/silo-server/internal/downloads" "github.com/Silo-Server/silo-server/internal/httpstream" + "github.com/Silo-Server/silo-server/internal/netaccess" "github.com/Silo-Server/silo-server/internal/nodepool" "github.com/Silo-Server/silo-server/internal/playback" "github.com/Silo-Server/silo-server/internal/streamtoken" @@ -584,11 +585,19 @@ func (h *DownloadHandler) redirectToProxy(w http.ResponseWriter, r *http.Request reservationKey = fmt.Sprintf("direct-%d-%d", userID, target.MediaFileID) } sessionID := fmt.Sprintf("download-%s-%d", reservationKey, time.Now().UnixNano()) - plan := h.nodePlanner.PlanDownload(sessionID, target.OriginNodeGroup) + accessPath := netaccess.PathFromContext(r.Context()) + plan := h.nodePlanner.PlanDownloadWith(sessionID, func(node *nodepool.Node) bool { + return node.ClientURLFor(accessPath) != "" + }, target.OriginNodeGroup) if plan.ProxyNode == nil { return false, nil } releaseReservation := func() { h.nodePlanner.ReleaseSession(sessionID) } + clientBase := plan.ProxyNode.ClientURLFor(accessPath) + if clientBase == "" { + releaseReservation() + return false, nil + } mediaPath := target.Path downloadFilename := "" if target.OriginArtifactID != "" { @@ -624,15 +633,22 @@ func (h *DownloadHandler) redirectToProxy(w http.ResponseWriter, r *http.Request return false, fmt.Errorf("sign proxy download token: %w", err) } // The location is what the client downloads from, so it uses the proxy's - // client-facing URL; the cache key below stays on the canonical backend - // URL, which is the node's identity everywhere else. - location := strings.TrimRight(plan.ProxyNode.ClientURL(), "/") + "/downloads/file/" + url.PathEscape(token) + // client-facing URL for this access path. The preflight below dials the + // proxy's backend URL instead: that is the address this server reaches the + // node on (health sweeps, force-reload), while a client-facing origin may + // be unreachable from here — a tailnet origin resolves only on tailnet + // members, and this process need not be one. The verdict is about the + // proxy's ability to read the file, not about any one access path, so the + // cache key is the backend URL plus the target and is shared by every path. + tokenPath := "/downloads/file/" + url.PathEscape(token) + location := clientBase + tokenPath targetKey := target.Path if target.OriginArtifactID != "" { targetKey = target.OriginNodeURL + "\x00" + target.OriginArtifactID } - cacheKey := strings.TrimRight(plan.ProxyNode.URL, "/") + "\x00" + targetKey - if !h.proxyCanServe(r.Context(), cacheKey, location) { + backendBase := strings.TrimRight(plan.ProxyNode.URL, "/") + cacheKey := backendBase + "\x00" + targetKey + if !h.proxyCanServe(r.Context(), cacheKey, backendBase+tokenPath) { releaseReservation() return false, nil } @@ -647,7 +663,9 @@ func (h *DownloadHandler) redirectToProxy(w http.ResponseWriter, r *http.Request return true, nil } -func (h *DownloadHandler) proxyCanServe(ctx context.Context, cacheKey, location string) bool { +// proxyCanServe reports whether the proxy answers a HEAD for the signed token +// at probeURL, its backend address. Verdicts are cached briefly per cacheKey. +func (h *DownloadHandler) proxyCanServe(ctx context.Context, cacheKey, probeURL string) bool { now := time.Now() h.preflightMu.Lock() for key, cached := range h.preflightCache { @@ -661,7 +679,7 @@ func (h *DownloadHandler) proxyCanServe(ctx context.Context, cacheKey, location } h.preflightMu.Unlock() - req, err := http.NewRequestWithContext(ctx, http.MethodHead, location, nil) + req, err := http.NewRequestWithContext(ctx, http.MethodHead, probeURL, nil) if err != nil { return false } diff --git a/internal/api/handlers/downloads_network_access_test.go b/internal/api/handlers/downloads_network_access_test.go new file mode 100644 index 0000000000..8e2650c8bf --- /dev/null +++ b/internal/api/handlers/downloads_network_access_test.go @@ -0,0 +1,163 @@ +package handlers + +import ( + "net/http" + "net/http/httptest" + "strings" + "testing" + + "github.com/Silo-Server/silo-server/internal/downloads" + "github.com/Silo-Server/silo-server/internal/netaccess" + "github.com/Silo-Server/silo-server/internal/nodepool" +) + +// recordingDownloadPlanner hands out one proxy and records reservations, so a +// test can prove the reservation of a proxy the client cannot reach is given +// back rather than pinned until it ages out. +type recordingDownloadPlanner struct { + proxy *nodepool.Node + plans int + released []string +} + +func (p *recordingDownloadPlanner) PlanDownloadWith(string, func(*nodepool.Node) bool, ...string) nodepool.Plan { + p.plans++ + return nodepool.Plan{ProxyNode: p.proxy} +} + +func (p *recordingDownloadPlanner) ReleaseSession(sessionID string) { + p.released = append(p.released, sessionID) +} + +// A download requested through a network access provider is served from this +// server when the planned proxy has no origin on that provider: the proxy is +// never even preflighted, and its reservation is released. +func TestDirectDownloadViaProxyServesLocallyWhenTheProxyIsUnreachableOnTheProviderPath(t *testing.T) { + proxyRequests := 0 + proxy := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { + proxyRequests++ + w.WriteHeader(http.StatusOK) + })) + defer proxy.Close() + svc := &proxyDownloadService{ + fakeDownloadService: &fakeDownloadService{}, + directTarget: &downloads.FileTarget{Path: "/media/movie.mkv", MediaFileID: 42, ProxyEligible: true}, + } + planner := &recordingDownloadPlanner{proxy: &nodepool.Node{ID: 3, URL: proxy.URL, Enabled: true, Healthy: true}} + h := NewDownloadHandler(svc) + h.SetProxyDelivery(planner, func() string { return "secret" }) + + req := downloadTestRequest(http.MethodGet, "/direct-download-proxy?file_id=42", nil, 7, "", "") + req = req.WithContext(netaccess.WithPath(req.Context(), netaccess.Path{Provider: "tailscale"})) + rec := httptest.NewRecorder() + h.HandleDirectDownloadViaProxy(rec, req) + + if rec.Code != http.StatusOK || rec.Body.String() != "served" || svc.gotDirectFileID != 42 { + t.Fatalf("status = %d body = %q file = %d, want the file served locally", rec.Code, rec.Body.String(), svc.gotDirectFileID) + } + if rec.Header().Get("Location") != "" { + t.Fatalf("Location = %q, want none", rec.Header().Get("Location")) + } + if proxyRequests != 0 { + t.Fatalf("proxy was preflighted %d times for a client that cannot reach it", proxyRequests) + } + if planner.plans != 1 || len(planner.released) != 1 { + t.Fatalf("plans = %d releases = %v, want the unusable reservation released", planner.plans, planner.released) + } +} + +// With a connected origin on the client's provider the redirect names that +// origin, while the preflight dials the proxy's backend address: the overlay +// origin is only resolvable by overlay members, which this server need not be. +func TestDirectDownloadViaProxyRedirectsToTheProviderOrigin(t *testing.T) { + preflights := 0 + proxy := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + preflights++ + if r.Method != http.MethodHead { + t.Errorf("preflight method = %s", r.Method) + } + if !strings.HasPrefix(r.URL.Path, "/downloads/file/") { + t.Errorf("preflight path = %q", r.URL.Path) + } + w.WriteHeader(http.StatusOK) + })) + defer proxy.Close() + // The tailnet origin is undialable from the test on purpose; only the + // backend URL (the test server) may be probed. + const origin = "https://proxy-1.tail1234.ts.net" + svc := &proxyDownloadService{ + fakeDownloadService: &fakeDownloadService{}, + directTarget: &downloads.FileTarget{Path: "/media/movie.mkv", MediaFileID: 42, ProxyEligible: true}, + } + proxies := nodepool.NewProxyPool() + proxies.SetNodes([]*nodepool.Node{{ + ID: 3, URL: proxy.URL, Enabled: true, Healthy: true, + NetworkAccess: netaccess.NodeNetworkAccess{"tailscale": {State: netaccess.StateConnected, Origin: origin}}, + }}) + h := NewDownloadHandler(svc) + h.SetProxyDelivery(nodepool.NewPlanner(proxies, nodepool.NewTranscodePool()), func() string { return "secret" }) + + req := downloadTestRequest(http.MethodGet, "/direct-download-proxy?file_id=42", nil, 7, "", "") + req = req.WithContext(netaccess.WithPath(req.Context(), netaccess.Path{Provider: "tailscale"})) + rec := httptest.NewRecorder() + h.HandleDirectDownloadViaProxy(rec, req) + + if rec.Code != http.StatusTemporaryRedirect { + t.Fatalf("status = %d, want 307 (body: %s)", rec.Code, rec.Body.String()) + } + location := rec.Header().Get("Location") + if !strings.HasPrefix(location, origin+"/downloads/file/") || strings.Contains(location, proxy.URL) { + t.Fatalf("Location = %q, want the provider origin and no backend address", location) + } + if preflights != 1 { + t.Fatalf("preflights = %d, want 1", preflights) + } + + // The same proxy on the default path redirects to its backend URL (no + // public_url is set). The preflight verdict is about the proxy, not the + // path, so it is shared and not repeated. + rec = httptest.NewRecorder() + h.HandleDirectDownloadViaProxy(rec, downloadTestRequest(http.MethodGet, "/direct-download-proxy?file_id=42", nil, 7, "", "")) + if rec.Code != http.StatusTemporaryRedirect || !strings.HasPrefix(rec.Header().Get("Location"), proxy.URL+"/downloads/file/") { + t.Fatalf("default-path status = %d Location = %q, want the backend origin", rec.Code, rec.Header().Get("Location")) + } + if preflights != 1 { + t.Fatalf("preflights after the default-path request = %d, want the cached verdict reused", preflights) + } +} + +func TestDirectDownloadViaProxySkipsUnreachablePreferredGroup(t *testing.T) { + unreachable := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { + t.Error("proxy without a provider origin was preflighted") + w.WriteHeader(http.StatusOK) + })) + defer unreachable.Close() + reachable := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { + w.WriteHeader(http.StatusOK) + })) + defer reachable.Close() + const origin = "https://proxy.example.test" + proxies := nodepool.NewProxyPool() + proxies.SetNodes([]*nodepool.Node{ + {ID: 1, URL: unreachable.URL, Group: new("origin-host"), Enabled: true, Healthy: true}, + {ID: 2, URL: reachable.URL, Enabled: true, Healthy: true, + NetworkAccess: netaccess.NodeNetworkAccess{"tailscale": {State: netaccess.StateConnected, Origin: origin}}}, + }) + svc := &proxyDownloadService{ + fakeDownloadService: &fakeDownloadService{}, + directTarget: &downloads.FileTarget{ + Path: "/media/movie.mkv", MediaFileID: 42, ProxyEligible: true, OriginNodeGroup: "origin-host", + }, + } + h := NewDownloadHandler(svc) + h.SetProxyDelivery(nodepool.NewPlanner(proxies, nodepool.NewTranscodePool()), func() string { return "secret" }) + for range 4 { + req := downloadTestRequest(http.MethodGet, "/direct-download-proxy?file_id=42", nil, 7, "", "") + req = req.WithContext(netaccess.WithPath(req.Context(), netaccess.Path{Provider: "tailscale"})) + rec := httptest.NewRecorder() + h.HandleDirectDownloadViaProxy(rec, req) + if rec.Code != http.StatusTemporaryRedirect || !strings.HasPrefix(rec.Header().Get("Location"), origin+"/downloads/file/") { + t.Fatalf("status = %d Location = %q, want the reachable proxy", rec.Code, rec.Header().Get("Location")) + } + } +} diff --git a/internal/api/handlers/events_socket_v2.go b/internal/api/handlers/events_socket_v2.go index cf0031386a..d0083aa28f 100644 --- a/internal/api/handlers/events_socket_v2.go +++ b/internal/api/handlers/events_socket_v2.go @@ -35,9 +35,10 @@ type EventsSocketV2 struct { Tickets *evt.SocketTicketStore Validate EventsSocketValidator // PublicOrigin is the configured external origin, never a forwarded header. - PublicOrigin string - publicOrigin atomic.Pointer[string] - checkInterval time.Duration + PublicOrigin string + publicOrigin atomic.Pointer[string] + overlayOrigins atomic.Pointer[OverlayOriginSource] + checkInterval time.Duration } func (h *EventsSocketV2) Mint(ctx context.Context, identity evt.SocketIdentity) (string, error) { @@ -132,7 +133,7 @@ func (h *EventsSocketV2) ServeHTTP(w http.ResponseWriter, r *http.Request) { } func (h *EventsSocketV2) validOrigin(r *http.Request) bool { - return socketOriginAllowed(r, h.currentPublicOrigin()) + return socketOriginAllowed(r, h.currentPublicOrigin(), overlayOriginsFrom(h.overlayOrigins.Load())) } func (h *EventsSocketV2) currentPublicOrigin() string { @@ -148,7 +149,30 @@ func (h *EventsSocketV2) SetPublicOrigin(origin string) { h.publicOrigin.Store(&normalized) } -func socketOriginAllowed(r *http.Request, publicOrigin string) bool { +// SetOverlayOrigins installs the source of overlay origins (connected +// network access providers on this host) accepted next to the public origin. +func (h *EventsSocketV2) SetOverlayOrigins(source OverlayOriginSource) { + h.overlayOrigins.Store(&source) +} + +// OverlayOriginSource lists the scheme://host[:port] origins of the network +// access providers currently connected on this host. netaccess.StatusCache's +// ConnectedOrigins is the production source; it is read per handshake so a +// provider that connects or drops is reflected without a config reload. +type OverlayOriginSource func() []string + +func overlayOriginsFrom(source *OverlayOriginSource) []string { + if source == nil || *source == nil { + return nil + } + return (*source)() +} + +// socketOriginAllowed accepts a browser Origin that matches the configured +// public origin (or, when none is configured, the request's own scheme and +// host) or one of the overlay origins connected providers report. Both are +// exact scheme and host matches; forwarded headers are never consulted here. +func socketOriginAllowed(r *http.Request, publicOrigin string, overlayOrigins []string) bool { origins := r.Header.Values("Origin") if len(origins) == 0 { return true @@ -168,8 +192,20 @@ func socketOriginAllowed(r *http.Request, publicOrigin string) bool { } expected = scheme + "://" + r.Host } + if originMatches(origin, expected) { + return true + } + for _, overlay := range overlayOrigins { + if originMatches(origin, overlay) { + return true + } + } + return false +} + +func originMatches(origin *url.URL, expected string) bool { target, err := url.Parse(expected) - return err == nil && strings.EqualFold(origin.Host, target.Host) && origin.Scheme == target.Scheme + return err == nil && strings.EqualFold(origin.Host, target.Host) && strings.EqualFold(origin.Scheme, target.Scheme) } type eventsSessionValidator interface { diff --git a/internal/api/handlers/nodes.go b/internal/api/handlers/nodes.go index 8d389e4b40..ebe55ae387 100644 --- a/internal/api/handlers/nodes.go +++ b/internal/api/handlers/nodes.go @@ -20,6 +20,7 @@ import ( "github.com/Silo-Server/silo-server/internal/cache" "github.com/Silo-Server/silo-server/internal/config" "github.com/Silo-Server/silo-server/internal/logredact" + "github.com/Silo-Server/silo-server/internal/netaccess" "github.com/Silo-Server/silo-server/internal/nodepool" "github.com/Silo-Server/silo-server/internal/playback" "github.com/go-chi/chi/v5" @@ -33,7 +34,7 @@ type NodeRepository interface { Create(ctx context.Context, input nodepool.CreateNodeInput) (*nodepool.Node, error) Update(ctx context.Context, id int, input nodepool.UpdateNodeInput) (*nodepool.Node, error) Delete(ctx context.Context, id int) error - UpdateHealth(ctx context.Context, id int, checkedURL string, healthy bool, activeJobs, egressKbps int, lastStats []byte) error + UpdateHealth(ctx context.Context, id int, checkedURL string, healthy bool, activeJobs, egressKbps int, lastStats []byte, networkAccess netaccess.NodeNetworkAccess) error } // NodeListEnabled queries enabled nodes by type for pool reload. @@ -473,17 +474,17 @@ func (h *NodeHandler) HandleCheckNode(w http.ResponseWriter, r *http.Request) { // Fenced on the node's URL by the pools themselves, like every other health // write: the row can be repointed while a check is in flight. func (h *NodeHandler) applyHealthToPools( - node *nodepool.Node, healthy bool, activeJobs, egressKbps int, capabilitiesHash string, lastStats []byte, + node *nodepool.Node, healthy bool, activeJobs, egressKbps int, capabilitiesHash string, lastStats []byte, networkAccess netaccess.NodeNetworkAccess, ) { checkedAt := time.Now() switch node.Type { case nodepool.NodeTypeProxy: if h.proxyPool != nil { - h.proxyPool.ApplyHealth(node.ID, node.URL, healthy, activeJobs, egressKbps, capabilitiesHash, lastStats, checkedAt) + h.proxyPool.ApplyHealth(node.ID, node.URL, healthy, activeJobs, egressKbps, capabilitiesHash, lastStats, networkAccess, checkedAt) } case nodepool.NodeTypeTranscode: if h.transcodePool != nil { - h.transcodePool.ApplyHealth(node.ID, node.URL, healthy, activeJobs, egressKbps, capabilitiesHash, lastStats, checkedAt) + h.transcodePool.ApplyHealth(node.ID, node.URL, healthy, activeJobs, egressKbps, capabilitiesHash, lastStats, networkAccess, checkedAt) } } } diff --git a/internal/api/handlers/nodes_test.go b/internal/api/handlers/nodes_test.go index b6a715b57a..0082146dc8 100644 --- a/internal/api/handlers/nodes_test.go +++ b/internal/api/handlers/nodes_test.go @@ -12,6 +12,7 @@ import ( "time" "github.com/Silo-Server/silo-server/internal/cache" + "github.com/Silo-Server/silo-server/internal/netaccess" "github.com/Silo-Server/silo-server/internal/nodepool" "github.com/go-chi/chi/v5" ) @@ -54,7 +55,7 @@ func (s *stubNodeRepository) Update(_ context.Context, _ int, input nodepool.Upd func (s *stubNodeRepository) Delete(context.Context, int) error { return nil } -func (s *stubNodeRepository) UpdateHealth(context.Context, int, string, bool, int, int, []byte) error { +func (s *stubNodeRepository) UpdateHealth(context.Context, int, string, bool, int, int, []byte, netaccess.NodeNetworkAccess) error { return nil } @@ -663,8 +664,8 @@ func TestHandleListNodesCarriesTheAdvertisedHashFromThePools(t *testing.T) { proxies := nodepool.NewProxyPool() proxies.SetNodes([]*nodepool.Node{{ID: 2, URL: "http://proxy-1", Enabled: true}}) // The sweep learns each node's advertised hash on its health check. - transcodes.ApplyHealth(1, "http://gpu-1", true, 0, 0, "sha256:newer", nil, time.Now()) - proxies.ApplyHealth(2, "http://proxy-1", true, 0, 0, "sha256:proxy", nil, time.Now()) + transcodes.ApplyHealth(1, "http://gpu-1", true, 0, 0, "sha256:newer", nil, nil, time.Now()) + proxies.ApplyHealth(2, "http://proxy-1", true, 0, 0, "sha256:proxy", nil, nil, time.Now()) handler := NewNodeHandler(repo, proxies, transcodes, nil, nil, nil, "secret") recorder := httptest.NewRecorder() @@ -799,7 +800,7 @@ func TestReadAdminNodesDoesNotMutateRepositoryRows(t *testing.T) { repo := &stubNodeRepository{nodes: []*nodepool.Node{stored}} pool := nodepool.NewProxyPool() pool.SetNodes([]*nodepool.Node{stored}) - pool.ApplyHealth(1, stored.URL, true, 0, 0, "advertised", nil, time.Now()) + pool.ApplyHealth(1, stored.URL, true, 0, 0, "advertised", nil, nil, time.Now()) rows, err := NewNodeHandler(repo, pool, nil, nil, nil, nil, "").ReadAdminNodes(t.Context()) if err != nil || len(rows) != 1 || rows[0] == stored || stored.AdvertisedCapabilitiesHash != nil || rows[0].AdvertisedCapabilitiesHash == nil || *rows[0].AdvertisedCapabilitiesHash != "advertised" { t.Fatal(rows, stored, err) diff --git a/internal/api/handlers/playback.go b/internal/api/handlers/playback.go index b0074d121d..fc19e35f2d 100644 --- a/internal/api/handlers/playback.go +++ b/internal/api/handlers/playback.go @@ -26,6 +26,7 @@ import ( "github.com/Silo-Server/silo-server/internal/httpstream" "github.com/Silo-Server/silo-server/internal/markers" "github.com/Silo-Server/silo-server/internal/models" + "github.com/Silo-Server/silo-server/internal/netaccess" "github.com/Silo-Server/silo-server/internal/nodepool" "github.com/Silo-Server/silo-server/internal/noderouting" "github.com/Silo-Server/silo-server/internal/playback" @@ -797,6 +798,7 @@ func identityRecipeCard(s *playback.Session) playback.RecipeCard { card = playback.NewDirectRecipeCard(s.ID, s.UserID, s.ProfileID, s.MediaFileID) } card.OriginalStartedAt = s.StartedAt + card.RoutingNetworkProvider = s.RoutingNetworkProvider card.RoutingWorkload = s.RoutingWorkload card.RoutingExecution = s.RoutingExecution card.RoutingExecutionNodeID = s.RoutingExecutionNodeID @@ -1963,9 +1965,13 @@ func (h *PlaybackHandler) HandleGetTranscodeSegment(w http.ResponseWriter, r *ht // never receives a token, and therefore never a proxy origin either, since a // proxy authenticates from the token in the URL path alone. It gets the // API-local manifest path, which the client fetches with its own credential. -func (h *PlaybackHandler) buildProxyManifestURL(card playback.RecipeCard, proxyNode *nodepool.Node, requireMediaAuth bool) string { +// +// path is the client's access path. A proxy with no origin on it is the same +// as no proxy: the manifest stays API-local and this server relays the node. +func (h *PlaybackHandler) buildProxyManifestURL(card playback.RecipeCard, proxyNode *nodepool.Node, requireMediaAuth bool, path netaccess.Path) string { localURL := fmt.Sprintf("/playback/transcode/%s/master.m3u8", card.SessionID) - if proxyNode == nil { + base := proxyNode.ClientURLFor(path) + if base == "" { return appendStreamToken(localURL, h.signSessionToken(card, requireMediaAuth)) } card.RoutingEgressNodeID = proxyNode.ID @@ -1973,7 +1979,7 @@ func (h *PlaybackHandler) buildProxyManifestURL(card playback.RecipeCard, proxyN if token == "" { return appendStreamToken(localURL, token) } - return nodepool.NodeEndpoint(proxyNode.ClientURL(), "/stream/transcode/"+token+"/master.m3u8") + return nodepool.NodeEndpoint(base, "/stream/transcode/"+token+"/master.m3u8") } // proxyToTranscodeNode forwards a request to the remote transcode node. diff --git a/internal/api/handlers/playback_control_socket_v2.go b/internal/api/handlers/playback_control_socket_v2.go index d062ce4750..1852b0f53d 100644 --- a/internal/api/handlers/playback_control_socket_v2.go +++ b/internal/api/handlers/playback_control_socket_v2.go @@ -184,9 +184,10 @@ type PlaybackControlSocketV2 struct { Tickets *PlaybackControlTicketStore Validate EventsSocketValidator // PublicOrigin is the configured external origin, never a forwarded header. - PublicOrigin string - publicOrigin atomic.Pointer[string] - checkInterval time.Duration + PublicOrigin string + publicOrigin atomic.Pointer[string] + overlayOrigins atomic.Pointer[OverlayOriginSource] + checkInterval time.Duration laneMu sync.Mutex lanes map[string]*playbackControlLane @@ -277,7 +278,7 @@ func (h *PlaybackControlSocketV2) ServeHTTP(w http.ResponseWriter, r *http.Reque http.Error(w, "invalid handshake", http.StatusBadRequest) return } - if !socketOriginAllowed(r, h.currentPublicOrigin()) { + if !socketOriginAllowed(r, h.currentPublicOrigin(), overlayOriginsFrom(h.overlayOrigins.Load())) { http.Error(w, "origin refused", http.StatusForbidden) return } @@ -334,7 +335,9 @@ func (h *PlaybackControlSocketV2) ServeHTTP(w http.ResponseWriter, r *http.Reque ctx, cancel := context.WithDeadline(validated, deadline) defer cancel() - upgrader := websocket.Upgrader{Subprotocols: []string{PlaybackControlSocketProtocol}, CheckOrigin: func(r *http.Request) bool { return socketOriginAllowed(r, h.currentPublicOrigin()) }} + upgrader := websocket.Upgrader{Subprotocols: []string{PlaybackControlSocketProtocol}, CheckOrigin: func(r *http.Request) bool { + return socketOriginAllowed(r, h.currentPublicOrigin(), overlayOriginsFrom(h.overlayOrigins.Load())) + }} conn, err := upgrader.Upgrade(w, r, nil) if err != nil { slog.ErrorContext(r.Context(), "playback control websocket upgrade failed", "component", "api", "error", err, "session", sessionID, "playback_session_id", sessionID) @@ -426,6 +429,12 @@ func (h *PlaybackControlSocketV2) SetPublicOrigin(origin string) { h.publicOrigin.Store(&normalized) } +// SetOverlayOrigins installs the source of overlay origins accepted next to +// the public origin. +func (h *PlaybackControlSocketV2) SetOverlayOrigins(source OverlayOriginSource) { + h.overlayOrigins.Store(&source) +} + func (h *PlaybackControlSocketV2) owns(sessionID string, lane *playbackControlLane) bool { h.laneMu.Lock() defer h.laneMu.Unlock() diff --git a/internal/api/handlers/playback_sessions.go b/internal/api/handlers/playback_sessions.go index 3c4784e3f1..0286601c49 100644 --- a/internal/api/handlers/playback_sessions.go +++ b/internal/api/handlers/playback_sessions.go @@ -65,33 +65,34 @@ type playbackSessionRow struct { // the target codec with no channel layout rather than reusing // SourceAudioChannels, which is what made a 7.1 source downmixed to AAC 5.1 // read as "AAC 7.1". - TargetAudioChannels *int `json:"target_audio_channels,omitempty"` - TargetBitrateKbps *int `json:"target_bitrate_kbps"` - TranscodeHWAccel string `json:"transcode_hw_accel,omitempty"` - ToneMapMode string `json:"tone_map_mode,omitempty"` - SourceContainer string `json:"source_container,omitempty"` - SourceBitrateKbps *int `json:"source_bitrate_kbps"` - SourceVideoCodec string `json:"source_video_codec,omitempty"` - SourceVideoResolution string `json:"source_video_resolution,omitempty"` - SourceAudioCodec string `json:"source_audio_codec,omitempty"` - SourceAudioChannels *int `json:"source_audio_channels"` - SourceAudioLanguage string `json:"source_audio_language,omitempty"` - SourceAudioTitle string `json:"source_audio_title,omitempty"` - SourceAudioLayout string `json:"source_audio_layout,omitempty"` - RequestedVideoCodec string `json:"requested_video_codec,omitempty"` - RequestedVideoResolution string `json:"requested_video_resolution,omitempty"` - VideoDecision string `json:"video_decision,omitempty"` - AudioDecision string `json:"audio_decision,omitempty"` - EffectivePlayMethod string `json:"effective_play_method,omitempty"` - IsJellyfinClient bool `json:"is_jellyfin_client,omitempty"` - RoutingWorkload string `json:"routing_workload,omitempty"` - RoutingExecution string `json:"routing_execution,omitempty"` - RoutingExecutionNodeID *int `json:"routing_execution_node_id,omitempty"` - RoutingExecutionNodeName string `json:"routing_execution_node_name,omitempty"` - RoutingEgress string `json:"routing_egress,omitempty"` - RoutingEgressNodeID *int `json:"routing_egress_node_id,omitempty"` - RoutingEgressNodeName string `json:"routing_egress_node_name,omitempty"` - CompatOrigin bool `json:"-"` + TargetAudioChannels *int `json:"target_audio_channels,omitempty"` + TargetBitrateKbps *int `json:"target_bitrate_kbps"` + TranscodeHWAccel string `json:"transcode_hw_accel,omitempty"` + ToneMapMode string `json:"tone_map_mode,omitempty"` + SourceContainer string `json:"source_container,omitempty"` + SourceBitrateKbps *int `json:"source_bitrate_kbps"` + SourceVideoCodec string `json:"source_video_codec,omitempty"` + SourceVideoResolution string `json:"source_video_resolution,omitempty"` + SourceAudioCodec string `json:"source_audio_codec,omitempty"` + SourceAudioChannels *int `json:"source_audio_channels"` + SourceAudioLanguage string `json:"source_audio_language,omitempty"` + SourceAudioTitle string `json:"source_audio_title,omitempty"` + SourceAudioLayout string `json:"source_audio_layout,omitempty"` + RequestedVideoCodec string `json:"requested_video_codec,omitempty"` + RequestedVideoResolution string `json:"requested_video_resolution,omitempty"` + VideoDecision string `json:"video_decision,omitempty"` + AudioDecision string `json:"audio_decision,omitempty"` + EffectivePlayMethod string `json:"effective_play_method,omitempty"` + IsJellyfinClient bool `json:"is_jellyfin_client,omitempty"` + RoutingNetworkProvider *string `json:"-"` + RoutingWorkload string `json:"routing_workload,omitempty"` + RoutingExecution string `json:"routing_execution,omitempty"` + RoutingExecutionNodeID *int `json:"routing_execution_node_id,omitempty"` + RoutingExecutionNodeName string `json:"routing_execution_node_name,omitempty"` + RoutingEgress string `json:"routing_egress,omitempty"` + RoutingEgressNodeID *int `json:"routing_egress_node_id,omitempty"` + RoutingEgressNodeName string `json:"routing_egress_node_name,omitempty"` + CompatOrigin bool `json:"-"` } // playbackSessionsCapabilitiesResponse advertises the additive fields of the @@ -315,7 +316,8 @@ func (l *PlaybackSessionsLoader) load(ctx context.Context, query PlaybackSession COALESCE(execution_node.name, ''), COALESCE(s.routing_egress, ''), s.routing_egress_node_id, - COALESCE(egress_node.name, '') + COALESCE(egress_node.name, ''), + s.routing_network_provider FROM playback_sessions_sync s LEFT JOIN users u ON u.id = s.user_id LEFT JOIN media_files mf ON mf.id = s.media_file_id @@ -385,7 +387,7 @@ func (l *PlaybackSessionsLoader) load(ctx context.Context, query PlaybackSession &s.TranscodeHWAccel, &s.ToneMapMode, &s.SourceContainer, &sourceBitrateKbps, &s.SourceVideoCodec, &s.SourceVideoResolution, &s.SourceAudioCodec, &sourceAudioChannels, &audioTracksJSON, &s.RequestedVideoCodec, &s.RequestedVideoResolution, &s.CompatOrigin, &s.RoutingWorkload, &s.RoutingExecution, &s.RoutingExecutionNodeID, - &s.RoutingExecutionNodeName, &s.RoutingEgress, &s.RoutingEgressNodeID, &s.RoutingEgressNodeName, + &s.RoutingExecutionNodeName, &s.RoutingEgress, &s.RoutingEgressNodeID, &s.RoutingEgressNodeName, &s.RoutingNetworkProvider, ); err != nil { return nil, 0, fmt.Errorf("scanning playback session: %w", err) } diff --git a/internal/api/handlers/playback_v3.go b/internal/api/handlers/playback_v3.go index 0b336b358d..93c4d4d44f 100644 --- a/internal/api/handlers/playback_v3.go +++ b/internal/api/handlers/playback_v3.go @@ -29,6 +29,7 @@ import ( "github.com/Silo-Server/silo-server/internal/config" "github.com/Silo-Server/silo-server/internal/logredact" "github.com/Silo-Server/silo-server/internal/models" + "github.com/Silo-Server/silo-server/internal/netaccess" "github.com/Silo-Server/silo-server/internal/nodepool" "github.com/Silo-Server/silo-server/internal/noderouting" "github.com/Silo-Server/silo-server/internal/playback" @@ -1375,13 +1376,19 @@ func (h *PlaybackHandler) resolveHLSRouteWithPolicyV3( return noderouting.Decision{Outcome: noderouting.OutcomePolicyUnsatisfied} } eligible := h.transcodeEligibilityV3(ctx, result, excludedNodes) + // A client that arrived through a network access provider can only use a + // proxy that has a connected origin on that same overlay; on the default + // path this is nil and every healthy proxy stays eligible. Filtering here + // rather than at the URL builder keeps the resolver's fallbacks (API + // egress, API relay to the transcode node) in charge of what happens next. + proxyEligible := nodepool.ClientReachableVia(netaccess.PathFromContext(ctx), nil) decision, err := noderouting.Resolve(noderouting.AdaptSessionPlanner(h.NodePlanner), noderouting.ResolveRequest{ Request: noderouting.Request{ Workload: workload, Delivery: delivery, Policy: policy, ProxyAllowed: proxyAllowed, }, SessionID: session.ID, CurrentTranscodeURL: session.TranscodeNodeURL, EstimatedBitrateKbps: result.TargetBitrateKbps, - TranscodeEligible: eligible, ExcludedShapeIDs: excludedShapes, + TranscodeEligible: eligible, ProxyEligible: proxyEligible, ExcludedShapeIDs: excludedShapes, }) if err != nil { slog.ErrorContext(ctx, "compile playback node route", "component", "noderouting", "error", err) @@ -2580,6 +2587,7 @@ func (h *PlaybackHandler) prepareIdentityTransportV3(r *http.Request, session *p // proxy's row or internal URL leak into an API-served replacement route. routeSession.RoutingEgressNodeID = 0 routeSession.RoutingEgressNodeURL = "" + routeSession.RoutingNetworkProvider = new(netaccess.PathFromContext(r.Context()).Provider) routeSession.RoutingWorkload = string(routingWorkloadV3(result)) routeSession.RoutingExecution = string(decision.Shape.Execution) routeSession.RoutingEgress = string(decision.Shape.Egress) @@ -2621,11 +2629,12 @@ func (h *PlaybackHandler) prepareIdentityTransportV3(r *http.Request, session *p // back: rolling back to the restored old plan leaves that plan's published // proxy URL live, and a deleted grant would 404 it. var priorGrant *playback.RecipeCard + accessPath := netaccess.PathFromContext(r.Context()) switch { case !mode.headerAuth: - streamURL, servedByProxy = h.identityStreamURLV3(&routeSession, file, proxyNode) + streamURL, servedByProxy = h.identityStreamURLV3(&routeSession, file, proxyNode, accessPath) case mode.proxyEgress: - streamURL, servedByProxy, priorGrant = h.identityGrantStreamURLV3(r.Context(), &routeSession, file, proxyNode) + streamURL, servedByProxy, priorGrant = h.identityGrantStreamURLV3(r.Context(), &routeSession, file, proxyNode, accessPath) } reservationReleased := false if proxyNode != nil && !servedByProxy { @@ -2804,10 +2813,17 @@ func (h *PlaybackHandler) revokeStaleProxyGrantOnCommitV3(ctx context.Context, s // which is exactly the behavior of a header-authenticated attempt that // negotiated no origins at all. The third value is the grant this write // displaced, for the caller's rollback. -func (h *PlaybackHandler) identityGrantStreamURLV3(ctx context.Context, s *playback.Session, file *models.MediaFile, proxyNode *nodepool.Node) (string, bool, *playback.RecipeCard) { +// +// path is the client's access path: a proxy with no origin on it is treated +// like no proxy at all, before any grant is written, so the API relays. +func (h *PlaybackHandler) identityGrantStreamURLV3(ctx context.Context, s *playback.Session, file *models.MediaFile, proxyNode *nodepool.Node, path netaccess.Path) (string, bool, *playback.RecipeCard) { if proxyNode == nil || file == nil || s == nil { return h.playbackStreamURL(s), false, nil } + base := proxyNode.ClientURLFor(path) + if base == "" { + return h.playbackStreamURL(s), false, nil + } card := identityRecipeCard(s) card.InputPath = file.FilePath card.DVProfile = file.PrimaryDVProfile() @@ -2817,7 +2833,7 @@ func (h *PlaybackHandler) identityGrantStreamURLV3(ctx context.Context, s *playb if !stored { return h.playbackStreamURL(s), false, nil } - return strings.TrimRight(proxyNode.ClientURL(), "/") + "/stream/v3/" + s.ID, true, prior + return base + "/stream/v3/" + s.ID, true, prior } // putProxyGrantV3 stores the recipe a designated proxy origin serves this @@ -2988,14 +3004,19 @@ func (h *PlaybackHandler) resolveIdentityRouteV3(r *http.Request, sessionID stri if delivery == noderouting.DeliveryProgressiveRemux { relayEligible = h.identityProxyRelayEligibilityV3(r.Context()) } + // Both proxy predicates are narrowed to proxies the client can actually + // reach on its access path (see resolveHLSRouteWithPolicyV3). With no + // reachable proxy the resolver falls through to the API-egress shapes, or + // reports capacity unavailable under a proxy_only policy. + accessPath := netaccess.PathFromContext(r.Context()) decision, err := noderouting.Resolve(noderouting.AdaptSessionPlanner(h.NodePlanner), noderouting.ResolveRequest{ Request: noderouting.Request{ Workload: workload, Delivery: delivery, Policy: policy, ProxyAllowed: proxyAllowed, }, SessionID: sessionID, EstimatedBitrateKbps: identityStreamBitrateKbpsV3(result), - TranscodeEligible: h.transcodeEligibilityV3(r.Context(), result, nil), ProxyEligible: relayEligible, - ProxyExecutionEligible: h.identityProxyEligibilityV3(r.Context(), result), ExcludedShapeIDs: excludedShapes, + TranscodeEligible: h.transcodeEligibilityV3(r.Context(), result, nil), ProxyEligible: nodepool.ClientReachableVia(accessPath, relayEligible), + ProxyExecutionEligible: nodepool.ClientReachableVia(accessPath, h.identityProxyEligibilityV3(r.Context(), result)), ExcludedShapeIDs: excludedShapes, }) if err != nil { return noderouting.Decision{}, &transportErrorV3{reason: string(noderouting.OutcomePolicyUnsatisfied), message: "The playback routing policy is invalid.", retryable: false, cause: err} @@ -3254,10 +3275,17 @@ func identityStreamBitrateKbpsV3(result playback.PlannerResultV3) int { // serves from the signed token alone, which is exactly the credential that mode // keeps out of client-visible URLs. It falls back to the API-local path, whose // builder omits the token for the same reason. -func (h *PlaybackHandler) identityStreamURLV3(s *playback.Session, file *models.MediaFile, proxyNode *nodepool.Node) (string, bool) { +// +// path is the client's access path; a proxy with no origin on it falls back +// the same way, and the caller releases the reservation. +func (h *PlaybackHandler) identityStreamURLV3(s *playback.Session, file *models.MediaFile, proxyNode *nodepool.Node, path netaccess.Path) (string, bool) { if proxyNode == nil || file == nil || (s != nil && s.RequireMediaAuthorization) { return h.playbackStreamURL(s), false } + base := proxyNode.ClientURLFor(path) + if base == "" { + return h.playbackStreamURL(s), false + } card := identityRecipeCard(s) card.InputPath = file.FilePath card.RoutingEgressNodeID = proxyNode.ID @@ -3268,7 +3296,6 @@ func (h *PlaybackHandler) identityStreamURLV3(s *playback.Session, file *models. if token == "" { return h.playbackStreamURL(s), false } - base := strings.TrimRight(proxyNode.ClientURL(), "/") if s.PlayMethod == playback.PlayRemux { if claims.PlayMethod == streamtoken.PlayMethodAudioDownmixRemux { return base + "/stream/remux/audio-v2/" + token, true @@ -3676,6 +3703,7 @@ func (h *PlaybackHandler) prepareLocalTransportV3(r *http.Request, session *play if !mode.headerAuth { card := playback.NewRecipeCard(session.UserID, session.ProfileID, file.ID, "", ts.Opts()) card.OriginalStartedAt = session.StartedAt + card.RoutingNetworkProvider = new(netaccess.PathFromContext(r.Context()).Provider) card.RoutingWorkload = string(routingWorkloadV3(result)) card.RoutingExecution = string(noderouting.ExecutionAPI) card.RoutingEgress = string(noderouting.EgressAPI) @@ -3869,6 +3897,8 @@ func (h *PlaybackHandler) prepareRemoteTransportV3(r *http.Request, session *pla } card := remoteTranscodeRecipeCardV3(session, file, node.URL, transportID, req, nodeResp, toneMapFilter) card.RoutingWorkload = string(routingWorkloadV3(result)) + card.RoutingNetworkProvider = new(netaccess.PathFromContext(r.Context()).Provider) + card.RoutingExecutionNodeID = node.ID card.RoutingExecution = string(noderouting.ExecutionTranscode) card.RoutingEgress = string(noderouting.EgressAPI) if nodePlan.ProxyNode != nil { @@ -3885,12 +3915,13 @@ func (h *PlaybackHandler) prepareRemoteTransportV3(r *http.Request, session *pla // See prepareIdentityTransportV3: the displaced grant is what a failed // replan of an already-proxy-served session has to put back. var priorGrant *playback.RecipeCard + accessPath := netaccess.PathFromContext(r.Context()) switch { case !mode.headerAuth: - url = h.buildProxyManifestURL(card, nodePlan.ProxyNode, mode.headerAuth) + url = h.buildProxyManifestURL(card, nodePlan.ProxyNode, mode.headerAuth, accessPath) servedByProxy = nodePlan.ProxyNode != nil && strings.HasPrefix(url, "http") case mode.proxyEgress: - url, servedByProxy, priorGrant = h.grantManifestURLV3(r.Context(), card, nodePlan.ProxyNode) + url, servedByProxy, priorGrant = h.grantManifestURLV3(r.Context(), card, nodePlan.ProxyNode, accessPath) } if mode.headerAuth { // No client-visible URL carries a stream token in this mode, so neither the @@ -4010,20 +4041,25 @@ func remoteTranscodeRecipeCardV3(session *playback.Session, file *models.MediaFi // credential-free manifest URL on that origin. Segment URIs stay relative to // the manifest, so the same /stream/v3/{session_id}/... family serves both. // -// Without a planned proxy — or when the grant cannot be stored — the client -// fetches the manifest from this server, which relays the same node. The third -// value is the grant this write displaced, for the caller's rollback. -func (h *PlaybackHandler) grantManifestURLV3(ctx context.Context, card playback.RecipeCard, proxyNode *nodepool.Node) (string, bool, *playback.RecipeCard) { +// Without a planned proxy — or a proxy the client cannot reach on its access +// path, or when the grant cannot be stored — the client fetches the manifest +// from this server, which relays the same node. The third value is the grant +// this write displaced, for the caller's rollback. +func (h *PlaybackHandler) grantManifestURLV3(ctx context.Context, card playback.RecipeCard, proxyNode *nodepool.Node, path netaccess.Path) (string, bool, *playback.RecipeCard) { localURL := fmt.Sprintf("/playback/transcode/%s/master.m3u8", card.SessionID) if proxyNode == nil { return localURL, false, nil } + base := proxyNode.ClientURLFor(path) + if base == "" { + return localURL, false, nil + } card.RoutingEgressNodeID = proxyNode.ID prior, stored := h.putProxyGrantV3(ctx, card.SessionID, card) if !stored { return localURL, false, nil } - return strings.TrimRight(proxyNode.ClientURL(), "/") + "/stream/v3/" + card.SessionID + "/master.m3u8", true, prior + return base + "/stream/v3/" + card.SessionID + "/master.m3u8", true, prior } // sourceExecutionMetadataV3 freezes the source facts used by a remote executor. @@ -4094,6 +4130,7 @@ func (h *PlaybackHandler) v3SessionStreamState(ctx context.Context, session *pla TranscodeNodeURL: transport.nodeURL, TranscodeTransportID: transport.transportID, TranscodeRouteSet: true, + RoutingNetworkProvider: new(netaccess.PathFromContext(ctx).Provider), RoutingWorkload: string(transport.routingWorkload), RoutingExecution: string(transport.routingExecution), RoutingExecutionNodeID: transport.routingExecutorID, diff --git a/internal/api/handlers/playback_v3_network_access_test.go b/internal/api/handlers/playback_v3_network_access_test.go new file mode 100644 index 0000000000..f62b08b229 --- /dev/null +++ b/internal/api/handlers/playback_v3_network_access_test.go @@ -0,0 +1,241 @@ +package handlers + +import ( + "context" + "net/http" + "net/http/httptest" + "net/url" + "strings" + "testing" + + "github.com/Silo-Server/silo-server/internal/config" + "github.com/Silo-Server/silo-server/internal/netaccess" + "github.com/Silo-Server/silo-server/internal/nodepool" + "github.com/Silo-Server/silo-server/internal/noderouting" + "github.com/Silo-Server/silo-server/internal/playback" + "github.com/Silo-Server/silo-server/internal/streamtoken" +) + +const ( + testLANProxyURL = "http://10.0.0.9:8083" + testTailnetOrigin = "https://proxy-1.tail1234.ts.net" + testTailnetProvider = "tailscale" +) + +func tailnetRequest(t *testing.T) *http.Request { + t.Helper() + return httptest.NewRequest(http.MethodPost, "/", nil). + WithContext(netaccess.WithPath(context.Background(), netaccess.Path{Provider: testTailnetProvider})) +} + +// lanOnlyProxyPlanner is a real planner over one healthy proxy that is only +// reachable on the LAN (no provider report), or, with connected true, one that +// also reports a connected tailnet origin. +func lanOnlyProxyPlanner(connected bool) (*nodepool.Planner, *nodepool.ProxyPool) { + node := &nodepool.Node{ID: 41, URL: testLANProxyURL, Enabled: true, Healthy: true} + if connected { + node.NetworkAccess = netaccess.NodeNetworkAccess{testTailnetProvider: {State: netaccess.StateConnected, Origin: testTailnetOrigin}} + } + proxies := nodepool.NewProxyPool() + proxies.SetNodes([]*nodepool.Node{node}) + return nodepool.NewPlanner(proxies, nodepool.NewTranscodePool()), proxies +} + +func assertAPIRelative(t *testing.T, rawURL string) { + t.Helper() + parsed, err := url.Parse(rawURL) + if err != nil || parsed.IsAbs() || parsed.Host != "" { + t.Fatalf("URL %q is not API-relative (err %v)", rawURL, err) + } + if strings.Contains(rawURL, "10.0.0.9") { + t.Fatalf("URL %q leaked the LAN origin", rawURL) + } +} + +// A client that arrived through a network access provider cannot reach a +// proxy's LAN origin. The route resolver must never hand it one: with the only +// proxy unreachable on that path the direct-play route falls back to API +// egress, the URL is API-relative, and no proxy reservation is left behind. +func TestPrepareTransportV3ProviderPathNeverReceivesALANOrigin(t *testing.T) { + for _, mode := range []struct { + name string + mode mediaAuthModeV3 + }{ + {"token mode", mediaAuthModeV3{}}, + {"authorized origins", authorizedOriginsModeV3()}, + } { + t.Run(mode.name, func(t *testing.T) { + handler := NewPlaybackHandler(playback.NewSessionManager(0, 0)) + handler.JWTSecret = "test-secret" + planner, proxies := lanOnlyProxyPlanner(false) + handler.NodePlanner = planner + grants := &recordingRecipeCardStoreV3{} + handler.ProxyGrantStore = grants + + transport, transportErr := handler.prepareTransportV3( + tailnetRequest(t), + &playback.Session{ID: "session-tailnet-direct", UserID: 7, ProfileID: "profile-1"}, + v3HandlerFixtureFile(t), + playback.PlannerResultV3{Plan: identityProxyPlanV3(playback.DeliveryOriginalHTTPV3), PlayMethod: playback.PlayDirect}, + mode.mode) + if transportErr != nil { + t.Fatalf("prepare identity transport: %v", transportErr) + } + defer transport.rollback() + + assertAPIRelative(t, transport.url) + if transport.routingEgress != noderouting.EgressAPI || transport.routingEgressID != 0 { + t.Fatalf("egress = %q on node %d, want API egress", transport.routingEgress, transport.routingEgressID) + } + if len(grants.cards) != 0 { + t.Fatalf("a proxy grant was written for a proxy the client cannot reach: %v", grants.cards) + } + // The same pool serves a LAN client with the LAN origin, so the + // exclusion is a property of the request's path, not of the node. + lanTransport, lanErr := handler.prepareTransportV3( + httptest.NewRequest(http.MethodPost, "/", nil), + &playback.Session{ID: "session-lan-direct", UserID: 7, ProfileID: "profile-1"}, + v3HandlerFixtureFile(t), + playback.PlannerResultV3{Plan: identityProxyPlanV3(playback.DeliveryOriginalHTTPV3), PlayMethod: playback.PlayDirect}, + mode.mode) + if lanErr != nil { + t.Fatalf("prepare LAN transport: %v", lanErr) + } + defer lanTransport.rollback() + if !strings.HasPrefix(lanTransport.url, testLANProxyURL+"/") { + t.Fatalf("LAN client url = %q, want the proxy's LAN origin", lanTransport.url) + } + if got := proxies.Nodes()[0].URL; got != testLANProxyURL { + t.Fatalf("pool node mutated: %q", got) + } + }) + } +} + +// Once the proxy reports a connected origin on the client's provider, that +// origin — and only that origin — is what the client is handed. +func TestPrepareTransportV3ProviderPathUsesTheProviderOrigin(t *testing.T) { + handler := NewPlaybackHandler(playback.NewSessionManager(0, 0)) + handler.JWTSecret = "test-secret" + planner, _ := lanOnlyProxyPlanner(true) + handler.NodePlanner = planner + handler.ProxyGrantStore = &recordingRecipeCardStoreV3{} + + for _, mode := range []struct { + name string + mode mediaAuthModeV3 + want string + }{ + {"token mode", mediaAuthModeV3{}, testTailnetOrigin + "/stream/direct/"}, + {"authorized origins", authorizedOriginsModeV3(), testTailnetOrigin + "/stream/v3/session-tailnet-"}, + } { + t.Run(mode.name, func(t *testing.T) { + transport, transportErr := handler.prepareTransportV3( + tailnetRequest(t), + &playback.Session{ID: "session-tailnet-" + strings.ReplaceAll(mode.name, " ", "-"), UserID: 7, ProfileID: "profile-1"}, + v3HandlerFixtureFile(t), + playback.PlannerResultV3{Plan: identityProxyPlanV3(playback.DeliveryOriginalHTTPV3), PlayMethod: playback.PlayDirect}, + mode.mode) + if transportErr != nil { + t.Fatalf("prepare identity transport: %v", transportErr) + } + defer transport.rollback() + if !strings.HasPrefix(transport.url, mode.want) || strings.Contains(transport.url, "10.0.0.9") { + t.Fatalf("url = %q, want prefix %q and no LAN address", transport.url, mode.want) + } + var provider *string + if mode.mode.headerAuth { + for _, card := range handler.ProxyGrantStore.(*recordingRecipeCardStoreV3).cards { + provider = card.RoutingNetworkProvider + } + } else { + parsed, err := url.Parse(transport.url) + if err != nil { + t.Fatal(err) + } + claims, err := streamtoken.Verify(strings.TrimPrefix(parsed.Path, "/stream/direct/"), handler.JWTSecret) + if err != nil { + t.Fatal(err) + } + provider = claims.RoutingNetworkProvider + } + if provider == nil || *provider != testTailnetProvider { + t.Fatalf("prepared recipe provider = %v", provider) + } + if transport.routingEgress != noderouting.EgressProxy || transport.routingEgressID != 41 { + t.Fatalf("egress = %q on node %d, want proxy 41", transport.routingEgress, transport.routingEgressID) + } + }) + } +} + +// proxy_only is a hard policy. A provider-path client with no reachable proxy +// gets the existing capacity-unavailable refusal rather than a LAN URL it +// cannot open or an API fallback the operator forbade. +func TestPrepareTransportV3ProviderPathProxyOnlyRefusesWithoutAReachableProxy(t *testing.T) { + handler := NewPlaybackHandler(playback.NewSessionManager(0, 0)) + handler.JWTSecret = "test-secret" + planner, _ := lanOnlyProxyPlanner(false) + handler.NodePlanner = planner + policy := config.DefaultPlaybackRoutingPolicy() + policy.DirectPlayEgress = config.PlaybackEgressProxyOnly + + _, transportErr := handler.prepareTransportWithPolicyV3( + tailnetRequest(t), + &playback.Session{ID: "session-tailnet-proxy-only", UserID: 7, ProfileID: "profile-1"}, + v3HandlerFixtureFile(t), + playback.PlannerResultV3{Plan: identityProxyPlanV3(playback.DeliveryOriginalHTTPV3), PlayMethod: playback.PlayDirect}, + mediaAuthModeV3{}, policy) + if transportErr == nil || transportErr.reason != string(noderouting.OutcomeCapacityUnavailable) { + t.Fatalf("transport error = %+v, want %s", transportErr, noderouting.OutcomeCapacityUnavailable) + } +} + +// The URL builders are the last line: even a caller that forgot the predicate +// cannot mint a LAN URL for a provider-path client. +func TestProxyURLBuildersRefuseALANOnlyProxyOnAProviderPath(t *testing.T) { + handler := NewPlaybackHandler(playback.NewSessionManager(0, 0)) + handler.JWTSecret = "test-secret" + file := v3HandlerFixtureFile(t) + lanOnly := &nodepool.Node{ID: 41, URL: testLANProxyURL} + tailnet := netaccess.Path{Provider: testTailnetProvider} + session := &playback.Session{ID: "session-builders", UserID: 7, ProfileID: "profile-1", MediaFileID: file.ID, PlayMethod: playback.PlayDirect} + card := playback.NewRecipeCard(session.UserID, session.ProfileID, file.ID, "", playback.TranscodeOpts{SessionID: session.ID, InputPath: file.FilePath}) + handler.ProxyGrantStore = &recordingRecipeCardStoreV3{} + + if got, byProxy := handler.identityStreamURLV3(session, file, lanOnly, tailnet); byProxy || got != handler.playbackStreamURL(session) { + t.Fatalf("identityStreamURLV3 = %q (proxy %v), want the API-local route", got, byProxy) + } + if got, byProxy, _ := handler.identityGrantStreamURLV3(context.Background(), session, file, lanOnly, tailnet); byProxy || got != handler.playbackStreamURL(session) { + t.Fatalf("identityGrantStreamURLV3 = %q (proxy %v), want the API-local route", got, byProxy) + } + if got := handler.buildProxyManifestURL(card, lanOnly, false, tailnet); !strings.HasPrefix(got, "/playback/transcode/session-builders/master.m3u8") { + t.Fatalf("buildProxyManifestURL = %q, want the API-local manifest", got) + } + if got, byProxy, _ := handler.grantManifestURLV3(context.Background(), card, lanOnly, tailnet); byProxy || got != "/playback/transcode/session-builders/master.m3u8" { + t.Fatalf("grantManifestURLV3 = %q (proxy %v), want the API-local manifest", got, byProxy) + } + + connected := &nodepool.Node{ID: 41, URL: testLANProxyURL, NetworkAccess: netaccess.NodeNetworkAccess{testTailnetProvider: {State: netaccess.StateConnected, Origin: testTailnetOrigin}}} + if got, byProxy := handler.identityStreamURLV3(session, file, connected, tailnet); !byProxy || !strings.HasPrefix(got, testTailnetOrigin+"/stream/direct/") { + t.Fatalf("identityStreamURLV3 on a connected proxy = %q (proxy %v)", got, byProxy) + } + if got := handler.buildProxyManifestURL(card, connected, false, tailnet); !strings.HasPrefix(got, testTailnetOrigin+"/stream/transcode/") { + t.Fatalf("buildProxyManifestURL on a connected proxy = %q", got) + } + // The default path is unchanged by the report: LAN clients keep the LAN URL. + if got, byProxy := handler.identityStreamURLV3(session, file, connected, netaccess.Path{}); !byProxy || !strings.HasPrefix(got, testLANProxyURL+"/stream/direct/") { + t.Fatalf("identityStreamURLV3 on the default path = %q (proxy %v)", got, byProxy) + } +} + +func TestV3SessionStateRecordsValidatedNetworkProvider(t *testing.T) { + handler := NewPlaybackHandler(playback.NewSessionManager(0, 0)) + for _, provider := range []string{"", "tailscale"} { + ctx := netaccess.WithPath(t.Context(), netaccess.Path{Provider: provider}) + state := handler.v3SessionStreamState(ctx, &playback.Session{}, nil, playback.PlannerResultV3{}, preparedTransportV3{}, mediaAuthModeV3{}) + if state.RoutingNetworkProvider == nil || *state.RoutingNetworkProvider != provider { + t.Fatalf("provider = %v, want %q", state.RoutingNetworkProvider, provider) + } + } +} diff --git a/internal/api/handlers/playback_v3_test.go b/internal/api/handlers/playback_v3_test.go index 43bc0dd501..6379c68026 100644 --- a/internal/api/handlers/playback_v3_test.go +++ b/internal/api/handlers/playback_v3_test.go @@ -25,6 +25,7 @@ import ( "github.com/Silo-Server/silo-server/internal/config" "github.com/Silo-Server/silo-server/internal/markers" "github.com/Silo-Server/silo-server/internal/models" + "github.com/Silo-Server/silo-server/internal/netaccess" "github.com/Silo-Server/silo-server/internal/nodepool" "github.com/Silo-Server/silo-server/internal/noderouting" "github.com/Silo-Server/silo-server/internal/playback" @@ -5565,7 +5566,7 @@ func TestIdentityStreamURLV3VersionsOnlyBoostedRemuxRoutes(t *testing.T) { PlayMethod: playback.PlayRemux, TranscodeAudio: true, TargetAudioCodec: "aac", SourceAudioChannels: 6, TargetAudioChannels: 2, } - boostedURL, servedByProxy := handler.identityStreamURLV3(boosted, file, proxy) + boostedURL, servedByProxy := handler.identityStreamURLV3(boosted, file, proxy, netaccess.Path{}) boostedPrefix := "http://proxy-1/stream/remux/audio-v2/" if !servedByProxy || !strings.HasPrefix(boostedURL, boostedPrefix) { t.Fatalf("boosted remux URL = %q (proxy %v), want the audio-v2 route", boostedURL, servedByProxy) @@ -5593,7 +5594,7 @@ func TestIdentityStreamURLV3VersionsOnlyBoostedRemuxRoutes(t *testing.T) { ordinary := *boosted ordinary.ID = "ordinary" test.mutate(&ordinary) - ordinaryURL, ordinaryByProxy := handler.identityStreamURLV3(&ordinary, file, proxy) + ordinaryURL, ordinaryByProxy := handler.identityStreamURLV3(&ordinary, file, proxy, netaccess.Path{}) legacyPrefix := "http://proxy-1/stream/remux/" if !ordinaryByProxy || !strings.HasPrefix(ordinaryURL, legacyPrefix) || strings.HasPrefix(ordinaryURL, boostedPrefix) { t.Fatalf("ordinary remux URL = %q (proxy %v), want the legacy route", ordinaryURL, ordinaryByProxy) diff --git a/internal/api/handlers/playback_v3_tokenless_test.go b/internal/api/handlers/playback_v3_tokenless_test.go index 69538f8a70..6dd4c3dc00 100644 --- a/internal/api/handlers/playback_v3_tokenless_test.go +++ b/internal/api/handlers/playback_v3_tokenless_test.go @@ -10,6 +10,7 @@ import ( "testing" "github.com/Silo-Server/silo-server/internal/models" + "github.com/Silo-Server/silo-server/internal/netaccess" "github.com/Silo-Server/silo-server/internal/nodepool" "github.com/Silo-Server/silo-server/internal/noderouting" "github.com/Silo-Server/silo-server/internal/playback" @@ -98,10 +99,10 @@ func TestPlaybackURLBuildersRefuseTokensForMediaAuthorizedSessions(t *testing.T) t.Fatalf("legacy stream URL = %q, want a restart token", got) } - if got, servedByProxy := handler.identityStreamURLV3(secure, file, proxy); servedByProxy || got != "/stream/session-secure" { + if got, servedByProxy := handler.identityStreamURLV3(secure, file, proxy, netaccess.Path{}); servedByProxy || got != "/stream/session-secure" { t.Fatalf("secure identity URL = %q (proxy %v), want the API-local route", got, servedByProxy) } - if got, servedByProxy := handler.identityStreamURLV3(&legacy, file, proxy); !servedByProxy || !strings.HasPrefix(got, proxy.URL) { + if got, servedByProxy := handler.identityStreamURLV3(&legacy, file, proxy, netaccess.Path{}); !servedByProxy || !strings.HasPrefix(got, proxy.URL) { t.Fatalf("legacy identity URL = %q (proxy %v), want the signed proxy route", got, servedByProxy) } @@ -109,10 +110,10 @@ func TestPlaybackURLBuildersRefuseTokensForMediaAuthorizedSessions(t *testing.T) card.RoutingWorkload = string(noderouting.WorkloadVideoTranscode) card.RoutingExecution = string(noderouting.ExecutionTranscode) card.RoutingEgress = string(noderouting.EgressProxy) - if got := handler.buildProxyManifestURL(card, proxy, true); got != "/playback/transcode/session-secure/master.m3u8" { + if got := handler.buildProxyManifestURL(card, proxy, true, netaccess.Path{}); got != "/playback/transcode/session-secure/master.m3u8" { t.Fatalf("secure manifest URL = %q, want the tokenless API-local manifest", got) } - legacyManifestURL := handler.buildProxyManifestURL(card, proxy, false) + legacyManifestURL := handler.buildProxyManifestURL(card, proxy, false, netaccess.Path{}) if !strings.HasPrefix(legacyManifestURL, proxy.URL+"/stream/transcode/") { t.Fatalf("legacy manifest URL = %q, want the signed proxy manifest", legacyManifestURL) } @@ -136,7 +137,7 @@ func TestPlaybackURLBuildersRefuseTokensForMediaAuthorizedSessions(t *testing.T) // must never fall back to minting the credential the mode removed. grants := &recordingRecipeCardStoreV3{} handler.ProxyGrantStore = grants - got, servedByProxy, _ := handler.identityGrantStreamURLV3(context.Background(), secure, file, proxy) + got, servedByProxy, _ := handler.identityGrantStreamURLV3(context.Background(), secure, file, proxy, netaccess.Path{}) if !servedByProxy || got != proxy.URL+"/stream/v3/session-secure" { t.Fatalf("origins identity URL = %q (proxy %v), want the credential-free proxy route", got, servedByProxy) } @@ -147,7 +148,7 @@ func TestPlaybackURLBuildersRefuseTokensForMediaAuthorizedSessions(t *testing.T) t.Fatalf("identity grant egress node ID = %d, want 71", grant.RoutingEgressNodeID) } - got, servedByProxy, _ = handler.grantManifestURLV3(context.Background(), card, proxy) + got, servedByProxy, _ = handler.grantManifestURLV3(context.Background(), card, proxy, netaccess.Path{}) if !servedByProxy || got != proxy.URL+"/stream/v3/session-secure/master.m3u8" { t.Fatalf("origins manifest URL = %q (proxy %v), want the credential-free proxy manifest", got, servedByProxy) } @@ -426,15 +427,15 @@ func TestProxyURLBuildersUseThePublicURLWhenSet(t *testing.T) { proxy := &nodepool.Node{URL: "http://10.0.0.9:8083", PublicURL: &public} session := &playback.Session{ID: "session-public", UserID: 7, ProfileID: "profile-1", MediaFileID: file.ID, PlayMethod: playback.PlayDirect} - if got, servedByProxy := handler.identityStreamURLV3(session, file, proxy); !servedByProxy || !strings.HasPrefix(got, public+"/stream/direct/") { + if got, servedByProxy := handler.identityStreamURLV3(session, file, proxy, netaccess.Path{}); !servedByProxy || !strings.HasPrefix(got, public+"/stream/direct/") { t.Fatalf("identity URL = %q (proxy %v), want the public origin", got, servedByProxy) } card := playback.NewRecipeCard(session.UserID, session.ProfileID, file.ID, "", playback.TranscodeOpts{SessionID: session.ID, InputPath: file.FilePath}) - if got := handler.buildProxyManifestURL(card, proxy, false); !strings.HasPrefix(got, public+"/stream/transcode/") { + if got := handler.buildProxyManifestURL(card, proxy, false, netaccess.Path{}); !strings.HasPrefix(got, public+"/stream/transcode/") { t.Fatalf("manifest URL = %q, want the public origin", got) } - if strings.Contains(handler.buildProxyManifestURL(card, proxy, false), "10.0.0.9") { + if strings.Contains(handler.buildProxyManifestURL(card, proxy, false, netaccess.Path{}), "10.0.0.9") { t.Fatalf("manifest URL leaked the backend address") } } diff --git a/internal/api/handlers/plugins.go b/internal/api/handlers/plugins.go index 25814a494a..a7debea67c 100644 --- a/internal/api/handlers/plugins.go +++ b/internal/api/handlers/plugins.go @@ -176,6 +176,22 @@ type PluginInstallationView struct { TaskBindings []PluginTaskBindingView `json:"task_bindings"` CreatedAt time.Time `json:"created_at"` UpdatedAt time.Time `json:"updated_at"` + // Runtime is the process state the v2 contract exposes as `runtime`. The + // frozen v1 bridge does not carry it. + Runtime *PluginRuntimeView `json:"-"` +} + +// PluginRuntimeView is one installation's process state. Resident marks a +// plugin the supervisor keeps running (network access providers); State is +// the supervisor's machine for those and running/stopped for lazily started +// plugins. +type PluginRuntimeView struct { + Resident bool + State string + RestartCount int + LastError string + LastStartedAt *time.Time + NextRestartAt *time.Time } type pluginCatalogSettingsResponse struct { @@ -1370,6 +1386,23 @@ func (h *PluginHandler) buildInstallationResponseWithBindings( if presentation != nil { repoURL = presentation.SourceURL } + var runtime *PluginRuntimeView + if h.service != nil { + state := h.service.RuntimeState(installation.ID) + runtime = &PluginRuntimeView{Resident: state.Resident, State: string(state.State), RestartCount: state.RestartCount, LastError: state.LastError, LastStartedAt: state.LastStartedAt, NextRestartAt: state.NextRestartAt} + // Before the supervisor arms (boot) the capability says what will be + // resident. Once armed, its entries are the truth: an enabled + // installation it deliberately does not own (a duplicate provider + // slug) is not resident, so the page must not offer a restart that + // would be a no-op. + if !state.Resident && !h.service.ResidentsArmed() { + for _, capability := range capabilities { + if plugins.IsResidentCapabilityType(capability.Type) { + runtime.Resident = true + } + } + } + } return PluginInstallationView{ ID: installation.ID, @@ -1397,6 +1430,7 @@ func (h *PluginHandler) buildInstallationResponseWithBindings( TaskBindings: taskBindingsForInstallation(installation.ID, taskBindings), CreatedAt: installation.CreatedAt, UpdatedAt: installation.UpdatedAt, + Runtime: runtime, }, nil } diff --git a/internal/api/handlers/socket_origin_overlay_test.go b/internal/api/handlers/socket_origin_overlay_test.go new file mode 100644 index 0000000000..fa4c75a7f8 --- /dev/null +++ b/internal/api/handlers/socket_origin_overlay_test.go @@ -0,0 +1,78 @@ +package handlers + +import ( + "net/http" + "net/http/httptest" + "testing" + + "github.com/Silo-Server/silo-server/internal/netaccess" +) + +// A browser reaching Silo through a connected network access provider sends +// the overlay origin. Both origin checks accept it while the provider is +// connected and refuse it again once the provider drops, without touching the +// public-origin rule. +func TestSocketOriginAcceptsConnectedOverlayOrigin(t *testing.T) { + cache := netaccess.NewStatusCache() + cache.Report(netaccess.Status{ + InstallationID: 3, Provider: "tailscale", State: netaccess.StateConnected, + Origin: "https://silo.tail1234.ts.net", + Listeners: []netaccess.Listener{{Name: "api", Origin: "https://silo.tail1234.ts.net"}}, + }) + overlay := cache.ConnectedOrigins() + + newReq := func(origin string) *http.Request { + r := httptest.NewRequest(http.MethodGet, "http://internal.test/api/v2/events/ws", nil) + r.Host = "internal.test" + r.Header.Set("Origin", origin) + return r + } + if !socketOriginAllowed(newReq("https://silo.tail1234.ts.net"), "https://public.test", overlay) { + t.Fatal("overlay origin refused while the provider is connected") + } + if !socketOriginAllowed(newReq("https://public.test"), "https://public.test", overlay) { + t.Fatal("public origin refused with an overlay configured") + } + if socketOriginAllowed(newReq("http://silo.tail1234.ts.net"), "https://public.test", overlay) { + t.Fatal("overlay host accepted with the wrong scheme") + } + if socketOriginAllowed(newReq("https://evil.test"), "https://public.test", overlay) { + t.Fatal("foreign origin accepted") + } + if socketOriginAllowed(newReq("https://silo.tail1234.ts.net"), "https://public.test", nil) { + t.Fatal("overlay origin accepted without a connected provider") + } + + // The handlers read the source per handshake: a dropped provider is + // refused on the next check without a config reload. + h := &EventsSocketV2{PublicOrigin: "https://public.test"} + h.SetOverlayOrigins(cache.ConnectedOrigins) + if !h.validOrigin(newReq("https://silo.tail1234.ts.net")) { + t.Fatal("events socket refused the overlay origin") + } + cache.Forget(3) + if h.validOrigin(newReq("https://silo.tail1234.ts.net")) { + t.Fatal("events socket accepted the overlay origin after the provider dropped") + } +} + +func TestCheckWebSocketOriginAcceptsConnectedOverlayOrigin(t *testing.T) { + cache := netaccess.NewStatusCache() + SetWebSocketOverlayOrigins(cache.ConnectedOrigins) + t.Cleanup(func() { SetWebSocketOverlayOrigins(nil) }) + + req := httptest.NewRequest(http.MethodGet, "/playback/sessions/abc/control/ws", nil) + req.Host = "origin.internal:8097" + req.Header.Set("Origin", "https://silo.tail1234.ts.net") + if checkWebSocketOrigin(req) { + t.Fatal("overlay origin accepted before the provider connected") + } + cache.Report(netaccess.Status{InstallationID: 3, Provider: "tailscale", State: netaccess.StateConnected, Origin: "https://silo.tail1234.ts.net"}) + if !checkWebSocketOrigin(req) { + t.Fatal("overlay origin refused while the provider is connected") + } + req.Header.Set("Origin", "https://evil.test") + if checkWebSocketOrigin(req) { + t.Fatal("foreign origin accepted") + } +} diff --git a/internal/api/handlers/socket_origin_proxy_test.go b/internal/api/handlers/socket_origin_proxy_test.go index c2a2567b1d..6e46fe7c8d 100644 --- a/internal/api/handlers/socket_origin_proxy_test.go +++ b/internal/api/handlers/socket_origin_proxy_test.go @@ -38,18 +38,18 @@ func TestSocketOriginTrustedProxy(t *testing.T) { r.Header["X-Forwarded-Proto"] = tc.proto r.Header.Set("X-Forwarded-Host", "foreign.test") clientip.Middleware(resolver)(http.HandlerFunc(func(_ http.ResponseWriter, r *http.Request) { - if got := socketOriginAllowed(r, tc.override); got != tc.want { + if got := socketOriginAllowed(r, tc.override, nil); got != tc.want { t.Fatalf("allowed=%v want %v", got, tc.want) } })).ServeHTTP(httptest.NewRecorder(), r) }) } r := httptest.NewRequest("GET", "http://example.test/", nil) - if !socketOriginAllowed(r, "") { + if !socketOriginAllowed(r, "", nil) { t.Fatal("native originless request refused") } r.Header["Origin"] = []string{"http://example.test", "http://example.test"} - if socketOriginAllowed(r, "") { + if socketOriginAllowed(r, "", nil) { t.Fatal("multiple origins allowed") } } diff --git a/internal/api/handlers/watch_together_socket_v2.go b/internal/api/handlers/watch_together_socket_v2.go index aa4a23661e..3040720891 100644 --- a/internal/api/handlers/watch_together_socket_v2.go +++ b/internal/api/handlers/watch_together_socket_v2.go @@ -21,12 +21,13 @@ type roomSocketTickets interface { Consume(context.Context, string, string) (watchtogether.RoomSocketCredential, error) } type WatchTogetherSocketV2 struct { - Room *WatchTogetherHandler - Tickets roomSocketTickets - Validate EventsSocketValidator - PublicOrigin string - publicOrigin atomic.Pointer[string] - checkInterval time.Duration + Room *WatchTogetherHandler + Tickets roomSocketTickets + Validate EventsSocketValidator + PublicOrigin string + publicOrigin atomic.Pointer[string] + overlayOrigins atomic.Pointer[OverlayOriginSource] + checkInterval time.Duration } func NewWatchTogetherSocketV2(room *WatchTogetherHandler, tickets *watchtogether.RoomSocketCredentialStore, sessions eventsSessionValidator, users access.UserRepository, resolver apimw.ViewerResolver, primary apimw.PrimaryProfileChecker, publicURL string) *WatchTogetherSocketV2 { @@ -82,7 +83,7 @@ func (h *WatchTogetherSocketV2) ServeHTTP(w http.ResponseWriter, r *http.Request http.Error(w, "invalid handshake", http.StatusBadRequest) return } - if !socketOriginAllowed(r, h.currentPublicOrigin()) { + if !socketOriginAllowed(r, h.currentPublicOrigin(), overlayOriginsFrom(h.overlayOrigins.Load())) { http.Error(w, "origin refused", http.StatusForbidden) return } @@ -126,7 +127,9 @@ func (h *WatchTogetherSocketV2) ServeHTTP(w http.ResponseWriter, r *http.Request } ctx, cancel := context.WithDeadline(validated, deadline) defer cancel() - upgrader := websocket.Upgrader{Subprotocols: []string{watchtogether.RoomSocketProtocol}, CheckOrigin: func(r *http.Request) bool { return socketOriginAllowed(r, h.currentPublicOrigin()) }} + upgrader := websocket.Upgrader{Subprotocols: []string{watchtogether.RoomSocketProtocol}, CheckOrigin: func(r *http.Request) bool { + return socketOriginAllowed(r, h.currentPublicOrigin(), overlayOriginsFrom(h.overlayOrigins.Load())) + }} conn, err := upgrader.Upgrade(w, r, nil) if err != nil { return @@ -173,3 +176,9 @@ func (h *WatchTogetherSocketV2) SetPublicOrigin(origin string) { normalized := strings.TrimRight(origin, "/") h.publicOrigin.Store(&normalized) } + +// SetOverlayOrigins installs the source of overlay origins accepted next to +// the public origin. +func (h *WatchTogetherSocketV2) SetOverlayOrigins(source OverlayOriginSource) { + h.overlayOrigins.Store(&source) +} diff --git a/internal/api/handlers/ws_shared.go b/internal/api/handlers/ws_shared.go index 59f424ff32..2ef5bff54b 100644 --- a/internal/api/handlers/ws_shared.go +++ b/internal/api/handlers/ws_shared.go @@ -6,6 +6,7 @@ import ( "net/http" "net/url" "strings" + "sync/atomic" "time" "github.com/gorilla/websocket" @@ -21,6 +22,17 @@ var wsUpgrader = websocket.Upgrader{ CheckOrigin: checkWebSocketOrigin, } +// wsOverlayOrigins feeds the shared v1 upgrader the overlay origins connected +// network access providers report. The upgrader is package state, so its +// source is too; SetWebSocketOverlayOrigins is called once at router build. +var wsOverlayOrigins atomic.Pointer[OverlayOriginSource] + +// SetWebSocketOverlayOrigins installs the overlay origin source consulted by +// the shared v1 WebSocket upgrader. +func SetWebSocketOverlayOrigins(source OverlayOriginSource) { + wsOverlayOrigins.Store(&source) +} + func checkWebSocketOrigin(r *http.Request) bool { origin := strings.TrimSpace(r.Header.Get("Origin")) if origin == "" { @@ -43,6 +55,14 @@ func checkWebSocketOrigin(r *http.Request) bool { return true } + // A browser reaching Silo through a network access provider's overlay + // listener sends the overlay origin; accept the ones connected right now. + for _, overlay := range overlayOriginsFrom(wsOverlayOrigins.Load()) { + if originMatches(originURL, overlay) { + return true + } + } + return false } diff --git a/internal/api/router.go b/internal/api/router.go index 76f8efdae5..c2920769d0 100644 --- a/internal/api/router.go +++ b/internal/api/router.go @@ -52,6 +52,7 @@ import ( "github.com/Silo-Server/silo-server/internal/metadata/tmdb" metatrakt "github.com/Silo-Server/silo-server/internal/metadata/trakt" metadatatranslation "github.com/Silo-Server/silo-server/internal/metadata/translation" + "github.com/Silo-Server/silo-server/internal/netaccess" "github.com/Silo-Server/silo-server/internal/nodemetrics" "github.com/Silo-Server/silo-server/internal/nodepool" "github.com/Silo-Server/silo-server/internal/noderecipe" @@ -164,13 +165,18 @@ type Dependencies struct { CatalogSearchVectorizer catalog.CatalogSearchQueryVectorizer // CatalogSearchSettings is the process-lifetime startup snapshot shared by // every native/jellycompat provider and the index maintenance worker. - CatalogSearchSettings *catalog.CatalogSearchSettings - RatingsRepo *catalog.RatingsRepo - PersonRepo *catalog.PersonRepository - PersonRefreshQueue handlers.PersonRefreshQueue - PersonRefresher handlers.PersonRefresher - RateLimitMW *ratelimit.Middleware - ClientIPResolver *clientip.Resolver + CatalogSearchSettings *catalog.CatalogSearchSettings + RatingsRepo *catalog.RatingsRepo + PersonRepo *catalog.PersonRepository + PersonRefreshQueue handlers.PersonRefreshQueue + PersonRefresher handlers.PersonRefresher + RateLimitMW *ratelimit.Middleware + ClientIPResolver *clientip.Resolver + // NetworkAccess is the ingress-token registry and provider status cache + // for network access provider plugins on this host. The token middleware + // runs on every native request and connected overlay origins are accepted + // by WebSocket handshakes. Nil disables both (tests, worker modes). + NetworkAccess *netaccess.Broker NodeID string LogStreamHub *logstream.Hub RealtimeHub *notifications.Hub @@ -328,6 +334,9 @@ func newChiRouter(deps Dependencies) chi.Router { r := chi.NewRouter() useBaseMiddleware(r, deps) + if overlay := deps.overlayOrigins(); overlay != nil { + handlers.SetWebSocketOverlayOrigins(overlay) + } // Build the readiness handler with optional S3 check. var s3Checker handlers.S3HealthChecker @@ -2034,6 +2043,9 @@ func newChiRouter(deps Dependencies) chi.Router { if deps.OnConfigChange != nil { deps.OnConfigChange(func(_, updated *config.Config) { socket.SetPublicOrigin(updated.Server.PublicURL) }) } + if overlay := deps.overlayOrigins(); overlay != nil { + socket.SetOverlayOrigins(overlay) + } } if autoscanHandler != nil { v2deps.AutoscanDelivery = autoscanHandler @@ -2136,6 +2148,9 @@ func newChiRouter(deps Dependencies) chi.Router { if deps.OnConfigChange != nil { deps.OnConfigChange(func(_, updated *config.Config) { socket.SetPublicOrigin(updated.Server.PublicURL) }) } + if overlay := deps.overlayOrigins(); overlay != nil { + socket.SetOverlayOrigins(overlay) + } } // Raw v2 delivery shares the byte-protocol handlers; fonts use the typed // service. Both retain token-carried reconstruction and deny markers. @@ -2218,6 +2233,9 @@ func newChiRouter(deps Dependencies) chi.Router { if deps.OnConfigChange != nil { deps.OnConfigChange(func(_, updated *config.Config) { socket.SetPublicOrigin(updated.Server.PublicURL) }) } + if overlay := deps.overlayOrigins(); overlay != nil { + socket.SetOverlayOrigins(overlay) + } } } if deps.EventsHub != nil { @@ -2230,6 +2248,9 @@ func newChiRouter(deps Dependencies) chi.Router { if deps.OnConfigChange != nil { deps.OnConfigChange(func(_, updated *config.Config) { socket.SetPublicOrigin(updated.Server.PublicURL) }) } + if overlay := deps.overlayOrigins(); overlay != nil { + socket.SetOverlayOrigins(overlay) + } } } if deps.Notifications != nil { @@ -2326,6 +2347,11 @@ func newChiRouter(deps Dependencies) chi.Router { v2deps.AdminNodesRead = nodeHandler v2deps.AdminNodeCommands = nodeHandler v2deps.AdminNodeReload = nodeHandler + // Network access admin operations fan out to every enabled proxy node + // over its bearer routes, the same way force-reload does. + if deps.PluginService != nil { + deps.PluginService.SetNetworkAccessNodes(nodeHandler) + } if deps.DB != nil { nodeHandler.SetConfigurationStore(nodepool.NewAdminConfigurationStore(deps.DB)) v2deps.AdminNodeConfiguration = nodeHandler @@ -2390,6 +2416,9 @@ func newChiRouter(deps Dependencies) chi.Router { v2deps.AdminPluginLifecycle = v2PluginHandler v2deps.AdminPluginUploads = v2PluginHandler } + if deps.PluginService != nil { + v2deps.NetworkAccess = deps.PluginService + } if deps.TaskManager != nil && deps.DB != nil { v2deps.AdminTasks = deps.TaskManager v2deps.AdminTaskMetrics = metadata.NewRefreshDebtRepository(deps.DB) @@ -4128,6 +4157,15 @@ func newChiRouter(deps Dependencies) chi.Router { return r } +// overlayOrigins returns the source of connected overlay origins the +// WebSocket handshakes accept, or nil when network access is not wired. +func (d Dependencies) overlayOrigins() handlers.OverlayOriginSource { + if d.NetworkAccess == nil || d.NetworkAccess.Status == nil { + return nil + } + return d.NetworkAccess.Status.ConnectedOrigins +} + // useBaseMiddleware mounts the middleware chain every native request passes // through, in order. It is factored out of NewRouter so a test can drive the // real chain over a real socket: re-declaring the stack in a test would let the @@ -4142,6 +4180,13 @@ func useBaseMiddleware(r chi.Router, deps Dependencies) { r.Use(clientip.Middleware(deps.ClientIPResolver)) } + // Ingress token from network access provider plugins: validated and + // stripped before anything can log or forward it; an unknown token is + // refused outright. Requests without it stay on the default access path. + if deps.NetworkAccess != nil { + r.Use(netaccess.Middleware(deps.NetworkAccess.Registry)) + } + r.Use(apimw.RequestLogger(deps.NodeID)) r.Use(middleware.Recoverer) r.Use(apimw.Metrics) diff --git a/internal/api/testdata/media_routes.txt b/internal/api/testdata/media_routes.txt index e133bba6c5..ac01b08687 100644 --- a/internal/api/testdata/media_routes.txt +++ b/internal/api/testdata/media_routes.txt @@ -173,6 +173,9 @@ GET /api/v2/admin/markers/items/{id}/history non-media GET /api/v2/admin/markers/providers non-media PUT /api/v2/admin/markers/providers/{provider} non-media POST /api/v2/admin/markers/providers/{provider}/validate non-media +POST /api/v2/admin/network-access/{provider}/connect non-media +POST /api/v2/admin/network-access/{provider}/disconnect non-media +GET /api/v2/admin/network-access/{provider}/status non-media GET /api/v2/admin/node-sessions non-media GET /api/v2/admin/nodes non-media POST /api/v2/admin/nodes non-media @@ -208,6 +211,7 @@ PUT /api/v2/admin/plugins/installations/{id} non-media PUT /api/v2/admin/plugins/installations/{id}/auth-binding non-media PUT /api/v2/admin/plugins/installations/{id}/config non-media POST /api/v2/admin/plugins/installations/{id}/config/test non-media +POST /api/v2/admin/plugins/installations/{id}/restart non-media PUT /api/v2/admin/plugins/installations/{id}/task-bindings/{capability_id} non-media POST /api/v2/admin/plugins/installations/{id}/update non-media GET /api/v2/admin/plugins/repositories non-media @@ -540,6 +544,7 @@ PUT /api/v2/markers/files/{file_id} non-media DELETE /api/v2/markers/files/{file_id}/{segment} non-media GET /api/v2/markers/items/{item_id} non-media PUT /api/v2/markers/items/{item_id} non-media +GET /api/v2/network-access/capabilities non-media GET /api/v2/notifications non-media GET /api/v2/notifications/capabilities non-media DELETE /api/v2/notifications/discord-link non-media @@ -1306,6 +1311,9 @@ GET /api/v2/admin/markers/items/{id}/history non-media GET /api/v2/admin/markers/providers non-media PUT /api/v2/admin/markers/providers/{provider} non-media POST /api/v2/admin/markers/providers/{provider}/validate non-media +POST /api/v2/admin/network-access/{provider}/connect non-media +POST /api/v2/admin/network-access/{provider}/disconnect non-media +GET /api/v2/admin/network-access/{provider}/status non-media GET /api/v2/admin/node-sessions non-media GET /api/v2/admin/nodes non-media POST /api/v2/admin/nodes non-media @@ -1341,6 +1349,7 @@ PUT /api/v2/admin/plugins/installations/{id} non-media PUT /api/v2/admin/plugins/installations/{id}/auth-binding non-media PUT /api/v2/admin/plugins/installations/{id}/config non-media POST /api/v2/admin/plugins/installations/{id}/config/test non-media +POST /api/v2/admin/plugins/installations/{id}/restart non-media PUT /api/v2/admin/plugins/installations/{id}/task-bindings/{capability_id} non-media POST /api/v2/admin/plugins/installations/{id}/update non-media GET /api/v2/admin/plugins/repositories non-media @@ -1673,6 +1682,7 @@ PUT /api/v2/markers/files/{file_id} non-media DELETE /api/v2/markers/files/{file_id}/{segment} non-media GET /api/v2/markers/items/{item_id} non-media PUT /api/v2/markers/items/{item_id} non-media +GET /api/v2/network-access/capabilities non-media GET /api/v2/notifications non-media GET /api/v2/notifications/capabilities non-media DELETE /api/v2/notifications/discord-link non-media diff --git a/internal/apiv2/admin_nodes_read.go b/internal/apiv2/admin_nodes_read.go index 1b4dee51fc..ebf5fb56d3 100644 --- a/internal/apiv2/admin_nodes_read.go +++ b/internal/apiv2/admin_nodes_read.go @@ -41,6 +41,19 @@ type AdminNode struct { CapabilityDriftBaseline json.RawMessage `json:"capability_drift_baseline,omitempty"` AdvertisedCapabilitiesHash *string `json:"advertised_capabilities_hash,omitempty" doc:"Absent when not checked in this process; empty when checked but no hash was advertised."` PhysicalGPUKeys []string `json:"physical_gpu_keys,omitempty"` + // NetworkAccess is keyed by provider slug ("tailscale"). Only proxy nodes + // report it; a transcode node never does because clients never reach one. + NetworkAccess map[string]AdminNodeNetworkAccess `json:"network_access,omitempty" doc:"Last network access provider status the node reported on its health check, keyed by provider slug. Omitted when the node reports no providers."` +} + +// AdminNodeNetworkAccess is one provider's status as the node last reported +// it. Stream URLs for clients arriving through that provider are built on +// origin while state is connected. +type AdminNodeNetworkAccess struct { + State string `json:"state" doc:"disconnected | awaiting_authorization | connecting | connected | error"` + Origin string `json:"origin,omitempty" doc:"scheme://host[:port] clients on the provider's overlay use to reach this node. Only used while state is connected."` + Hostname string `json:"hostname,omitempty" doc:"Overlay DNS name of the node."` + UpdatedAt *Instant `json:"updated_at,omitempty" doc:"When the node last heard from the provider, on the node's clock."` } type AdminNodesListInput struct { LimitParam @@ -67,6 +80,16 @@ func adminNodeOf(n *nodepool.Node) AdminNode { if n.CapabilitiesRefreshedAt != nil { out.CapabilitiesRefreshedAt = new(NewInstant(*n.CapabilitiesRefreshedAt)) } + if len(n.NetworkAccess) > 0 { + out.NetworkAccess = make(map[string]AdminNodeNetworkAccess, len(n.NetworkAccess)) + for provider, status := range n.NetworkAccess { + entry := AdminNodeNetworkAccess{State: status.State, Origin: status.Origin, Hostname: status.Hostname} + if !status.UpdatedAt.IsZero() { + entry.UpdatedAt = new(NewInstant(status.UpdatedAt)) + } + out.NetworkAccess[provider] = entry + } + } return out } func registerAdminNodesRead(reg *Registry) { diff --git a/internal/apiv2/admin_plugin_installations.go b/internal/apiv2/admin_plugin_installations.go index 2cecc3ddf5..29e7b52d77 100644 --- a/internal/apiv2/admin_plugin_installations.go +++ b/internal/apiv2/admin_plugin_installations.go @@ -169,6 +169,21 @@ type AdminPluginCatalogEntry struct { Metadata PluginJSONValue `json:"metadata"` } +// AdminPluginRuntime is one installation's process state. A resident plugin +// (one declaring a capability the server keeps running, such as a network +// access provider) is supervised: started at boot, restarted after a crash +// with exponential backoff, and parked as failed after repeated failures +// until an administrator restarts or reconfigures it. Other plugins start on +// first use and report only running or stopped. +type AdminPluginRuntime struct { + Resident bool `json:"resident" doc:"True when the server supervises this plugin's process"` + State string `json:"state" enum:"stopped,starting,running,backoff,failed" doc:"Process state; backoff and failed occur only for resident plugins"` + RestartCount int `json:"restart_count" doc:"Automatic restarts since the plugin last ran stably or was restarted by an administrator"` + LastError string `json:"last_error,omitempty" doc:"Why the process last stopped or failed to start"` + LastStartedAt *Instant `json:"last_started_at,omitempty"` + NextRestartAt *Instant `json:"next_restart_at,omitempty" doc:"Scheduled automatic restart while in backoff"` +} + // AdminPluginInstallation is one manageable installation with its redacted // configuration and bindings. type AdminPluginInstallation struct { @@ -195,6 +210,7 @@ type AdminPluginInstallation struct { GlobalConfigs []AdminPluginConfigValue `json:"global_configs"` AuthBindings []AdminPluginAuthBinding `json:"auth_bindings"` TaskBindings []AdminPluginTaskBinding `json:"task_bindings"` + Runtime AdminPluginRuntime `json:"runtime"` CreatedAt Instant `json:"created_at"` UpdatedAt Instant `json:"updated_at"` } @@ -307,6 +323,23 @@ func adminPluginCatalogEntryOf(v handlers.PluginCatalogEntryView) (AdminPluginCa return AdminPluginCatalogEntry{RepositoryID: IDFromInt(int64(v.RepositoryID)), PluginID: v.PluginID, Version: v.Version, ArchiveURL: v.ArchiveURL, SourceKind: v.SourceKind, RepositoryName: v.RepositoryName, RepoURL: v.RepoURL, Presentation: adminPluginPresentationOf(v.Presentation), Capabilities: caps, GlobalConfigSchema: global, UserConfigSchema: user, Routes: pluginRoutesOf(v.Routes), Assets: pluginAssetsOf(v.Assets), Metadata: NonNilMap(PluginJSONValue(v.Metadata))}, nil } +func adminPluginRuntimeOf(v *handlers.PluginRuntimeView) AdminPluginRuntime { + if v == nil { + return AdminPluginRuntime{State: string(plugins.ResidentStopped)} + } + out := AdminPluginRuntime{Resident: v.Resident, State: v.State, RestartCount: v.RestartCount, LastError: v.LastError} + if out.State == "" { + out.State = string(plugins.ResidentStopped) + } + if v.LastStartedAt != nil { + out.LastStartedAt = new(NewInstant(*v.LastStartedAt)) + } + if v.NextRestartAt != nil { + out.NextRestartAt = new(NewInstant(*v.NextRestartAt)) + } + return out +} + func adminPluginInstallationOf(v handlers.PluginInstallationView) (AdminPluginInstallation, error) { caps, err := adminPluginCapabilitiesOf(v.Capabilities) if err != nil { @@ -320,7 +353,7 @@ func adminPluginInstallationOf(v handlers.PluginInstallationView) (AdminPluginIn if err != nil { return AdminPluginInstallation{}, err } - out := AdminPluginInstallation{ID: IDFromInt(int64(v.ID)), PluginID: v.PluginID, Version: v.Version, InstallPath: v.InstallPath, Enabled: v.Enabled, Kind: v.Kind, UpdatePolicy: v.UpdatePolicy, SourceKind: v.SourceKind, RepositoryName: v.RepositoryName, RepoURL: v.RepoURL, Presentation: adminPluginPresentationOf(v.Presentation), UpdatesPaused: v.UpdatesPaused, Capabilities: caps, GlobalConfigSchema: global, UserConfigSchema: user, Routes: pluginRoutesOf(v.Routes), Assets: pluginAssetsOf(v.Assets), Metadata: NonNilMap(PluginJSONValue(v.Metadata)), GlobalConfigs: make([]AdminPluginConfigValue, 0, len(v.GlobalConfigs)), AuthBindings: make([]AdminPluginAuthBinding, 0, len(v.AuthBindings)), TaskBindings: make([]AdminPluginTaskBinding, 0, len(v.TaskBindings)), CreatedAt: NewInstant(v.CreatedAt), UpdatedAt: NewInstant(v.UpdatedAt)} + out := AdminPluginInstallation{ID: IDFromInt(int64(v.ID)), PluginID: v.PluginID, Version: v.Version, InstallPath: v.InstallPath, Enabled: v.Enabled, Kind: v.Kind, UpdatePolicy: v.UpdatePolicy, SourceKind: v.SourceKind, RepositoryName: v.RepositoryName, RepoURL: v.RepoURL, Presentation: adminPluginPresentationOf(v.Presentation), UpdatesPaused: v.UpdatesPaused, Capabilities: caps, GlobalConfigSchema: global, UserConfigSchema: user, Routes: pluginRoutesOf(v.Routes), Assets: pluginAssetsOf(v.Assets), Metadata: NonNilMap(PluginJSONValue(v.Metadata)), GlobalConfigs: make([]AdminPluginConfigValue, 0, len(v.GlobalConfigs)), AuthBindings: make([]AdminPluginAuthBinding, 0, len(v.AuthBindings)), TaskBindings: make([]AdminPluginTaskBinding, 0, len(v.TaskBindings)), Runtime: adminPluginRuntimeOf(v.Runtime), CreatedAt: NewInstant(v.CreatedAt), UpdatedAt: NewInstant(v.UpdatedAt)} if v.RepositoryID != nil { out.RepositoryID = new(IDFromInt(int64(*v.RepositoryID))) } diff --git a/internal/apiv2/admin_plugin_installations_test.go b/internal/apiv2/admin_plugin_installations_test.go index 02dfe3d186..5fd9931270 100644 --- a/internal/apiv2/admin_plugin_installations_test.go +++ b/internal/apiv2/admin_plugin_installations_test.go @@ -109,7 +109,7 @@ func TestAdminPluginInstallationsRead(t *testing.T) { t.Fatal(rec.Code, rec.Body.String()) } raw := rec.Body.String() - for _, want := range []string{`"value":{"region":"us-east"}`, `"configured_secrets":["api_key"]`, `"trigger":{"cron":"0 * * * *"}`, `"display_order":2`, `2026-09-01T00:00:00.123Z`, `"capabilities":[]`, `"metadata":{}`} { + for _, want := range []string{`"value":{"region":"us-east"}`, `"configured_secrets":["api_key"]`, `"trigger":{"cron":"0 * * * *"}`, `"display_order":2`, `2026-09-01T00:00:00.123Z`, `"capabilities":[]`, `"metadata":{}`, `"runtime":{"resident":false,"state":"stopped","restart_count":0}`} { if !strings.Contains(raw, want) { t.Fatalf("missing %s in %s", want, raw) } diff --git a/internal/apiv2/admin_plugin_lifecycle.go b/internal/apiv2/admin_plugin_lifecycle.go index b910e84060..1ed08cdd10 100644 --- a/internal/apiv2/admin_plugin_lifecycle.go +++ b/internal/apiv2/admin_plugin_lifecycle.go @@ -7,17 +7,19 @@ import ( "strings" "github.com/Silo-Server/silo-server/internal/api/handlers" + "github.com/Silo-Server/silo-server/internal/plugins" ) // AdminPluginLifecycleService is the slice of *handlers.PluginHandler the // installation lifecycle uses: install from catalog or archive URL, partial -// assignment, apply the recorded update, and uninstall. Every method refuses -// the reserved builtin row and reports an unknown installation as -// plugins.ErrInstallationNotFound. +// assignment, apply the recorded update, restart the process, and uninstall. +// Every method refuses the reserved builtin row and reports an unknown +// installation as plugins.ErrInstallationNotFound. type AdminPluginLifecycleService interface { CreateAdminPluginInstallation(context.Context, handlers.PluginInstallationCreateInput) (handlers.PluginInstallationView, error) UpdateAdminPluginInstallation(context.Context, int, handlers.PluginInstallationUpdateInput) (handlers.PluginInstallationView, error) ApplyAdminPluginUpdate(context.Context, int) (handlers.PluginInstallationView, error) + RestartAdminPluginInstallation(context.Context, int) (handlers.PluginInstallationView, error) DeleteAdminPluginInstallation(context.Context, int) error } @@ -54,6 +56,8 @@ func adminPluginLifecycleProblem(err error) error { switch { case errors.Is(err, handlers.ErrPluginUpdateUnavailable): return NewProblem(TypeConflict, "This installation has no applicable update.") + case errors.Is(err, plugins.ErrInstallationDisabled): + return NewProblem(TypeConflict, "This installation is disabled; enable it before restarting.") case errors.As(err, &apiErr) && apiErr.Status == http.StatusBadRequest && apiErr.Field != "": return validationProblem(locationBody+"."+apiErr.Field, codeInvalid, apiErr.Message) } @@ -130,6 +134,18 @@ func registerAdminPluginLifecycle(reg *Registry) { } return adminPluginInstallationOutput(reg.deps.AdminPluginLifecycle.ApplyAdminPluginUpdate(ctx, id)) }) + restart := op(http.MethodPost, "/{id}/restart", "restartAdminPluginInstallation", "Stop the installation's process and, for a resident plugin (one the server supervises, such as a network access provider), start it again with a fresh failure budget; the response's runtime reports the outcome, including a launch that failed. A non-resident plugin is only stopped and launches on its next use. A disabled installation is 409. Repeating the request converges on one running process.", RetrySafetyNaturalIdempotent) + restart.Errors = append(restart.Errors, http.StatusNotFound, http.StatusConflict) + Register(reg, restart, func(ctx context.Context, in *AdminPluginInstallationIDInput) (*AdminPluginInstallationOutput, error) { + if reg.deps.AdminPluginLifecycle == nil { + return nil, unavailable("plugin lifecycle") + } + id, p := adminPluginInstallationID(in.ID) + if p != nil { + return nil, p + } + return adminPluginInstallationOutput(reg.deps.AdminPluginLifecycle.RestartAdminPluginInstallation(ctx, id)) + }) remove := op(http.MethodDelete, "/{id}", "deleteAdminPluginInstallation", "Stop the plugin, delete the installation row (configuration, bindings and archives cascade) and remove its files. Row delete and file removal are not one transaction and a repeat finds no row: a later 404 is not this caller's receipt; never automatically retry.", RetrySafetyNonRetryable) remove.DefaultStatus = http.StatusNoContent remove.Errors = append(remove.Errors, http.StatusNotFound, http.StatusConflict) diff --git a/internal/apiv2/admin_plugin_lifecycle_test.go b/internal/apiv2/admin_plugin_lifecycle_test.go index 3579492dcf..49f5b8a931 100644 --- a/internal/apiv2/admin_plugin_lifecycle_test.go +++ b/internal/apiv2/admin_plugin_lifecycle_test.go @@ -58,6 +58,17 @@ func (f *fakePluginLifecycle) ApplyAdminPluginUpdate(_ context.Context, id int) v.Version = "1.1.0" return v, nil } +func (f *fakePluginLifecycle) RestartAdminPluginInstallation(_ context.Context, id int) (handlers.PluginInstallationView, error) { + f.calls++ + f.lastID = id + if f.err != nil { + return handlers.PluginInstallationView{}, f.err + } + v := f.view(id) + at := time.Date(2026, 9, 7, 0, 0, 1, 0, time.UTC) + v.Runtime = &handlers.PluginRuntimeView{Resident: true, State: "running", RestartCount: 2, LastStartedAt: &at} + return v, nil +} func (f *fakePluginLifecycle) DeleteAdminPluginInstallation(_ context.Context, id int) error { f.calls++ f.lastID = id @@ -131,18 +142,29 @@ func TestAdminPluginLifecycleUpdateApplyDelete(t *testing.T) { if rec.Code != http.StatusOK || f.lastID != 7 || !strings.Contains(rec.Body.String(), `"version":"1.1.0"`) { t.Fatal(rec.Code, rec.Body.String()) } + requireProblem(t, do(t, h, http.MethodPost, base+"/restart", "", bearer(memberToken)), TypePermissionDenied) + rec = do(t, h, http.MethodPost, base+"/restart", "", bearer(adminToken)) + if rec.Code != http.StatusOK || f.lastID != 7 { + t.Fatal(rec.Code, rec.Body.String()) + } + for _, want := range []string{`"runtime":{"resident":true,"state":"running","restart_count":2,"last_started_at":"2026-09-07T00:00:01.000Z"}`} { + if !strings.Contains(rec.Body.String(), want) { + t.Fatalf("restart body missing %s: %s", want, rec.Body.String()) + } + } rec = do(t, h, http.MethodDelete, base, "", bearer(adminToken)) - if rec.Code != http.StatusNoContent || rec.Body.Len() != 0 || f.calls != 3 { + if rec.Code != http.StatusNoContent || rec.Body.Len() != 0 || f.calls != 4 { t.Fatal(rec.Code, rec.Body.String(), f.calls) } // Seam errors map to one problem each on every mutation. for _, tc := range []struct { err error want ProblemType - }{{plugins.ErrInstallationNotFound, TypeNotFound}, {handlers.ErrPluginBuiltinInstallation, TypeConflict}, {handlers.ErrPluginUpdateUnavailable, TypeConflict}, {errors.New("stop plugin: boom"), TypeInternalError}} { + }{{plugins.ErrInstallationNotFound, TypeNotFound}, {handlers.ErrPluginBuiltinInstallation, TypeConflict}, {handlers.ErrPluginUpdateUnavailable, TypeConflict}, {plugins.ErrInstallationDisabled, TypeConflict}, {errors.New("stop plugin: boom"), TypeInternalError}} { f.err = tc.err requireProblem(t, do(t, h, http.MethodPut, base, `{"enabled":true}`, bearer(adminToken)), tc.want) requireProblem(t, do(t, h, http.MethodPost, base+"/update", "", bearer(adminToken)), tc.want) + requireProblem(t, do(t, h, http.MethodPost, base+"/restart", "", bearer(adminToken)), tc.want) requireProblem(t, do(t, h, http.MethodDelete, base, "", bearer(adminToken)), tc.want) } deps.AdminPluginLifecycle = nil diff --git a/internal/apiv2/admin_sessions.go b/internal/apiv2/admin_sessions.go index ea7ec0dbe5..2c67d6a2ad 100644 --- a/internal/apiv2/admin_sessions.go +++ b/internal/apiv2/admin_sessions.go @@ -72,6 +72,7 @@ type AdminPlaybackSession struct { AudioDecision string `json:"audio_decision,omitempty"` EffectivePlayMethod string `json:"effective_play_method,omitempty"` IsJellyfinClient bool `json:"is_jellyfin_client,omitzero"` + RoutingNetworkProvider *string `json:"routing_network_provider,omitempty" doc:"Access network selected for playback: empty means default; absent means unknown; otherwise the validated provider identifier."` RoutingWorkload string `json:"routing_workload,omitempty"` RoutingExecution string `json:"routing_execution,omitempty"` RoutingExecutionNodeID *ID `json:"routing_execution_node_id,omitempty"` @@ -146,6 +147,7 @@ func adminPlaybackSessionOf(v handlers.AdminPlaybackSessionView) AdminPlaybackSe AudioDecision: v.AudioDecision, EffectivePlayMethod: v.EffectivePlayMethod, IsJellyfinClient: v.IsJellyfinClient, + RoutingNetworkProvider: v.RoutingNetworkProvider, RoutingWorkload: v.RoutingWorkload, RoutingExecution: v.RoutingExecution, RoutingExecutionNodeID: adminSessionNodeID(v.RoutingExecutionNodeID), @@ -186,6 +188,7 @@ type AdminPlaybackSessionCapabilitiesOutputBody struct { ClientBuild bool `json:"client_build"` ClientChannel bool `json:"client_channel"` TargetAudioChannels bool `json:"target_audio_channels"` + NetworkAccessRoute bool `json:"network_access_route"` NodeRouting bool `json:"node_routing"` } @@ -213,6 +216,7 @@ func registerAdminPlaybackSessions(reg *Registry) { out.Body.ClientChannel = f.ClientChannel out.Body.TargetAudioChannels = f.TargetAudioChannels out.Body.NodeRouting = f.NodeRouting + out.Body.NetworkAccessRoute = true return out, nil }) Register(reg, op("", opListAdminPlaybackSessions), func(ctx context.Context, in *AdminPlaybackSessionsInput) (*AdminPlaybackSessionsOutput, error) { diff --git a/internal/apiv2/admin_sessions_postgres_test.go b/internal/apiv2/admin_sessions_postgres_test.go index 8a557f8f27..febc046800 100644 --- a/internal/apiv2/admin_sessions_postgres_test.go +++ b/internal/apiv2/admin_sessions_postgres_test.go @@ -49,6 +49,10 @@ func TestAdminSessionPagesBeyondBridgeLimitPostgres(t *testing.T) { SELECT 'session-'||lpad(i::text,3,'0'),1,'primary',0,'direct_play','api',now()+i*interval '1 second',now() FROM generate_series(1,205) i`); err != nil { t.Fatal(err) } + + if _, err := pool.Exec(t.Context(), `UPDATE playback_sessions_sync SET routing_network_provider = CASE WHEN session_id = 'session-001' THEN 'tailscale' ELSE '' END WHERE session_id IN ('session-001', 'session-002')`); err != nil { + t.Fatal(err) + } loader := handlers.NewPlaybackSessionsLoader(pool, nil, nil) legacy, err := loader.Load(t.Context(), handlers.PlaybackSessionsQuery{}) if err != nil || len(legacy) != 200 || legacy[0].SessionID != "session-205" || legacy[199].SessionID != "session-006" { @@ -76,6 +80,21 @@ func TestAdminSessionPagesBeyondBridgeLimitPostgres(t *testing.T) { t.Fatalf("page%d: %s", pageNumber, rec.Body.String()) } for _, row := range page.Items { + + switch row.SessionID { + case "session-001": + if row.RoutingNetworkProvider == nil || *row.RoutingNetworkProvider != "tailscale" { + t.Fatalf("overlay provider: %#v", row) + } + case "session-002": + if row.RoutingNetworkProvider == nil || *row.RoutingNetworkProvider != "" { + t.Fatalf("default provider: %#v", row) + } + default: + if row.RoutingNetworkProvider != nil { + t.Fatalf("unknown provider: %#v", row) + } + } ids = append(ids, row.SessionID) } cursor = page.Page.NextCursor diff --git a/internal/apiv2/admin_sessions_test.go b/internal/apiv2/admin_sessions_test.go index 7d97745571..e35fef7a45 100644 --- a/internal/apiv2/admin_sessions_test.go +++ b/internal/apiv2/admin_sessions_test.go @@ -23,7 +23,7 @@ func (f *fakeAdminPlaybackSessions) ReadAdminPlaybackSessions(_ context.Context, f.lastLimit = limit at := time.Date(2026, 1, 2, 3, 4, 5, 123456789, time.FixedZone("offset", 3600)) rows := []handlers.AdminPlaybackSessionView{ - {SessionID: "b", UserID: 7, ProfileID: "child", MediaFileID: 42, RequestedMediaFileID: 41, StartedAt: at, UpdatedAt: at, RoutingExecutionNodeID: new(9), TargetAudioChannels: new(2), SourceAudioChannels: new(8), EffectivePlayMethod: "transcode", IsJellyfinClient: true, HasPlaybackControl: true}, + {SessionID: "b", UserID: 7, ProfileID: "child", MediaFileID: 42, RequestedMediaFileID: 41, StartedAt: at, UpdatedAt: at, RoutingExecutionNodeID: new(9), RoutingNetworkProvider: new("tailscale"), TargetAudioChannels: new(2), SourceAudioChannels: new(8), EffectivePlayMethod: "transcode", IsJellyfinClient: true, HasPlaybackControl: true}, {SessionID: "a", UserID: 7, ProfileID: "primary", MediaFileID: 44, RequestedMediaFileID: 44, StartedAt: at, UpdatedAt: at}, } slices.SortFunc(rows, func(a, b handlers.AdminPlaybackSessionView) int { return cmp.Compare(a.SessionID, b.SessionID) }) @@ -81,7 +81,7 @@ func TestAdminPlaybackSessionReadProjection(t *testing.T) { t.Fatal(err) } row := page.Items[0] - if rec.Code != 200 || row.SessionID != "b" || row.ProfileID != "child" || row.MediaFileID != "42" || row.RequestedMediaFileID != "41" || row.RoutingExecutionNodeID == nil || *row.RoutingExecutionNodeID != "9" || *row.TargetAudioChannels != 2 || *row.SourceAudioChannels != 8 || !row.IsJellyfinClient || !row.HasPlaybackControl || page.Page.HasMore { + if rec.Code != 200 || row.SessionID != "b" || row.RoutingNetworkProvider == nil || *row.RoutingNetworkProvider != "tailscale" || row.ProfileID != "child" || row.MediaFileID != "42" || row.RequestedMediaFileID != "41" || row.RoutingExecutionNodeID == nil || *row.RoutingExecutionNodeID != "9" || *row.TargetAudioChannels != 2 || *row.SourceAudioChannels != 8 || !row.IsJellyfinClient || !row.HasPlaybackControl || page.Page.HasMore { t.Fatal(rec.Body.String()) } if !strings.Contains(rec.Body.String(), `"started_at":"2026-01-02T02:04:05.123Z"`) { @@ -163,3 +163,34 @@ func adminSessionFixtureCases() []fixtureCase { {name: "admin_playback_sessions", operationID: "listAdminPlaybackSessions", method: "GET", path: Prefix + "/admin/sessions", headers: bearer(adminToken), status: 200, schema: "#/components/schemas/CollectionAdminPlaybackSession", assertHeaders: []string{"Content-Type"}, scenario: "Diagnostic rows retain account/profile and chosen/requested file distinctions with canonical string IDs and instants."}, } } + +func TestAdminSessionNetworkProviderProjection(t *testing.T) { + for _, provider := range []*string{nil, new(""), new("tailscale")} { + view := handlers.AdminPlaybackSessionView{RoutingNetworkProvider: provider, StartedAt: time.Now(), UpdatedAt: time.Now()} + wire, err := json.Marshal(adminPlaybackSessionOf(view)) + if err != nil { + t.Fatal(err) + } + var fields map[string]json.RawMessage + if err := json.Unmarshal(wire, &fields); err != nil { + t.Fatal(err) + } + raw, present := fields["routing_network_provider"] + if present != (provider != nil) { + t.Fatalf("wrong field presence: %s", wire) + } + if present { + var got string + if err := json.Unmarshal(raw, &got); err != nil || got != *provider { + t.Fatalf("provider = %s", raw) + } + } + legacy, err := json.Marshal(view) + if err != nil { + t.Fatal(err) + } + if strings.Contains(string(legacy), "routing_network_provider") { + t.Fatal("v2 field leaked into v1") + } + } +} diff --git a/internal/apiv2/direct_download_test.go b/internal/apiv2/direct_download_test.go index 09ef7e3384..cff066454b 100644 --- a/internal/apiv2/direct_download_test.go +++ b/internal/apiv2/direct_download_test.go @@ -140,7 +140,7 @@ type directPlannerFixture struct { plans, releases int } -func (p *directPlannerFixture) PlanDownload(string, ...string) nodepool.Plan { +func (p *directPlannerFixture) PlanDownloadWith(string, func(*nodepool.Node) bool, ...string) nodepool.Plan { p.plans++ return nodepool.Plan{ProxyNode: &nodepool.Node{URL: p.url}} } diff --git a/internal/apiv2/document.go b/internal/apiv2/document.go index c648decd98..ddd507d5f5 100644 --- a/internal/apiv2/document.go +++ b/internal/apiv2/document.go @@ -517,6 +517,7 @@ func registerAll(reg *Registry) { registerOrderedApplePush(reg) registerEmailVerification(reg) registerEventsCapability(reg) + registerNetworkAccess(reg) registerEventsSocket(reg) registerAdminLogsSocket(reg) registerPlaybackControlSocket(reg) diff --git a/internal/apiv2/document_test.go b/internal/apiv2/document_test.go index 0730ae176b..823c8a38e8 100644 --- a/internal/apiv2/document_test.go +++ b/internal/apiv2/document_test.go @@ -147,6 +147,10 @@ func TestGeneratedDocumentStatuses(t *testing.T) { bodies := 0 expect := map[string]map[int]bool{ "getEventsCapabilities": {http.StatusOK: true, http.StatusServiceUnavailable: true}, + "getNetworkAccessCapabilities": {http.StatusOK: true, http.StatusServiceUnavailable: true}, + "getAdminNetworkAccessStatus": {http.StatusOK: true, http.StatusNotFound: true}, + "connectNetworkAccess": {http.StatusAccepted: true, http.StatusOK: false, http.StatusNotFound: true, http.StatusRequestTimeout: true}, + "disconnectNetworkAccess": {http.StatusAccepted: true, http.StatusOK: false, http.StatusNotFound: true, http.StatusRequestTimeout: true}, "getSetupStatus": {http.StatusServiceUnavailable: true}, "getSystemInfo": {http.StatusServiceUnavailable: false}, "getOpenAPIDocument": {http.StatusServiceUnavailable: false}, @@ -336,7 +340,7 @@ func TestGeneratedDocumentStatuses(t *testing.T) { } expect[opCreateProgressSnapshot] = map[int]bool{http.StatusCreated: true, http.StatusConflict: true, http.StatusRequestEntityTooLarge: true, http.StatusTooManyRequests: true} expect[opGetProgressSnapshot] = map[int]bool{http.StatusOK: true, http.StatusConflict: true, http.StatusNotFound: true} - for _, id := range []string{"listAdminPluginCatalog", "listAdminPluginInstallations", "createAdminPluginInstallation", "updateAdminPluginInstallation", "applyAdminPluginUpdate", "deleteAdminPluginInstallation", "uploadAdminPluginInstallation", "createAdminPluginUpload", "putAdminPluginUploadChunk", "completeAdminPluginUpload", "cancelAdminPluginUpload", "updateAdminPluginInstallationConfig", "testAdminPluginInstallationConfig", "updateAdminPluginAuthBinding", "updateAdminPluginTaskBinding", "forceReloadAdminNodes", "forceReloadAdminNode", "checkAdminNode", "reprobeAdminNode", "triggerAdminAutoscan", "createAdminAutoscanSourceWebhook", "rotateAdminAutoscanSourceWebhook", "deleteAdminAutoscanSourceWebhook", "createAdminAutoscanSource", "updateAdminAutoscanSource", "saveAdminDashboardLayout", "deleteAdminAutoscanSource", "resetAdminDashboardLayout", "updateAdminAutoscanSettings", "listAdminAutoscanEvents", "listAdminAutoscanScans", "deleteAdminPluginRepository", "updateAdminPluginRepository", "deleteAdminAutoscanConnection", "updateAdminAutoscanConnection", "createAdminAutoscanConnection", "testAdminAutoscanConnection", "createAdminPluginRepository", "getAdminStreamTelemetryParity", "listAdminPluginRepositories", "getAdminHardwareAcceleration", "getAdminDashboardLayout", "getAdminAutoscanRewriteSuggestions", "listAdminAutoscanAvailableSources", "listAdminAuditLogs", "listAdminOperationalLogs", "getAdminDashboardCapabilities", "updateAdminJellyfinCompatSettings", "getAdminJellyfinCompatStatus", "getAdminSetting", "getAdminSectionSettings", "getAdminPlaybackRoutingCapabilities", "updateAdminRateLimitConfig", "getAdminRateLimitConfig", "getAdminRateLimitStatus", "sendAdminTestEmail", "getAdminServerStatus", "listAdminNodes", "getAdminDashboardTimeseries", "getAdminDashboardPlaybackActivity", "getAdminDashboardTopActivity", "getAdminDashboardDownloadsStats", "deleteAdminDiagnosticReport", "listAdminDiagnosticReports", "getAdminDiagnosticReport", "downloadAdminDiagnosticReport", "getAdminBuildInfo", "getAdminSystemResources", "getAdminResourceCapabilities", "getAdminStoredSettings", "updateAdminSettings", "updateAdminSetting", "getAdminEffectiveSettings", "getAdminRestartKeys", "getAdminSensitiveSettingsStatus", "checkAdminSettingsConnection", "getAdminPluginCatalogSettings", "getAdminPluginCatalogStatus", "updateAdminPluginCatalogSettings", "createAdminNode", "updateAdminNode", "deleteAdminNode", "uploadAdminBrandingAsset", "deleteAdminBrandingAsset", "installAdminJellyfinCompatWeb", "removeAdminJellyfinCompatWeb", "requestAdminServerRestart"} { + for _, id := range []string{"listAdminPluginCatalog", "listAdminPluginInstallations", "createAdminPluginInstallation", "updateAdminPluginInstallation", "applyAdminPluginUpdate", "restartAdminPluginInstallation", "deleteAdminPluginInstallation", "uploadAdminPluginInstallation", "createAdminPluginUpload", "putAdminPluginUploadChunk", "completeAdminPluginUpload", "cancelAdminPluginUpload", "updateAdminPluginInstallationConfig", "testAdminPluginInstallationConfig", "updateAdminPluginAuthBinding", "updateAdminPluginTaskBinding", "forceReloadAdminNodes", "forceReloadAdminNode", "checkAdminNode", "reprobeAdminNode", "triggerAdminAutoscan", "createAdminAutoscanSourceWebhook", "rotateAdminAutoscanSourceWebhook", "deleteAdminAutoscanSourceWebhook", "createAdminAutoscanSource", "updateAdminAutoscanSource", "saveAdminDashboardLayout", "deleteAdminAutoscanSource", "resetAdminDashboardLayout", "updateAdminAutoscanSettings", "listAdminAutoscanEvents", "listAdminAutoscanScans", "deleteAdminPluginRepository", "updateAdminPluginRepository", "deleteAdminAutoscanConnection", "updateAdminAutoscanConnection", "createAdminAutoscanConnection", "testAdminAutoscanConnection", "createAdminPluginRepository", "getAdminStreamTelemetryParity", "listAdminPluginRepositories", "getAdminHardwareAcceleration", "getAdminDashboardLayout", "getAdminAutoscanRewriteSuggestions", "listAdminAutoscanAvailableSources", "listAdminAuditLogs", "listAdminOperationalLogs", "getAdminDashboardCapabilities", "updateAdminJellyfinCompatSettings", "getAdminJellyfinCompatStatus", "getAdminSetting", "getAdminSectionSettings", "getAdminPlaybackRoutingCapabilities", "updateAdminRateLimitConfig", "getAdminRateLimitConfig", "getAdminRateLimitStatus", "sendAdminTestEmail", "getAdminServerStatus", "listAdminNodes", "getAdminDashboardTimeseries", "getAdminDashboardPlaybackActivity", "getAdminDashboardTopActivity", "getAdminDashboardDownloadsStats", "deleteAdminDiagnosticReport", "listAdminDiagnosticReports", "getAdminDiagnosticReport", "downloadAdminDiagnosticReport", "getAdminBuildInfo", "getAdminSystemResources", "getAdminResourceCapabilities", "getAdminStoredSettings", "updateAdminSettings", "updateAdminSetting", "getAdminEffectiveSettings", "getAdminRestartKeys", "getAdminSensitiveSettingsStatus", "checkAdminSettingsConnection", "getAdminPluginCatalogSettings", "getAdminPluginCatalogStatus", "updateAdminPluginCatalogSettings", "createAdminNode", "updateAdminNode", "deleteAdminNode", "uploadAdminBrandingAsset", "deleteAdminBrandingAsset", "installAdminJellyfinCompatWeb", "removeAdminJellyfinCompatWeb", "requestAdminServerRestart", "getAdminNetworkAccessStatus", "connectNetworkAccess", "disconnectNetworkAccess"} { profileToken[id] = true } for _, id := range []string{"listAdminItemImages", "applyAdminItemImage", "listAdminUnmatchedFiles", "searchAdminItemMatches", "applyAdminItemMatch", "listAdminItemFiles", "splitAdminItem", "mergeAdminItem", "refreshAdminItemMetadata", "updateAdminItemMetadata", "translateAdminItemMetadata", "listAdminMetadataTranslationJobs", "cancelAdminMetadataTranslation", "refreshAdminPerson", "updateAdminPerson", "getAdminRecommendationsStatus", "triggerAdminRecommendationEmbeddings", "triggerAdminRecommendationTasteProfiles", "triggerAdminRecommendationCowatch", "triggerAdminRecommendationRefresh", "listAdminLiteraryCandidates", "linkAdminLiteraryItems", "confirmAdminLiteraryMatch", "ignoreAdminLiteraryMatch", "unlinkAdminLiteraryItem", opExportAdminCatalog, "createCatalogExportJob", "createCatalogImportJob", "importAdminCatalog", "publishCatalogExportJob", "getAdminCatalogSearchStatus", "listCatalogImportSources", "listLocalCatalogImportSources", "browseAdminFilesystem", "listAdminTasks", "getAdminTask", "runAdminTask", "cancelAdminTask", "getAdminTaskSchedule", "updateAdminTaskSchedule", "listAdminTaskHistory", "getAdminTaskMetrics", "listAdminJobs"} { diff --git a/internal/apiv2/fixtures_test.go b/internal/apiv2/fixtures_test.go index 0f54a70900..e33a8c07b8 100644 --- a/internal/apiv2/fixtures_test.go +++ b/internal/apiv2/fixtures_test.go @@ -1690,7 +1690,7 @@ func fixtureCases() []fixtureCase { cases = append(cases, playbackFixtureCases()...) cases = append(cases, playbackRouteEventFixtureCases()...) cases = append(cases, playbackReplanFixtureCases()...) - return append(cases, []fixtureCase{ + cases = append(cases, []fixtureCase{ {name: "get_image_capabilities_ok", operationID: "getImageCapabilities", scenario: "Image discovery advertises the supported season-list artwork parameter.", method: http.MethodGet, path: "/api/v2/images/capabilities", headers: viewer, @@ -1715,6 +1715,9 @@ func fixtureCases() []fixtureCase { method: http.MethodPost, path: Prefix + "/auth/device/start", body: `{"temporary":null}`, status: http.StatusUnprocessableEntity, assertHeaders: []string{"Content-Type", "Cache-Control"}, schema: problem}, }...) + // Request ids are positional: new fixtures append here so committed + // fixtures keep their ids. + return append(cases, networkAccessFixtureCases()...) } // fixtureMultipartType is the multipart Content-Type of the avatar fixtures, @@ -1739,6 +1742,7 @@ func profileOwner() map[string]string { return with(bearer(memberToken), "X-Prof // produced by the gate translation the production limiter goes through. func fixtureDeps() Dependencies { deps := pilotDeps(&fakeProgress{entries: progressRows()}, nil) + deps.NetworkAccess = newFakeNetworkAccess() deps.SubtitleAIReads = &fakeSubtitleAIReads{} deps.Downloads = &fakeDownloadRegistry{} deps.DownloadCreation = &fakeDownloadCreation{row: &downloads.Download{ID: "entry", ContentID: "movie", MediaFileID: 42, Revision: 1, CreatedAt: time.Date(2026, 1, 2, 3, 4, 5, 0, time.UTC), Status: downloads.StatusReady, Quality: downloads.QualityOriginal, EffectiveQuality: downloads.QualityOriginal, Format: downloads.FormatOriginal, DeviceID: "device-one"}, page: downloads.CreatePage{BatchID: "intent", Skipped: []downloads.SkippedDownload{{EpisodeID: "missing", Reason: "no_file"}}}} diff --git a/internal/apiv2/network_access.go b/internal/apiv2/network_access.go new file mode 100644 index 0000000000..2dd48d4c45 --- /dev/null +++ b/internal/apiv2/network_access.go @@ -0,0 +1,240 @@ +package apiv2 + +import ( + "bytes" + "context" + "encoding/json" + "errors" + "net/http" + + "github.com/Silo-Server/silo-server/internal/netaccess" + "github.com/Silo-Server/silo-server/internal/plugins" +) + +// NetworkAccessService is the slice of *plugins.Service the network access +// operations use: provider discovery from manifests, and per-host status, +// connect and disconnect through the resident provider plugin. +type NetworkAccessService interface { + ListNetworkAccessProviders(context.Context) ([]plugins.NetworkAccessProvider, error) + NetworkAccessStatus(ctx context.Context, provider string) (plugins.NetworkAccessReport, error) + ConnectNetworkAccess(ctx context.Context, provider string, hosts []string) (plugins.NetworkAccessReport, error) + DisconnectNetworkAccess(ctx context.Context, provider string, hosts []string) (plugins.NetworkAccessReport, error) +} + +// NetworkAccessProviderSummary is one installed provider as the capability +// document lists it, read from the plugin manifest without launching it. +type NetworkAccessProviderSummary struct { + Provider string `json:"provider" doc:"Stable provider slug used in the admin routes, e.g. tailscale"` + DisplayName string `json:"display_name"` + InstallationID ID `json:"installation_id" doc:"Plugin installation declaring network_access_provider.v1"` +} + +// NetworkAccessCapabilities is the server-wide capability document: available +// when at least one enabled installation declares a provider, not_configured +// otherwise. Provider health is not state; read it from the admin status. +type NetworkAccessCapabilities struct { + Capability + Providers []NetworkAccessProviderSummary `json:"providers"` +} +type NetworkAccessCapabilitiesOutput struct { + Status int + ETag string `header:"ETag"` + CacheControl string `header:"Cache-Control"` + Body NetworkAccessCapabilities +} + +func (c NetworkAccessCapabilities) capabilityState() string { + return configuredCapabilityState(len(c.Providers) > 0) +} + +// NetworkAccessHostRef identifies one process running the provider. +type NetworkAccessHostRef struct { + ID string `json:"id" doc:"api for the API server; node: for a proxy node"` + Role string `json:"role" enum:"api,proxy"` + Name string `json:"name" doc:"Server name for the API host; node name for a proxy"` +} + +// NetworkAccessHostStatus is the provider's state on one host. +type NetworkAccessHostStatus struct { + Host NetworkAccessHostRef `json:"host"` + State string `json:"state" enum:"disconnected,awaiting_authorization,connecting,connected,error,unavailable" doc:"Published host state; provider states outside this vocabulary map to error"` + RawState string `json:"raw_state,omitempty" doc:"the provider's own state string when it is not one of the published states"` + Hostname string `json:"hostname,omitempty" doc:"Overlay DNS name"` + Origin string `json:"origin,omitempty" doc:"scheme://host[:port] clients reach the API listener at over the overlay"` + Addresses []string `json:"addresses" doc:"Overlay IP addresses"` + AuthURL string `json:"auth_url,omitempty" doc:"Interactive enrollment URL, only while awaiting_authorization; admin-only, never logged"` + Error string `json:"error,omitempty"` + ProviderVersion string `json:"provider_version,omitempty"` + UpdatedAt *Instant `json:"updated_at,omitempty" doc:"When the host last heard from the provider; absent while unavailable"` +} + +// NetworkAccessStatus is the provider's state across every host. +type NetworkAccessStatus struct { + Provider string `json:"provider"` + Hosts []NetworkAccessHostStatus `json:"hosts"` +} +type NetworkAccessStatusOutput struct{ Body NetworkAccessStatus } + +type NetworkAccessProviderInput struct { + Provider string `path:"provider" minLength:"1" maxLength:"64" pattern:"^[a-z0-9]+(?:[._-][a-z0-9]+)*$"` +} + +// NetworkAccessCommand selects the hosts a connect or disconnect targets. +type NetworkAccessCommand struct { + Hosts []string `json:"hosts,omitempty" maxItems:"256" doc:"Host ids to act on (api, node:); omitted means every host"` + // hostsNull records an explicit {"hosts": null}. The decoder leaves Hosts + // nil for null and for omission alike, and omission means every host; a + // null must not widen a malformed request to the whole deployment. The + // body is optional; RawBody separately records a top-level null. + hostsNull bool +} + +func (c *NetworkAccessCommand) UnmarshalJSON(data []byte) error { + type plain NetworkAccessCommand + var decoded plain + if err := json.Unmarshal(data, &decoded); err != nil { + return err + } + var members map[string]json.RawMessage + if err := json.Unmarshal(data, &members); err != nil { + return err + } + *c = NetworkAccessCommand(decoded) + c.hostsNull = bytes.Equal(bytes.TrimSpace(members["hosts"]), jsonNull) + return nil +} + +type NetworkAccessCommandInput struct { + NetworkAccessProviderInput + Body *NetworkAccessCommand `required:"false"` + // RawBody distinguishes omitted bytes from a literal null, including + // requests whose transfer encoding leaves ContentLength unknown. + RawBody []byte +} + +// publishedNetworkAccessStates is the closed enum on NetworkAccessHostStatus.State. +// The plugin vocabulary is open, so anything else is coerced to error with the +// original value kept in raw_state. +var publishedNetworkAccessStates = map[string]bool{ + netaccess.StateDisconnected: true, + netaccess.StateAwaitingAuthorization: true, + netaccess.StateConnecting: true, + netaccess.StateConnected: true, + netaccess.StateError: true, + netaccess.StateUnavailable: true, +} + +func networkAccessStatusOf(report plugins.NetworkAccessReport) NetworkAccessStatus { + out := NetworkAccessStatus{Provider: report.Provider.Provider, Hosts: make([]NetworkAccessHostStatus, 0, len(report.Hosts))} + for _, host := range report.Hosts { + status := host.Status + entry := NetworkAccessHostStatus{ + Host: NetworkAccessHostRef{ID: host.Host.ID, Role: host.Host.Role, Name: host.Host.Name}, + State: status.State, + Hostname: status.Hostname, + Origin: status.Origin, + Addresses: append([]string{}, status.Addresses...), + AuthURL: status.AuthURL, + Error: status.Error, + ProviderVersion: status.ProviderVersion, + } + if entry.State == "" { + entry.State = netaccess.StateUnavailable + } else if !publishedNetworkAccessStates[entry.State] { + entry.RawState = entry.State + entry.State = netaccess.StateError + } + if !status.UpdatedAt.IsZero() { + entry.UpdatedAt = ptr(NewInstant(status.UpdatedAt)) + } + out.Hosts = append(out.Hosts, entry) + } + return out +} + +func networkAccessProblem(err error) error { + switch { + case errors.Is(err, plugins.ErrNetworkAccessProviderNotFound): + return NewProblem(TypeNotFound, "Network access provider not found.") + case errors.Is(err, plugins.ErrNetworkAccessHostUnknown): + return validationProblem(locationBody+".hosts", codeInvalid, "Unknown network access host.") + } + return serviceProblem(err) +} + +func registerNetworkAccess(reg *Registry) { + capability := Operation{Operation: humaOp(http.MethodGet, Prefix+"/network-access/capabilities", "getNetworkAccessCapabilities", "network-access", + "List the installed network access providers (overlay networks such as Tailscale that reach this server without port forwarding). available when an enabled plugin declares one; not_configured otherwise. Read from plugin manifests; no plugin is launched and no health is implied."), Class: ClassAuthenticated, ServiceBacked: true} + Register(reg, capability, func(ctx context.Context, _ *CapabilityInput) (*NetworkAccessCapabilitiesOutput, error) { + out := NetworkAccessCapabilities{Providers: []NetworkAccessProviderSummary{}} + if reg.deps.NetworkAccess == nil { + out.State = StateNotConfigured + return &NetworkAccessCapabilitiesOutput{Body: out}, nil + } + providers, err := reg.deps.NetworkAccess.ListNetworkAccessProviders(ctx) + if err != nil { + return nil, serviceProblem(err) + } + for _, provider := range providers { + out.Providers = append(out.Providers, NetworkAccessProviderSummary{Provider: provider.Provider, DisplayName: provider.DisplayName, InstallationID: IDFromInt(int64(provider.InstallationID))}) + } + return &NetworkAccessCapabilitiesOutput{Body: out}, nil + }) + + admin := func(method, path, id, summary string) Operation { + o := Operation{Operation: humaOp(method, Prefix+"/admin/network-access/{provider}"+path, id, "network-access", summary), Class: ClassActingAdmin, ServiceBacked: true} + o.Errors = append(o.Errors, http.StatusNotFound) + return o + } + Register(reg, admin(http.MethodGet, "/status", "getAdminNetworkAccessStatus", + "Read the provider's live state on every host that runs it, asking each plugin instance directly with a ten-second timeout. A host whose plugin process is not running answers state unavailable with the supervisor's last error. An unknown provider slug is 404."), + func(ctx context.Context, in *NetworkAccessProviderInput) (*NetworkAccessStatusOutput, error) { + if reg.deps.NetworkAccess == nil { + return nil, unavailable("network access") + } + report, err := reg.deps.NetworkAccess.NetworkAccessStatus(ctx, in.Provider) + if err != nil { + return nil, networkAccessProblem(err) + } + return &NetworkAccessStatusOutput{Body: networkAccessStatusOf(report)}, nil + }) + command := func(path, id, summary string, run func(NetworkAccessService, context.Context, string, []string) (plugins.NetworkAccessReport, error)) { + op := admin(http.MethodPost, path, id, summary) + op.DefaultStatus = http.StatusAccepted + op.DemoRestricted = true + op.RetrySafety = RetrySafetyNaturalIdempotent + op.MaxBodyBytes = 64 << 10 + Register(reg, op, func(ctx context.Context, in *NetworkAccessCommandInput) (*NetworkAccessStatusOutput, error) { + if reg.deps.NetworkAccess == nil { + return nil, unavailable("network access") + } + if in.Body == nil && len(in.RawBody) != 0 { + return nil, validationProblem(locationBody, codeInvalidType, "null is not a request body; send an object or omit the body to act on every host.") + } + if in.Body != nil && in.Body.hostsNull { + return nil, validationProblem(locationBody+".hosts", codeInvalidType, "null is not a value for this member; omit it to act on every host.") + } + var hosts []string + if in.Body != nil && in.Body.Hosts != nil { + hosts = in.Body.Hosts + if len(hosts) == 0 { + return nil, validationProblem(locationBody+".hosts", codeInvalid, "Name at least one host or omit hosts to act on every host.") + } + } + report, err := run(reg.deps.NetworkAccess, ctx, in.Provider, hosts) + if err != nil { + return nil, networkAccessProblem(err) + } + return &NetworkAccessStatusOutput{Body: networkAccessStatusOf(report)}, nil + }) + // Huma infers required=true from RawBody. Keep this operation's + // documented optional body in both decoding and the OpenAPI document. + registeredOperation(reg.api.OpenAPI(), op).RequestBody.Required = false + } + command("/connect", "connectNetworkAccess", + "Ask the provider on the named hosts (every host when hosts is omitted) to bring its overlay identity up and start proxying. Answers 202 with the state each host reached within ten seconds; enrollment may continue in the background (awaiting_authorization carries the auth_url), so poll status for the final state. Repeating the request converges on one connected instance per host. Hosts not named answer their current status; an unknown host id is 422.", + NetworkAccessService.ConnectNetworkAccess) + command("/disconnect", "disconnectNetworkAccess", + "Ask the provider on the named hosts (every host when hosts is omitted) to tear its overlay listener down and clear its desired-connected intent. Answers 202 with the state each host reached within ten seconds. Repeating the request converges on disconnected. Hosts not named answer their current status; an unknown host id is 422.", + NetworkAccessService.DisconnectNetworkAccess) +} diff --git a/internal/apiv2/network_access_fixtures_test.go b/internal/apiv2/network_access_fixtures_test.go new file mode 100644 index 0000000000..cda58cb6ba --- /dev/null +++ b/internal/apiv2/network_access_fixtures_test.go @@ -0,0 +1,17 @@ +package apiv2 + +import "net/http" + +func networkAccessFixtureCases() []fixtureCase { + problem := "#/components/schemas/Problem" + status := "#/components/schemas/NetworkAccessStatus" + return []fixtureCase{ + {name: "network_access_capabilities_ok", operationID: "getNetworkAccessCapabilities", scenario: "An authenticated account lists the installed overlay-network providers read from plugin manifests.", method: http.MethodGet, path: Prefix + "/network-access/capabilities", headers: bearer(memberToken), status: 200, assertHeaders: []string{"Content-Type", "Cache-Control", "ETag"}, schema: "#/components/schemas/NetworkAccessCapabilities"}, + {name: "admin_network_access_status_ok", operationID: "getAdminNetworkAccessStatus", scenario: "The provider's live state on the API host, with overlay origin and addresses.", method: http.MethodGet, path: Prefix + "/admin/network-access/stub/status", headers: actingRequestAdmin, status: 200, assertHeaders: []string{"Content-Type"}, schema: status}, + {name: "admin_network_access_status_unavailable_host", operationID: "getAdminNetworkAccessStatus", scenario: "A host whose provider plugin is not running answers state unavailable with the supervisor's reason and no updated_at.", method: http.MethodGet, path: Prefix + "/admin/network-access/down/status", headers: actingRequestAdmin, status: 200, assertHeaders: []string{"Content-Type"}, schema: status}, + {name: "admin_network_access_status_not_found", operationID: "getAdminNetworkAccessStatus", scenario: "A provider slug no enabled installation declares.", method: http.MethodGet, path: Prefix + "/admin/network-access/netbird/status", headers: actingRequestAdmin, status: 404, assertHeaders: []string{"Content-Type"}, schema: problem}, + {name: "admin_network_access_connect_accepted", operationID: "connectNetworkAccess", scenario: "Connect on the named host is acknowledged with the state reached so far; enrollment continues and the auth URL is admin-only.", method: http.MethodPost, path: Prefix + "/admin/network-access/stub/connect", headers: actingRequestAdmin, body: `{"hosts":["api"]}`, status: 202, assertHeaders: []string{"Content-Type"}, schema: status}, + {name: "admin_network_access_disconnect_accepted", operationID: "disconnectNetworkAccess", scenario: "Disconnect without a body acts on every host and is acknowledged with the state reached.", method: http.MethodPost, path: Prefix + "/admin/network-access/stub/disconnect", headers: actingRequestAdmin, status: 202, assertHeaders: []string{"Content-Type"}, schema: status}, + {name: "admin_network_access_unknown_host", operationID: "connectNetworkAccess", scenario: "A host id this deployment does not run the provider on is a validation failure at body.hosts; nothing is applied.", method: http.MethodPost, path: Prefix + "/admin/network-access/stub/connect", headers: actingRequestAdmin, body: `{"hosts":["node:99"]}`, status: 422, assertHeaders: []string{"Content-Type"}, schema: problem}, + } +} diff --git a/internal/apiv2/network_access_test.go b/internal/apiv2/network_access_test.go new file mode 100644 index 0000000000..bcb97ba150 --- /dev/null +++ b/internal/apiv2/network_access_test.go @@ -0,0 +1,400 @@ +package apiv2 + +import ( + "context" + "encoding/json" + "errors" + "net/http" + "net/http/httptest" + "strings" + "testing" + "time" + + "github.com/Silo-Server/silo-server/internal/netaccess" + "github.com/Silo-Server/silo-server/internal/plugins" +) + +// fakeNetworkAccess stands in for *plugins.Service: one provider ("stub") on +// the api host whose plugin answers, and an optional second provider whose +// plugin is not running. +type fakeNetworkAccess struct { + providers []plugins.NetworkAccessProvider + unavailable map[string]bool + listErr error + connects []string + disconnects []string + hosts [][]string + state string +} + +func newFakeNetworkAccess() *fakeNetworkAccess { + return &fakeNetworkAccess{ + providers: []plugins.NetworkAccessProvider{ + {InstallationID: 7, CapabilityID: "stub", Provider: "stub", DisplayName: "Stub Overlay"}, + {InstallationID: 9, CapabilityID: "down", Provider: "down", DisplayName: "Down Overlay"}, + }, + unavailable: map[string]bool{"down": true}, + state: netaccess.StateDisconnected, + } +} + +func (f *fakeNetworkAccess) ListNetworkAccessProviders(context.Context) ([]plugins.NetworkAccessProvider, error) { + if f.listErr != nil { + return nil, f.listErr + } + return f.providers, nil +} + +func (f *fakeNetworkAccess) report(slug, state string) (plugins.NetworkAccessReport, error) { + for _, provider := range f.providers { + if provider.Provider != slug { + continue + } + host := plugins.NetworkAccessHost{ID: plugins.HostScopeAPI, Role: "api", Name: "Living Room"} + status := netaccess.Status{InstallationID: provider.InstallationID, Provider: slug} + if f.unavailable[slug] { + status.State = netaccess.StateUnavailable + status.Error = "plugin process is backoff: plugin process exited" + } else { + status.State = state + status.Hostname = "silo.overlay.example.test" + status.Origin = "https://silo.overlay.example.test" + status.Addresses = []string{"127.0.0.1"} + status.ProviderVersion = "stub 0.1.0" + status.UpdatedAt = fixedTime() + if state == netaccess.StateAwaitingAuthorization { + status.AuthURL = "https://login.example.test/a/abc" + } + } + return plugins.NetworkAccessReport{Provider: provider, Hosts: []plugins.NetworkAccessHostStatus{{Host: host, Status: status}}}, nil + } + return plugins.NetworkAccessReport{}, plugins.ErrNetworkAccessProviderNotFound +} + +func (f *fakeNetworkAccess) NetworkAccessStatus(_ context.Context, provider string) (plugins.NetworkAccessReport, error) { + return f.report(provider, f.state) +} + +func (f *fakeNetworkAccess) checkHosts(hosts []string) error { + f.hosts = append(f.hosts, hosts) + for _, id := range hosts { + if id != plugins.HostScopeAPI { + return plugins.ErrNetworkAccessHostUnknown + } + } + return nil +} + +func (f *fakeNetworkAccess) ConnectNetworkAccess(_ context.Context, provider string, hosts []string) (plugins.NetworkAccessReport, error) { + if _, err := f.report(provider, ""); err != nil { + return plugins.NetworkAccessReport{}, err + } + if err := f.checkHosts(hosts); err != nil { + return plugins.NetworkAccessReport{}, err + } + f.connects = append(f.connects, provider) + return f.report(provider, netaccess.StateAwaitingAuthorization) +} + +func (f *fakeNetworkAccess) DisconnectNetworkAccess(_ context.Context, provider string, hosts []string) (plugins.NetworkAccessReport, error) { + if _, err := f.report(provider, ""); err != nil { + return plugins.NetworkAccessReport{}, err + } + if err := f.checkHosts(hosts); err != nil { + return plugins.NetworkAccessReport{}, err + } + f.disconnects = append(f.disconnects, provider) + return f.report(provider, netaccess.StateDisconnected) +} + +type networkAccessStatusDoc struct { + Provider string `json:"provider"` + Hosts []struct { + Host struct { + ID string `json:"id"` + Role string `json:"role"` + Name string `json:"name"` + } `json:"host"` + State string `json:"state"` + Hostname string `json:"hostname"` + Origin string `json:"origin"` + Addresses []string `json:"addresses"` + AuthURL string `json:"auth_url"` + Error string `json:"error"` + ProviderVersion string `json:"provider_version"` + UpdatedAt string `json:"updated_at"` + } `json:"hosts"` +} + +func decodeNetworkAccessStatus(t *testing.T, body string) networkAccessStatusDoc { + t.Helper() + var doc networkAccessStatusDoc + if err := json.Unmarshal([]byte(body), &doc); err != nil { + t.Fatalf("decode %s: %v", body, err) + } + return doc +} + +func TestNetworkAccessStatusOfCoercesUnknownState(t *testing.T) { + base := plugins.NetworkAccessReport{Provider: plugins.NetworkAccessProvider{Provider: "stub"}, Hosts: []plugins.NetworkAccessHostStatus{{ + Host: plugins.NetworkAccessHost{ID: plugins.HostScopeAPI, Role: "api", Name: "API"}, + Status: netaccess.Status{State: "reconnecting"}, + }}} + got := networkAccessStatusOf(base) + if got.Hosts[0].State != "error" || got.Hosts[0].RawState != "reconnecting" { + t.Fatalf("unknown state = %+v", got.Hosts[0]) + } + base.Hosts[0].Status.State = netaccess.StateConnected + got = networkAccessStatusOf(base) + if got.Hosts[0].State != netaccess.StateConnected || got.Hosts[0].RawState != "" { + t.Fatalf("known state = %+v", got.Hosts[0]) + } +} + +func TestNetworkAccessCapabilities(t *testing.T) { + deps := pilotDeps(nil, nil) + f := newFakeNetworkAccess() + deps.NetworkAccess = f + h := newTestHandler(t, deps) + path := Prefix + "/network-access/capabilities" + + requireProblem(t, do(t, h, http.MethodGet, path, "", nil), TypeAuthenticationRequired) + + rec := do(t, h, http.MethodGet, path, "", bearer(memberToken)) + if rec.Code != http.StatusOK { + t.Fatal(rec.Code, rec.Body.String()) + } + var doc struct { + State string `json:"state"` + Allowed bool `json:"allowed"` + Providers []struct { + Provider string `json:"provider"` + DisplayName string `json:"display_name"` + InstallationID string `json:"installation_id"` + } `json:"providers"` + } + if err := json.Unmarshal(rec.Body.Bytes(), &doc); err != nil { + t.Fatal(err) + } + if doc.State != StateAvailable || !doc.Allowed || len(doc.Providers) != 2 || doc.Providers[0].Provider != "stub" || doc.Providers[0].DisplayName != "Stub Overlay" || doc.Providers[0].InstallationID != "7" { + t.Fatalf("capabilities = %s", rec.Body.String()) + } + if rec.Header().Get("ETag") == "" { + t.Fatal("capability document has no ETag") + } + + // No provider installed: not_configured, providers is an empty array. + f.providers = nil + rec = do(t, h, http.MethodGet, path, "", bearer(memberToken)) + if rec.Code != http.StatusOK || !strings.Contains(rec.Body.String(), `"state":"not_configured"`) || !strings.Contains(rec.Body.String(), `"providers":[]`) { + t.Fatal(rec.Code, rec.Body.String()) + } + + f.listErr = errors.New("db down") + requireProblem(t, do(t, h, http.MethodGet, path, "", bearer(memberToken)), TypeInternalError) + + // Not wired at all (worker modes): still 200 not_configured. + deps.NetworkAccess = nil + rec = do(t, newTestHandler(t, deps), http.MethodGet, path, "", bearer(memberToken)) + if rec.Code != http.StatusOK || !strings.Contains(rec.Body.String(), `"state":"not_configured"`) { + t.Fatal(rec.Code, rec.Body.String()) + } +} + +func TestAdminNetworkAccessStatus(t *testing.T) { + deps := pilotDeps(nil, nil) + f := newFakeNetworkAccess() + deps.NetworkAccess = f + h := newTestHandler(t, deps) + path := Prefix + "/admin/network-access/stub/status" + + requireProblem(t, do(t, h, http.MethodGet, path, "", bearer(memberToken)), TypePermissionDenied) + + rec := do(t, h, http.MethodGet, path, "", bearer(adminToken)) + if rec.Code != http.StatusOK { + t.Fatal(rec.Code, rec.Body.String()) + } + doc := decodeNetworkAccessStatus(t, rec.Body.String()) + if doc.Provider != "stub" || len(doc.Hosts) != 1 { + t.Fatalf("status = %+v", doc) + } + host := doc.Hosts[0] + if host.Host.ID != "api" || host.Host.Role != "api" || host.Host.Name != "Living Room" { + t.Fatalf("host = %+v", host.Host) + } + if host.State != "disconnected" || host.Hostname != "silo.overlay.example.test" || host.Origin != "https://silo.overlay.example.test" || len(host.Addresses) != 1 || host.ProviderVersion != "stub 0.1.0" || host.AuthURL != "" { + t.Fatalf("host status = %+v", host) + } + if host.UpdatedAt != NewInstant(fixedTime()).String() { + t.Fatalf("updated_at = %q", host.UpdatedAt) + } + + // A host whose plugin is not running: unavailable with the reason, no + // updated_at, addresses still an array. + rec = do(t, h, http.MethodGet, Prefix+"/admin/network-access/down/status", "", bearer(adminToken)) + if rec.Code != http.StatusOK { + t.Fatal(rec.Code, rec.Body.String()) + } + doc = decodeNetworkAccessStatus(t, rec.Body.String()) + if len(doc.Hosts) != 1 || doc.Hosts[0].State != "unavailable" || !strings.Contains(doc.Hosts[0].Error, "backoff") || doc.Hosts[0].UpdatedAt != "" || !strings.Contains(rec.Body.String(), `"addresses":[]`) { + t.Fatalf("unavailable host = %s", rec.Body.String()) + } + + // Unknown provider is 404; a slug the pattern refuses is 422. + requireProblem(t, do(t, h, http.MethodGet, Prefix+"/admin/network-access/netbird/status", "", bearer(adminToken)), TypeNotFound) + requireProblem(t, do(t, h, http.MethodGet, Prefix+"/admin/network-access/Not%20A%20Slug/status", "", bearer(adminToken)), TypeValidationFailed) + + deps.NetworkAccess = nil + requireProblem(t, do(t, newTestHandler(t, deps), http.MethodGet, path, "", bearer(adminToken)), TypeDependencyUnavailable) +} + +func TestAdminNetworkAccessConnectDisconnect(t *testing.T) { + deps := pilotDeps(nil, nil) + f := newFakeNetworkAccess() + deps.NetworkAccess = f + h := newTestHandler(t, deps) + connect := Prefix + "/admin/network-access/stub/connect" + disconnect := Prefix + "/admin/network-access/stub/disconnect" + + requireProblem(t, do(t, h, http.MethodPost, connect, "", bearer(memberToken)), TypePermissionDenied) + if len(f.connects) != 0 { + t.Fatal("unauthorized connect reached the service") + } + + // No body: every host. 202 with the state reached and the auth URL. + rec := do(t, h, http.MethodPost, connect, "", bearer(adminToken)) + if rec.Code != http.StatusAccepted { + t.Fatal(rec.Code, rec.Body.String()) + } + doc := decodeNetworkAccessStatus(t, rec.Body.String()) + if len(f.connects) != 1 || f.hosts[0] != nil || len(doc.Hosts) != 1 || doc.Hosts[0].State != "awaiting_authorization" || doc.Hosts[0].AuthURL != "https://login.example.test/a/abc" { + t.Fatalf("connect: hosts=%v body=%s", f.hosts, rec.Body.String()) + } + + // Named host. + rec = do(t, h, http.MethodPost, connect, `{"hosts":["api"]}`, bearer(adminToken)) + if rec.Code != http.StatusAccepted || len(f.connects) != 2 || len(f.hosts[1]) != 1 || f.hosts[1][0] != "api" { + t.Fatal(rec.Code, rec.Body.String(), f.hosts) + } + // Empty body object behaves like no body. + if rec = do(t, h, http.MethodPost, connect, `{}`, bearer(adminToken)); rec.Code != http.StatusAccepted || f.hosts[2] != nil { + t.Fatal(rec.Code, rec.Body.String(), f.hosts) + } + + // An unknown host and an empty host list are validation failures that + // name body.hosts. + p := requireProblem(t, do(t, h, http.MethodPost, connect, `{"hosts":["node:99"]}`, bearer(adminToken)), TypeValidationFailed) + if len(p.Errors) != 1 || p.Errors[0].Location != "body.hosts" { + t.Fatalf("unknown host problem = %+v", p) + } + p = requireProblem(t, do(t, h, http.MethodPost, connect, `{"hosts":[]}`, bearer(adminToken)), TypeValidationFailed) + if len(p.Errors) != 1 || p.Errors[0].Location != "body.hosts" { + t.Fatalf("empty hosts problem = %+v", p) + } + requireProblem(t, do(t, h, http.MethodPost, connect, `{"hosts":`, bearer(adminToken)), TypeMalformedRequest) + + // Disconnect mirrors connect. + rec = do(t, h, http.MethodPost, disconnect, "", bearer(adminToken)) + if rec.Code != http.StatusAccepted || len(f.disconnects) != 1 { + t.Fatal(rec.Code, rec.Body.String()) + } + if doc = decodeNetworkAccessStatus(t, rec.Body.String()); doc.Hosts[0].State != "disconnected" || doc.Hosts[0].AuthURL != "" { + t.Fatalf("disconnect = %s", rec.Body.String()) + } + + // A host whose plugin is not running still answers 202: the per-host + // row says unavailable, the command was acknowledged, not applied. + rec = do(t, h, http.MethodPost, Prefix+"/admin/network-access/down/connect", "", bearer(adminToken)) + if rec.Code != http.StatusAccepted || !strings.Contains(rec.Body.String(), `"state":"unavailable"`) { + t.Fatal(rec.Code, rec.Body.String()) + } + + requireProblem(t, do(t, h, http.MethodPost, Prefix+"/admin/network-access/netbird/connect", "", bearer(adminToken)), TypeNotFound) + requireProblem(t, do(t, h, http.MethodPost, Prefix+"/admin/network-access/netbird/disconnect", "", bearer(adminToken)), TypeNotFound) + + deps.NetworkAccess = nil + requireProblem(t, do(t, newTestHandler(t, deps), http.MethodPost, connect, "", bearer(adminToken)), TypeDependencyUnavailable) +} + +// TestNetworkAccessStatusOfDefaultsState: a status the service could not fill +// in (zero value) is rendered unavailable rather than an empty enum value. +func TestNetworkAccessStatusOfDefaultsState(t *testing.T) { + report := plugins.NetworkAccessReport{ + Provider: plugins.NetworkAccessProvider{Provider: "stub"}, + Hosts: []plugins.NetworkAccessHostStatus{{Host: plugins.NetworkAccessHost{ID: "api", Role: "api"}}}, + } + out := networkAccessStatusOf(report) + if out.Provider != "stub" || len(out.Hosts) != 1 || out.Hosts[0].State != netaccess.StateUnavailable || out.Hosts[0].UpdatedAt != nil || out.Hosts[0].Addresses == nil { + t.Fatalf("status = %+v", out) + } + report.Hosts[0].Status = netaccess.Status{State: netaccess.StateConnected, UpdatedAt: time.Date(2026, 3, 4, 5, 6, 7, 0, time.UTC)} + if out = networkAccessStatusOf(report); out.Hosts[0].UpdatedAt == nil || out.Hosts[0].UpdatedAt.String() != "2026-03-04T05:06:07.000Z" { + t.Fatalf("updated_at = %v", out.Hosts[0].UpdatedAt) + } +} + +// An explicit null selector must not be read as "every host". +func TestNetworkAccessCommandRejectsExplicitNullHosts(t *testing.T) { + deps := pilotDeps(nil, nil) + f := newFakeNetworkAccess() + deps.NetworkAccess = f + h := newTestHandler(t, deps) + connect := Prefix + "/admin/network-access/stub/connect" + p := requireProblem(t, do(t, h, http.MethodPost, connect, `{"hosts":null}`, bearer(adminToken)), TypeValidationFailed) + if len(p.Errors) != 1 || p.Errors[0].Location != "body.hosts" { + t.Fatalf("problem = %+v", p) + } + if len(f.connects) != 0 { + t.Fatal("null hosts reached the service") + } +} + +// A literal null body is malformed, not "omitted": it must not fan out to +// every host. +func TestNetworkAccessCommandRejectsNullBody(t *testing.T) { + deps := pilotDeps(nil, nil) + f := newFakeNetworkAccess() + deps.NetworkAccess = f + h := newTestHandler(t, deps) + connect := Prefix + "/admin/network-access/stub/connect" + p := requireProblem(t, do(t, h, http.MethodPost, connect, `null`, bearer(adminToken)), TypeValidationFailed) + if len(p.Errors) != 1 || p.Errors[0].Location != "body" { + t.Fatalf("problem = %+v", p) + } + if len(f.connects) != 0 { + t.Fatal("null body reached the service") + } + // An omitted body still means every host. + if rec := do(t, h, http.MethodPost, connect, "", bearer(adminToken)); rec.Code != http.StatusAccepted || len(f.connects) != 1 { + t.Fatalf("omitted body: %d, connects=%d", rec.Code, len(f.connects)) + } +} + +func TestNetworkAccessCommandChunkedBody(t *testing.T) { + for _, action := range []string{"connect", "disconnect"} { + for _, body := range []string{"", "null", "{}", `{"hosts":null}`, `{"hosts":["api"]}`} { + t.Run(action+"/"+body, func(t *testing.T) { + deps := pilotDeps(nil, nil) + f := newFakeNetworkAccess() + deps.NetworkAccess = f + h := newTestHandler(t, deps) + req := httptest.NewRequest(http.MethodPost, Prefix+"/admin/network-access/stub/"+action, strings.NewReader(body)) + req.ContentLength = -1 + req.TransferEncoding = []string{"chunked"} + req.Header.Set("Content-Type", "application/json") + req.Header.Set("Authorization", "Bearer "+adminToken) + rec := httptest.NewRecorder() + h.ServeHTTP(rec, req) + invalid := body == "null" || body == `{"hosts":null}` + if invalid { + requireProblem(t, rec, TypeValidationFailed) + if len(f.connects)+len(f.disconnects) != 0 { + t.Fatal("invalid body reached the service") + } + } else if rec.Code != http.StatusAccepted || len(f.connects)+len(f.disconnects) != 1 { + t.Fatalf("status=%d, calls=%d: %s", rec.Code, len(f.connects)+len(f.disconnects), rec.Body.String()) + } + }) + } + } +} diff --git a/internal/apiv2/router.go b/internal/apiv2/router.go index dc8233230f..097a097b86 100644 --- a/internal/apiv2/router.go +++ b/internal/apiv2/router.go @@ -134,6 +134,7 @@ type Dependencies struct { AdminLogsSocket AdminLogsSocketService PlaybackControlSocket PlaybackControlSocketService EventsCapability EventsCapabilityService + NetworkAccess NetworkAccessService NotificationDestinationCreate NotificationDestinationCreateService AdminUnmatchedFiles AdminUnmatchedFilesService AdminCatalogImages AdminCatalogImagesService diff --git a/internal/apiv2/worker_protocols.go b/internal/apiv2/worker_protocols.go index c01c57ff54..34702bff62 100644 --- a/internal/apiv2/worker_protocols.go +++ b/internal/apiv2/worker_protocols.go @@ -46,5 +46,6 @@ func describeWorkerProtocols() workerProtocolRegistry { operations = append(operations, proxy.ProtocolDirectPlayback()...) operations = append(operations, proxy.ProtocolStreaming()...) operations = append(operations, transcodenode.ProtocolStreaming()...) + operations = append(operations, proxy.ProtocolNetworkAccess(schemas)...) return workerProtocolRegistry{Operations: operations, Schemas: schemas.Map()} } diff --git a/internal/apiv2/worker_streaming_test.go b/internal/apiv2/worker_streaming_test.go index ad8fa5428d..fbcf94987b 100644 --- a/internal/apiv2/worker_streaming_test.go +++ b/internal/apiv2/worker_streaming_test.go @@ -36,7 +36,7 @@ func TestWorkerDeliveryInventoryDescriptionCoverage(t *testing.T) { t.Fatal("undescribed worker registration", route.Listener, route.Method, route.Path) } } - if count != 40 || len(described) != 40 { + if count != 44 || len(described) != 44 { t.Fatal("retained inventory drift", count, len(described)) } } diff --git a/internal/cache/redis.go b/internal/cache/redis.go index 241609cffd..7cf3a37578 100644 --- a/internal/cache/redis.go +++ b/internal/cache/redis.go @@ -53,6 +53,11 @@ const ( EventOperationalLogAppended = "operational_log_appended" EventAuditLogAppended = "audit_log_appended" EventEventsNotification = "events_notification" + // EventPluginsChanged is published on ChannelAdmin by the API server after + // every plugin lifecycle change (install, enable, disable, config save, + // auto-update, uninstall) so proxy nodes running resident plugins from the + // same installations reconcile at once instead of on their next poll. + EventPluginsChanged = "plugins_changed" ) // --------------------------------------------------------------------------- diff --git a/internal/cache/replicas.go b/internal/cache/replicas.go new file mode 100644 index 0000000000..235e61f2e3 --- /dev/null +++ b/internal/cache/replicas.go @@ -0,0 +1,100 @@ +package cache + +import ( + "context" + "fmt" + "time" + + "github.com/google/uuid" + "github.com/redis/go-redis/v9" +) + +// apiReplicaKeyPrefix keys one live API replica's presence marker. +const apiReplicaKeyPrefix = "silo:api-replica:" + +// APIReplicaTTL is how long a presence marker outlives its last refresh. It is +// three refresh intervals so one missed refresh does not drop a live replica. +const APIReplicaTTL = 90 * time.Second + +// apiReplicaRefreshInterval is how often a registered replica renews its marker. +const apiReplicaRefreshInterval = 30 * time.Second + +// APIReplicaPresence is a best-effort census of live API replicas. Network +// access providers keep their overlay node key under one shared "api" state +// scope, so two API replicas would present the same overlay identity from two +// machines; the resident supervisor warns when it sees more than one. Nothing +// routes on this count and a Redis failure only silences the warning. +type APIReplicaPresence struct { + client *redis.Client + id string + ttl time.Duration + refresh time.Duration +} + +// NewAPIReplicaPresence returns a unique presence for this process, even when +// several replicas use the same logical node id. A nil client or empty id yields +// a nil presence, on which every method is a no-op that reports one replica. +func NewAPIReplicaPresence(client *redis.Client, id string) *APIReplicaPresence { + if client == nil || id == "" { + return nil + } + return &APIReplicaPresence{client: client, id: id + ":" + uuid.NewString(), ttl: APIReplicaTTL, refresh: apiReplicaRefreshInterval} +} + +// Register writes this replica's marker and renews it until ctx ends. +func (p *APIReplicaPresence) Register(ctx context.Context) error { + if p == nil { + return nil + } + if err := p.touch(ctx); err != nil { + return err + } + go func() { + ticker := time.NewTicker(p.refresh) + defer ticker.Stop() + for { + select { + case <-ctx.Done(): + cleanup, cancel := context.WithTimeout(context.Background(), 2*time.Second) + _ = p.client.Del(cleanup, p.key()).Err() + cancel() + return + case <-ticker.C: + _ = p.touch(ctx) + } + } + }() + return nil +} + +func (p *APIReplicaPresence) key() string { return apiReplicaKeyPrefix + p.id } + +func (p *APIReplicaPresence) touch(ctx context.Context) error { + if err := p.client.Set(ctx, p.key(), time.Now().UTC().Format(time.RFC3339), p.ttl).Err(); err != nil { + return fmt.Errorf("register api replica presence: %w", err) + } + return nil +} + +// Count returns how many API replicas currently hold a marker, including this +// one. A nil presence reports one. +func (p *APIReplicaPresence) Count(ctx context.Context) (int, error) { + if p == nil { + return 1, nil + } + var ( + cursor uint64 + count int + ) + for { + keys, next, err := p.client.Scan(ctx, cursor, apiReplicaKeyPrefix+"*", 100).Result() + if err != nil { + return 0, fmt.Errorf("count api replicas: %w", err) + } + count += len(keys) + if next == 0 { + return count, nil + } + cursor = next + } +} diff --git a/internal/cache/replicas_test.go b/internal/cache/replicas_test.go new file mode 100644 index 0000000000..c3cec8c0cd --- /dev/null +++ b/internal/cache/replicas_test.go @@ -0,0 +1,94 @@ +package cache + +import ( + "context" + "fmt" + "os" + "testing" + "time" + + "github.com/redis/go-redis/v9" +) + +func TestAPIReplicaPresenceNilReportsOneReplica(t *testing.T) { + var presence *APIReplicaPresence + if err := presence.Register(context.Background()); err != nil { + t.Fatal(err) + } + if count, err := presence.Count(context.Background()); err != nil || count != 1 { + t.Fatalf("Count on nil presence = (%d, %v), want (1, nil)", count, err) + } + if NewAPIReplicaPresence(nil, "a") != nil || NewAPIReplicaPresence(redis.NewClient(&redis.Options{}), "") != nil { + t.Fatal("presence built without a client or an id") + } +} + +func TestAPIReplicaPresenceSeparatesIdenticalNodeNames(t *testing.T) { + client := redis.NewClient(&redis.Options{}) + t.Cleanup(func() { _ = client.Close() }) + first := NewAPIReplicaPresence(client, "shared-api-name") + second := NewAPIReplicaPresence(client, "shared-api-name") + if first.key() == second.key() { + t.Fatal("replicas with the same node name share a presence marker") + } +} + +// Two replicas registering see each other; one whose context ends drops its +// marker so the census shrinks again. +func TestAPIReplicaPresenceCountsLiveReplicas(t *testing.T) { + rawURL := os.Getenv("SILO_TEST_REDIS_URL") + if rawURL == "" { + t.Skip("SILO_TEST_REDIS_URL not set") + } + options, err := redis.ParseURL(rawURL) + if err != nil { + t.Fatalf("parse SILO_TEST_REDIS_URL: %v", err) + } + client := redis.NewClient(options) + t.Cleanup(func() { _ = client.Close() }) + ctx := context.Background() + + unique := fmt.Sprintf("%d", time.Now().UnixNano()) + baseline, err := (&APIReplicaPresence{client: client, id: "census-" + unique, ttl: time.Minute, refresh: time.Minute}).Count(ctx) + if err != nil { + t.Fatal(err) + } + + newPresence := func(id string) *APIReplicaPresence { + p := NewAPIReplicaPresence(client, id+"-"+unique) + p.ttl, p.refresh = time.Minute, 10*time.Millisecond + t.Cleanup(func() { _ = client.Del(context.Background(), p.key()).Err() }) + return p + } + first := newPresence("shared-api-name") + firstCtx, stopFirst := context.WithCancel(ctx) + defer stopFirst() + if err := first.Register(firstCtx); err != nil { + t.Fatal(err) + } + second := newPresence("shared-api-name") + secondCtx, stopSecond := context.WithCancel(ctx) + defer stopSecond() + if err := second.Register(secondCtx); err != nil { + t.Fatal(err) + } + if count, err := second.Count(ctx); err != nil || count != baseline+2 { + t.Fatalf("Count with two replicas = (%d, %v), want %d", count, err, baseline+2) + } + + stopFirst() + deadline := time.Now().Add(10 * time.Second) + for { + count, err := second.Count(ctx) + if err != nil { + t.Fatal(err) + } + if count == baseline+1 { + break + } + if time.Now().After(deadline) { + t.Fatalf("count after the first replica left = %d, want %d", count, baseline+1) + } + time.Sleep(10 * time.Millisecond) + } +} diff --git a/internal/contractledger/ledger_test.go b/internal/contractledger/ledger_test.go index e7b6cc875b..55bc6a6b2e 100644 --- a/internal/contractledger/ledger_test.go +++ b/internal/contractledger/ledger_test.go @@ -1214,6 +1214,9 @@ var mutationWithoutLegacyRow = map[string]string{ "uploadAdminCollectionPoster": "V2 separates administrator poster upload from legacy multipart definition create and update; those legacy operations retain their own mappings.", "uploadAdminCollectionBackdrop": "V2 separates administrator backdrop upload from legacy multipart definition create and update; those legacy operations retain their own mappings.", "uploadCollectionPoster": "V2 separates poster upload from legacy multipart collection create and update; their legacy rows remain mapped separately.", + "restartAdminPluginInstallation": "V2-only resident plugin restart: v1 had no supervised plugin processes, so no legacy route stops and relaunches one. Repeating the restart converges on one running process.", + "connectNetworkAccess": "V2-only network access provider command: v1 had no overlay network providers. Repeating the connect converges on one connected instance per host.", + "disconnectNetworkAccess": "V2-only network access provider command: v1 had no overlay network providers. Repeating the disconnect converges on disconnected.", } // retrySafetyMismatches compares every operation the v2 registry declares diff --git a/internal/jellycompat/handlers_playback.go b/internal/jellycompat/handlers_playback.go index 8625681a77..bb5441ad97 100644 --- a/internal/jellycompat/handlers_playback.go +++ b/internal/jellycompat/handlers_playback.go @@ -28,6 +28,7 @@ import ( "github.com/Silo-Server/silo-server/internal/config" "github.com/Silo-Server/silo-server/internal/logredact" "github.com/Silo-Server/silo-server/internal/models" + "github.com/Silo-Server/silo-server/internal/netaccess" "github.com/Silo-Server/silo-server/internal/nodepool" "github.com/Silo-Server/silo-server/internal/noderouting" "github.com/Silo-Server/silo-server/internal/playback" @@ -183,6 +184,7 @@ type nodeRoutingAssignmentSetter interface { // Child HLS requests trust the durable compat marker, so its write is fatal; // the native mirror remains best-effort as it was before route markers existed. func (h *PlaybackHandler) recordNodeRoutingAssignment(ctx context.Context, playSessionID, sessionID string, assignment playback.NodeRoutingAssignment) error { + assignment.NetworkProvider = new(netaccess.PathFromContext(ctx).Provider) if h.playbackStore != nil { if err := h.playbackStore.Update(playSessionID, func(current *PlaybackSession) error { committed := assignment @@ -626,7 +628,10 @@ func (h *PlaybackHandler) resolveCompatIdentityRouteWithPolicy( workload = noderouting.WorkloadRemux delivery = noderouting.DeliveryProgressiveRemux } - proxyEligible := h.compatProxyEligibility(ctx, requiresAudioBoost) + // Narrowed to proxies the client can reach on its access path: a client + // that arrived through a network access provider never receives a LAN + // origin, and with no reachable proxy the API-egress shapes apply. + proxyEligible := nodepool.ClientReachableVia(netaccess.PathFromContext(ctx), h.compatProxyEligibility(ctx, requiresAudioBoost)) decision, err := noderouting.Resolve(noderouting.AdaptSessionPlanner(h.NodePlanner), noderouting.ResolveRequest{ Request: noderouting.Request{ Workload: workload, Delivery: delivery, @@ -982,7 +987,11 @@ func (h *PlaybackHandler) resolveCompatHLSRouteOnNodeWithPolicy( }, SessionID: session.ID, CurrentTranscodeURL: currentTranscodeURL, EstimatedBitrateKbps: source.Version.Bitrate, - TranscodeEligible: eligible, ExcludedShapeIDs: excludedShapes, + TranscodeEligible: eligible, + // Proxy egress only through a proxy the client can reach on its access + // path; nil on the default path, so every healthy proxy stays eligible. + ProxyEligible: nodepool.ClientReachableVia(netaccess.PathFromContext(ctx), nil), + ExcludedShapeIDs: excludedShapes, }) } @@ -1203,7 +1212,10 @@ func (h *PlaybackHandler) CleanupOrphanedTranscodes() (int, error) { } // buildProxyRedirectURL signs a stream token and builds the redirect URL for -// the given proxy node (the planner's pick for this session). +// the given proxy node (the planner's pick for this session) on the client's +// access path. A proxy with no origin on that path is an error, which every +// caller treats like an unavailable proxy transport: the reservation is +// released and the stream is served from this server. func (h *PlaybackHandler) buildProxyRedirectURL( playSessionID string, upstreamSessionID string, @@ -1215,10 +1227,15 @@ func (h *PlaybackHandler) buildProxyRedirectURL( transcodeNodeURL string, seekSeconds float64, proxyNode *nodepool.Node, + path netaccess.Path, ) (string, error) { if proxyNode == nil || h.JWTSecret == "" { return "", fmt.Errorf("proxy transport unavailable") } + base := proxyNode.ClientURLFor(path) + if base == "" { + return "", fmt.Errorf("proxy %d has no client origin on access path %q", proxyNode.ID, path.Provider) + } audioTrackIndex := 0 if resolvedAudioTrackIndex, ok := compatAudioTrackIndex(source); ok { @@ -1236,31 +1253,34 @@ func (h *PlaybackHandler) buildProxyRedirectURL( targetAudioCodec = compatCopyCodec } claims := streamtoken.Claims{ - SessionID: upstreamSessionID, - MediaPath: file.FilePath, - PlayMethod: method, - TranscodeAudio: source.TranscodeAudio, - TargetCodecAudio: targetAudioCodec, - AudioTrackIndex: audioTrackIndex, - SourceAudioChannels: sourceAudioChannels, - AudioOnly: file.IsAudioOnly(), - TranscodeNode: transcodeNodeURL, - DVProfile: file.PrimaryDVProfile(), - RoutingWorkload: string(noderouting.WorkloadDirectPlay), - RoutingExecution: string(noderouting.ExecutionNone), - RoutingEgress: string(noderouting.EgressProxy), - RoutingEgressNodeID: proxyNode.ID, + SessionID: upstreamSessionID, + MediaPath: file.FilePath, + PlayMethod: method, + TranscodeAudio: source.TranscodeAudio, + TargetCodecAudio: targetAudioCodec, + AudioTrackIndex: audioTrackIndex, + SourceAudioChannels: sourceAudioChannels, + AudioOnly: file.IsAudioOnly(), + TranscodeNode: transcodeNodeURL, + DVProfile: file.PrimaryDVProfile(), + RoutingWorkload: string(noderouting.WorkloadDirectPlay), + RoutingExecution: string(noderouting.ExecutionNone), + RoutingEgress: string(noderouting.EgressProxy), + RoutingEgressNodeID: proxyNode.ID, + RoutingNetworkProvider: new(path.Provider), } switch method { case string(playback.PlayRemux): claims.RoutingWorkload = string(noderouting.WorkloadRemux) claims.RoutingExecution = string(noderouting.ExecutionProxy) + claims.RoutingExecutionNodeID = proxyNode.ID case string(playback.PlayTranscode): claims.RoutingWorkload = string(noderouting.WorkloadVideoTranscode) if compatHLSCopiesVideo(source) { claims.RoutingWorkload = string(noderouting.WorkloadRemux) } claims.RoutingExecution = string(noderouting.ExecutionTranscode) + claims.RoutingExecutionNodeID = h.compatTranscodeNodeID(transcodeNodeURL, nil) } if playback.IsAudioToAACStereoDownmixV3(claims.SourceAudioChannels, claims.TargetCodecAudio, claims.TargetAudioChannels) { // Compatibility AAC output is stereo by default. Freeze that effective @@ -1304,19 +1324,19 @@ func (h *PlaybackHandler) buildProxyRedirectURL( switch method { case string(playback.PlayDirect): - return nodepool.NodeEndpoint(proxyNode.ClientURL(), "/stream/direct/"+token), nil + return nodepool.NodeEndpoint(base, "/stream/direct/"+token), nil case string(playback.PlayRemux): remuxPath := "/stream/remux/" if claims.PlayMethod == streamtoken.PlayMethodAudioDownmixRemux { remuxPath = "/stream/remux/audio-v2/" } - redirectURL := nodepool.NodeEndpoint(proxyNode.ClientURL(), remuxPath+token) + redirectURL := nodepool.NodeEndpoint(base, remuxPath+token) if seekSeconds > 0 { redirectURL += "?seek=" + strconv.FormatFloat(seekSeconds, 'f', -1, 64) } return redirectURL, nil case string(playback.PlayTranscode): - return nodepool.NodeEndpoint(proxyNode.ClientURL(), + return nodepool.NodeEndpoint(base, "/stream/transcode/"+token+"/master.m3u8?"+playback.SourceTimelineQueryParam+"=1"), nil default: return "", fmt.Errorf("unsupported proxy method %q", method) @@ -1810,6 +1830,8 @@ func (h *PlaybackHandler) persistTranscodeRecipe( if playSession != nil { card.OriginalStartedAt = playSession.CreatedAt if assignment := playSession.RoutingAssignment; assignment != nil { + card.RoutingNetworkProvider = assignment.NetworkProvider + card.RoutingExecutionNodeID = assignment.ExecutionNodeID card.RoutingWorkload = assignment.Workload card.RoutingExecution = assignment.Execution card.RoutingEgress = assignment.Egress diff --git a/internal/jellycompat/network_access_test.go b/internal/jellycompat/network_access_test.go new file mode 100644 index 0000000000..cd7c161b35 --- /dev/null +++ b/internal/jellycompat/network_access_test.go @@ -0,0 +1,137 @@ +package jellycompat + +import ( + "context" + "net/url" + "strings" + "testing" + "time" + + "github.com/Silo-Server/silo-server/internal/config" + "github.com/Silo-Server/silo-server/internal/models" + "github.com/Silo-Server/silo-server/internal/netaccess" + "github.com/Silo-Server/silo-server/internal/nodepool" + "github.com/Silo-Server/silo-server/internal/noderouting" + "github.com/Silo-Server/silo-server/internal/playback" + "github.com/Silo-Server/silo-server/internal/streamtoken" +) + +const ( + compatLANProxyURL = "http://10.0.0.9:8083" + compatTailnetOrigin = "https://proxy-1.tail1234.ts.net" +) + +func compatTailnetContext() context.Context { + return netaccess.WithPath(context.Background(), netaccess.Path{Provider: "tailscale"}) +} + +// The Jellyfin-compatible redirect is built on the same accessor as the +// native API: a provider-path client is never redirected to a LAN origin. +func TestBuildProxyRedirectURLUsesTheAccessPathOrigin(t *testing.T) { + h := &PlaybackHandler{JWTSecret: "test-secret"} + file := &models.MediaFile{FilePath: "/media/movie.mkv"} + tailnet := netaccess.Path{Provider: "tailscale"} + lanOnly := &nodepool.Node{ID: 41, URL: compatLANProxyURL} + connected := &nodepool.Node{ID: 41, URL: compatLANProxyURL, NetworkAccess: netaccess.NodeNetworkAccess{ + "tailscale": {State: netaccess.StateConnected, Origin: compatTailnetOrigin}, + }} + + for _, method := range []string{string(playback.PlayDirect), string(playback.PlayRemux), string(playback.PlayTranscode)} { + t.Run(method, func(t *testing.T) { + if got, err := h.buildProxyRedirectURL("play", "upstream", method, file, PlaybackMediaSource{}, nil, time.Time{}, "http://transcode-1", 0, lanOnly, tailnet); err == nil { + t.Fatalf("LAN-only proxy produced %q for a tailnet client, want an error", got) + } + got, err := h.buildProxyRedirectURL("play", "upstream", method, file, PlaybackMediaSource{}, nil, time.Time{}, "http://transcode-1", 0, connected, tailnet) + if err != nil { + t.Fatal(err) + } + if !strings.HasPrefix(got, compatTailnetOrigin+"/stream/") || strings.Contains(got, "10.0.0.9") { + t.Fatalf("redirect = %q, want the tailnet origin and no LAN address", got) + } + + parsed, err := url.Parse(got) + if err != nil { + t.Fatal(err) + } + parts := strings.Split(strings.Trim(parsed.Path, "/"), "/") + claims, err := streamtoken.Verify(parts[2], h.JWTSecret) + if err != nil { + t.Fatal(err) + } + if claims.RoutingNetworkProvider == nil || *claims.RoutingNetworkProvider != "tailscale" { + t.Fatalf("provider = %v", claims.RoutingNetworkProvider) + } + lan, err := h.buildProxyRedirectURL("play", "upstream", method, file, PlaybackMediaSource{}, nil, time.Time{}, "http://transcode-1", 0, connected, netaccess.Path{}) + if err != nil { + t.Fatal(err) + } + if !strings.HasPrefix(lan, compatLANProxyURL+"/stream/") { + t.Fatalf("default-path redirect = %q, want the LAN origin", lan) + } + }) + } +} + +// Route resolution excludes proxies the client cannot reach before reserving, +// so a tailnet client with only LAN proxies is routed to API egress under the +// default policy, refused under proxy_only, and given the proxy once it has a +// tailnet origin. +func TestResolveCompatIdentityRouteExcludesProxiesUnreachableOnTheAccessPath(t *testing.T) { + proxies := nodepool.NewProxyPool() + proxies.SetNodes([]*nodepool.Node{{ID: 41, URL: compatLANProxyURL, Enabled: true, Healthy: true}}) + handler := &PlaybackHandler{JWTSecret: "secret", NodePlanner: nodepool.NewPlanner(proxies, nodepool.NewTranscodePool())} + policy := config.DefaultPlaybackRoutingPolicy() + + decision := handler.resolveCompatIdentityRouteWithPolicy(compatTailnetContext(), "compat-tailnet-direct", string(playback.PlayDirect), 8_000, false, policy) + if !decision.Selected() || decision.Shape.Egress != noderouting.EgressAPI || decision.Plan.ProxyNode != nil { + t.Fatalf("tailnet decision = %#v, want API egress with no proxy", decision) + } + lan := handler.resolveCompatIdentityRouteWithPolicy(context.Background(), "compat-lan-direct", string(playback.PlayDirect), 8_000, false, policy) + if !lan.Selected() || lan.Plan.ProxyNode == nil || lan.Plan.ProxyNode.ID != 41 { + t.Fatalf("LAN decision = %#v, want the LAN proxy", lan) + } + + hard := policy + hard.DirectPlayEgress = config.PlaybackEgressProxyOnly + refused := handler.resolveCompatIdentityRouteWithPolicy(compatTailnetContext(), "compat-tailnet-proxy-only", string(playback.PlayDirect), 8_000, false, hard) + if refused.Selected() || refused.Outcome != noderouting.OutcomeCapacityUnavailable { + t.Fatalf("proxy_only tailnet decision = %#v, want %s", refused, noderouting.OutcomeCapacityUnavailable) + } + + proxies.SetNodes([]*nodepool.Node{{ID: 41, URL: compatLANProxyURL, Enabled: true, Healthy: true, NetworkAccess: netaccess.NodeNetworkAccess{ + "tailscale": {State: netaccess.StateConnected, Origin: compatTailnetOrigin}, + }}}) + reachable := handler.resolveCompatIdentityRouteWithPolicy(compatTailnetContext(), "compat-tailnet-reachable", string(playback.PlayDirect), 8_000, false, hard) + if !reachable.Selected() || reachable.Plan.ProxyNode == nil || reachable.Plan.ProxyNode.ClientURLFor(netaccess.Path{Provider: "tailscale"}) != compatTailnetOrigin { + t.Fatalf("reachable decision = %#v, want the tailnet-connected proxy", reachable) + } +} + +func TestCompatNetworkRouteIsMirroredAndRecoverable(t *testing.T) { + for _, provider := range []string{"", "tailscale"} { + manager := playback.NewSessionManager(0, 0) + session, err := manager.StartSession(7, "profile", 42, playback.PlayDirect, false) + if err != nil { + t.Fatal(err) + } + store := NewPlaybackSessionStore(0, nil) + store.Put(PlaybackSession{ID: "play", UpstreamSessionID: session.ID}) + handler := &PlaybackHandler{sessionMgr: manager, playbackStore: store} + ctx := netaccess.WithPath(t.Context(), netaccess.Path{Provider: provider}) + if err := handler.recordNodeRoutingAssignment(ctx, "play", session.ID, playback.NodeRoutingAssignment{Workload: "remux", Execution: "transcode", ExecutionNodeID: 7, Egress: "proxy", EgressNodeID: 11}); err != nil { + t.Fatal(err) + } + current, err := manager.GetSession(session.ID) + if err != nil { + t.Fatal(err) + } + if current.RoutingNetworkProvider == nil || *current.RoutingNetworkProvider != provider { + t.Fatalf("native mirror = %v", current.RoutingNetworkProvider) + } + stored, _ := store.Get("play") + card := handler.upstreamRecipeCard(stored, &Session{StreamAppUserID: 7, ProfileID: "profile"}, PlaybackMediaSource{FileID: 42}, "remux") + if card.RoutingNetworkProvider == nil || *card.RoutingNetworkProvider != provider || card.RoutingExecutionNodeID != 7 || card.RoutingEgressNodeID != 11 { + t.Fatalf("recovery route = %#v", card) + } + } +} diff --git a/internal/jellycompat/router.go b/internal/jellycompat/router.go index d9c9df917d..b36d9f8d88 100644 --- a/internal/jellycompat/router.go +++ b/internal/jellycompat/router.go @@ -17,6 +17,7 @@ import ( "github.com/Silo-Server/silo-server/internal/clientip" "github.com/Silo-Server/silo-server/internal/config" "github.com/Silo-Server/silo-server/internal/httpstream" + "github.com/Silo-Server/silo-server/internal/netaccess" "github.com/Silo-Server/silo-server/internal/playback" "github.com/Silo-Server/silo-server/internal/recommendations" "github.com/Silo-Server/silo-server/internal/sections" @@ -34,6 +35,9 @@ func NewRouter(deps Dependencies) chi.Router { if deps.ClientIPResolver != nil { r.Use(clientip.Middleware(deps.ClientIPResolver)) } + if deps.IngressTokens != nil { + r.Use(netaccess.Middleware(deps.IngressTokens)) + } r.Use(cors.Handler(cors.Options{ AllowedOrigins: []string{"*"}, AllowedMethods: []string{"GET", "POST", "PUT", "DELETE", "OPTIONS", "HEAD"}, diff --git a/internal/jellycompat/routing_policy_test.go b/internal/jellycompat/routing_policy_test.go index 34c89807e5..0ed5228dbe 100644 --- a/internal/jellycompat/routing_policy_test.go +++ b/internal/jellycompat/routing_policy_test.go @@ -4,6 +4,7 @@ import ( "context" "net/http" "net/http/httptest" + "reflect" "strings" "testing" "time" @@ -249,7 +250,8 @@ func TestRecordNodeRoutingAssignmentPersistsChildRouteBinding(t *testing.T) { store.Put(PlaybackSession{ID: "play-1", CompatToken: "compat-token"}) handler := &PlaybackHandler{playbackStore: store} want := playback.NodeRoutingAssignment{ - Workload: string(noderouting.WorkloadVideoTranscode), Execution: string(noderouting.ExecutionTranscode), + NetworkProvider: new(""), + Workload: string(noderouting.WorkloadVideoTranscode), Execution: string(noderouting.ExecutionTranscode), ExecutionNodeID: 2, ExecutionNodeURL: "http://worker-1", Egress: string(noderouting.EgressAPI), } @@ -258,7 +260,7 @@ func TestRecordNodeRoutingAssignmentPersistsChildRouteBinding(t *testing.T) { t.Fatalf("record route: %v", err) } stored, ok := store.Get("play-1") - if !ok || stored.RoutingAssignment == nil || *stored.RoutingAssignment != want { + if !ok || stored.RoutingAssignment == nil || !reflect.DeepEqual(*stored.RoutingAssignment, want) { t.Fatalf("stored route = %#v, want %#v", stored.RoutingAssignment, want) } } diff --git a/internal/jellycompat/server.go b/internal/jellycompat/server.go index 7a3dfe8251..0da09e4891 100644 --- a/internal/jellycompat/server.go +++ b/internal/jellycompat/server.go @@ -13,6 +13,7 @@ import ( "github.com/Silo-Server/silo-server/internal/catalog" "github.com/Silo-Server/silo-server/internal/clientip" "github.com/Silo-Server/silo-server/internal/config" + "github.com/Silo-Server/silo-server/internal/netaccess" "github.com/Silo-Server/silo-server/internal/nodepool" "github.com/Silo-Server/silo-server/internal/recommendations" "github.com/Silo-Server/silo-server/internal/scantrigger" @@ -42,6 +43,9 @@ type Dependencies struct { DB *pgxpool.Pool SecretCipher *secret.Cipher // at-rest credential cipher (required when DB is set) ClientIPResolver *clientip.Resolver + // IngressTokens validates the X-Silo-Ingress-Token network access + // provider plugins stamp on proxied requests. Nil accepts no tokens. + IngressTokens *netaccess.Registry // StreamTelemetry is the local observation-only registry shared with the // native API process. May be nil, which makes every media route unobserved. StreamTelemetry *streamtelemetry.Registry diff --git a/internal/jellycompat/streams.go b/internal/jellycompat/streams.go index c0b6780d4c..14ade23039 100644 --- a/internal/jellycompat/streams.go +++ b/internal/jellycompat/streams.go @@ -24,6 +24,7 @@ import ( "github.com/Silo-Server/silo-server/internal/config" "github.com/Silo-Server/silo-server/internal/httpstream" "github.com/Silo-Server/silo-server/internal/models" + "github.com/Silo-Server/silo-server/internal/netaccess" "github.com/Silo-Server/silo-server/internal/nodepool" "github.com/Silo-Server/silo-server/internal/noderouting" "github.com/Silo-Server/silo-server/internal/playback" @@ -488,7 +489,7 @@ func (h *PlaybackHandler) HandleVideoStream(w http.ResponseWriter, r *http.Reque return } if proxyNode := decision.Plan.ProxyNode; proxyNode != nil { - if redirectURL, redirectErr := h.buildProxyRedirectURL(playSession.ID, playSession.UpstreamSessionID, method, file, *source, session, playSession.CreatedAt, "", seekSeconds, proxyNode); redirectErr == nil { + if redirectURL, redirectErr := h.buildProxyRedirectURL(playSession.ID, playSession.UpstreamSessionID, method, file, *source, session, playSession.CreatedAt, "", seekSeconds, proxyNode, netaccess.PathFromContext(r.Context())); redirectErr == nil { assignment := playback.NodeRoutingAssignment{ Workload: string(decision.Shape.Workload), Execution: string(decision.Shape.Execution), Egress: string(decision.Shape.Egress), EgressNodeID: proxyNode.ID, EgressNodeURL: proxyNode.URL, @@ -792,7 +793,7 @@ func (h *PlaybackHandler) HandleMasterManifest(w http.ResponseWriter, r *http.Re executionNodeID := h.compatTranscodeNodeID(remoteNodeURL, tcNode) if decision.Shape.Egress == noderouting.EgressProxy { - redirectURL, redirectErr := h.buildProxyRedirectURL(playSession.ID, playSession.UpstreamSessionID, string(playback.PlayTranscode), file, *source, session, playSession.CreatedAt, remoteNodeURL, 0, plan.ProxyNode) + redirectURL, redirectErr := h.buildProxyRedirectURL(playSession.ID, playSession.UpstreamSessionID, string(playback.PlayTranscode), file, *source, session, playSession.CreatedAt, remoteNodeURL, 0, plan.ProxyNode, netaccess.PathFromContext(r.Context())) if redirectErr == nil { if err := h.recordNodeRoutingAssignment(r.Context(), playSession.ID, playSession.UpstreamSessionID, playback.NodeRoutingAssignment{ Workload: string(decision.Shape.Workload), Execution: string(noderouting.ExecutionTranscode), @@ -1036,9 +1037,12 @@ func (h *PlaybackHandler) HandleHLSSegment(w http.ResponseWriter, r *http.Reques // Recover the playback session and local runtime as one transaction. If a // frozen tone-map recipe cannot be rebuilt, the manager rolls back the exact // provisional playback session before this handler returns an error. + // The master binds its route after persisting the executable recipe, so + // overlay that durable assignment before reconstructing from a child request. + card := h.upstreamRecipeCard(playSession, session, *source, playSession.UpstreamPlayMethod) _, transcodeSession, status, reconstructErr := h.tm.LoadOrReconstructTranscodeWithError( r.Context(), h.sessionMgr.GetSession, playSession.UpstreamSessionID, - session.StreamAppUserID, requestedSegment, playSession.Recipe, + session.StreamAppUserID, requestedSegment, &card, ) switch status { case playback.SessionMissing: @@ -2232,6 +2236,8 @@ func (h *PlaybackHandler) upstreamRecipeCard(ps *PlaybackSession, cs *Session, s card.OriginalStartedAt = ps.CreatedAt } if ps != nil && ps.RoutingAssignment != nil { + card.RoutingNetworkProvider = ps.RoutingAssignment.NetworkProvider + card.RoutingExecutionNodeID = ps.RoutingAssignment.ExecutionNodeID card.RoutingWorkload = ps.RoutingAssignment.Workload card.RoutingExecution = ps.RoutingAssignment.Execution card.RoutingEgress = ps.RoutingAssignment.Egress diff --git a/internal/jellycompat/streams_segment_error_test.go b/internal/jellycompat/streams_segment_error_test.go index da03f17769..dcc096a08d 100644 --- a/internal/jellycompat/streams_segment_error_test.go +++ b/internal/jellycompat/streams_segment_error_test.go @@ -16,10 +16,84 @@ import ( "github.com/Silo-Server/silo-server/internal/catalog" "github.com/Silo-Server/silo-server/internal/models" + "github.com/Silo-Server/silo-server/internal/netaccess" "github.com/Silo-Server/silo-server/internal/playback" "github.com/Silo-Server/silo-server/internal/tonemap" ) +func TestHandleHLSSegmentRecoversNetworkRouteRecordedAfterRecipe(t *testing.T) { + for _, provider := range []string{"", "tailscale"} { + t.Run("provider="+provider, func(t *testing.T) { + const upstreamID = "upstream-network" + source := PlaybackMediaSource{ID: "source-42", FileID: 42, Version: catalog.FileVersion{FileID: 42}} + sessions := playback.NewSessionManager(0, 0) + sessions.RegisterReconstructed(&playback.Session{ + ID: upstreamID, UserID: 7, ProfileID: "profile-1", MediaFileID: 42, PlayMethod: playback.PlayTranscode, + }) + store := NewPlaybackSessionStore(time.Hour, nil) + store.Put(PlaybackSession{ + ID: "play-network", CompatToken: "compat-token", RouteItemID: "item", UpstreamSessionID: upstreamID, + UpstreamPlayMethod: "transcode", MediaSources: []PlaybackMediaSource{source}, + }) + handler := &PlaybackHandler{playbackStore: store, sessionMgr: sessions} + // Startup persists the executable recipe before the master manifest binds + // its route. Recovery must use that later assignment, not the old card. + if err := handler.persistTranscodeRecipe(t.Context(), "play-network", upstreamID, playback.TranscodeOpts{ + SessionID: upstreamID, InputPath: "/media/movie.mkv", TargetCodecVideo: "h264", TargetCodecAudio: "aac", + AudioTrackIndex: compatAudioTrackIndexOrDefault(source), SegmentDuration: 2, + }); err != nil { + t.Fatal(err) + } + ctx := netaccess.WithPath(t.Context(), netaccess.Path{Provider: provider}) + if err := handler.recordNodeRoutingAssignment(ctx, "play-network", upstreamID, *compatLocalVideoAPIRoutingAssignment()); err != nil { + t.Fatal(err) + } + stored, _ := store.Get("play-network") + if stored.Recipe.RoutingNetworkProvider != nil { + t.Fatal("fixture must retain the recipe written before route assignment") + } + + // A restart leaves only the durable compat row and cached HLS bytes. + recovered := playback.NewSessionManager(0, 0) + tm := playback.NewTranscodeManager() + tm.Sessions = recovered + root := t.TempDir() + output := filepath.Join(root, upstreamID) + if err := os.MkdirAll(output, 0o755); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(filepath.Join(output, "seg_00000.ts"), []byte("cached HLS segment"), 0o644); err != nil { + t.Fatal(err) + } + ffmpegPath := writeCompatTestFFmpeg(t) + tm.Config = func() playback.TranscodeRuntimeConfig { + return playback.TranscodeRuntimeConfig{TranscodeDir: root, FFmpegPath: ffmpegPath, HWAccel: playback.HWAccelNone} + } + t.Cleanup(func() { tm.CloseTranscodeSession(upstreamID, "") }) + handler.sessionMgr, handler.tm = recovered, tm + req := httptest.NewRequest(http.MethodGet, "/Videos/item/hls/play-network/seg_00000.ts", nil) + routeCtx := chi.NewRouteContext() + routeCtx.URLParams.Add("playlistId", "play-network") + routeCtx.URLParams.Add("segmentId", "seg_00000") + routeCtx.URLParams.Add("segmentContainer", "ts") + requestCtx := context.WithValue(req.Context(), chi.RouteCtxKey, routeCtx) + requestCtx = context.WithValue(requestCtx, compatSessionKey, &Session{Token: "compat-token", StreamAppUserID: 7}) + recorder := httptest.NewRecorder() + handler.HandleHLSSegment(recorder, req.WithContext(requestCtx)) + if recorder.Code != http.StatusOK || recorder.Body.String() != "cached HLS segment" { + t.Fatalf("segment response = %d %s", recorder.Code, recorder.Body.String()) + } + current, err := recovered.GetSession(upstreamID) + if err != nil { + t.Fatal(err) + } + if current.RoutingNetworkProvider == nil || *current.RoutingNetworkProvider != provider || current.RoutingWorkload != "video_transcode" || current.RoutingExecution != "api" || current.RoutingEgress != "api" { + t.Fatalf("recovered route = %#v", current) + } + }) + } +} + // TestHLSSegmentErrorResponse pins the never-500 contract for the HLS segment // handler: a segment that will never materialize (absent, or whose transcode // process started then died) maps to 404 like Jellyfin, and only genuinely diff --git a/internal/jellycompat/streams_test.go b/internal/jellycompat/streams_test.go index c8b2e1a93a..6be0eb8f0d 100644 --- a/internal/jellycompat/streams_test.go +++ b/internal/jellycompat/streams_test.go @@ -20,6 +20,7 @@ import ( "github.com/Silo-Server/silo-server/internal/config" "github.com/Silo-Server/silo-server/internal/mediaprobe" "github.com/Silo-Server/silo-server/internal/models" + "github.com/Silo-Server/silo-server/internal/netaccess" "github.com/Silo-Server/silo-server/internal/nodepool" "github.com/Silo-Server/silo-server/internal/noderouting" "github.com/Silo-Server/silo-server/internal/playback" @@ -560,6 +561,7 @@ func TestBuildProxyRedirectURLRequestsSourceAlignedCompatManifest(t *testing.T) "http://transcode-1", 0, &nodepool.Node{URL: "http://proxy-1"}, + netaccess.Path{}, ) if err != nil { t.Fatalf("buildProxyRedirectURL: %v", err) @@ -575,7 +577,7 @@ func TestBuildProxyRedirectURLMarksCopyFMP4ForOldReaderRejection(t *testing.T) { redirectURL, err := h.buildProxyRedirectURL( "play-1", "upstream-1", string(playback.PlayTranscode), &models.MediaFile{FilePath: "/media/movie.mkv"}, source, nil, time.Time{}, - "http://transcode-1", 0, &nodepool.Node{URL: "http://proxy-1"}, + "http://transcode-1", 0, &nodepool.Node{URL: "http://proxy-1"}, netaccess.Path{}, ) if err != nil { t.Fatal(err) @@ -599,7 +601,7 @@ func TestBuildProxyRedirectURLCarriesMPEGTSForRemoteCopyRecipe(t *testing.T) { redirectURL, err := h.buildProxyRedirectURL( "play-1", "upstream-1", string(playback.PlayTranscode), &models.MediaFile{FilePath: "/media/movie.mkv"}, source, nil, time.Time{}, - "http://transcode-1", 0, &nodepool.Node{URL: "http://proxy-1"}, + "http://transcode-1", 0, &nodepool.Node{URL: "http://proxy-1"}, netaccess.Path{}, ) if err != nil { t.Fatal(err) @@ -616,7 +618,7 @@ func TestBuildProxyRedirectURLCarriesMPEGTSForRemoteCopyRecipe(t *testing.T) { func TestBuildProxyRedirectURLMarksToneMapForOldReaderRejection(t *testing.T) { h := &PlaybackHandler{JWTSecret: "test-secret"} source := PlaybackMediaSource{Version: catalog.FileVersion{HDR: true, VideoTracks: []models.VideoTrack{{VideoRangeType: "HDR10", ColorTransfer: "smpte2084"}}}} - redirectURL, err := h.buildProxyRedirectURL("play-1", "upstream-1", string(playback.PlayTranscode), &models.MediaFile{FilePath: "/media/hdr.mkv"}, source, nil, time.Time{}, "http://transcode-1", 0, &nodepool.Node{URL: "http://proxy-1"}) + redirectURL, err := h.buildProxyRedirectURL("play-1", "upstream-1", string(playback.PlayTranscode), &models.MediaFile{FilePath: "/media/hdr.mkv"}, source, nil, time.Time{}, "http://transcode-1", 0, &nodepool.Node{URL: "http://proxy-1"}, netaccess.Path{}) if err != nil { t.Fatal(err) } @@ -653,6 +655,7 @@ func TestBuildProxyRedirectURLCarriesAudioOnlyRemuxClaim(t *testing.T) { "", 0, &nodepool.Node{URL: "http://proxy-1"}, + netaccess.Path{}, ) if err != nil { t.Fatalf("buildProxyRedirectURL: %v", err) @@ -894,11 +897,11 @@ func TestProxyRedirectURLClaimGrowthBudget(t *testing.T) { for _, method := range []string{string(playback.PlayDirect), string(playback.PlayRemux), string(playback.PlayTranscode)} { t.Run(method, func(t *testing.T) { - withClaims, err := h.buildProxyRedirectURL("play", "upstream", method, file, source, session, createdAt, transcodeNodeURL, 12.5, proxyNode) + withClaims, err := h.buildProxyRedirectURL("play", "upstream", method, file, source, session, createdAt, transcodeNodeURL, 12.5, proxyNode, netaccess.Path{}) if err != nil { t.Fatal(err) } - withoutClaims, err := h.buildProxyRedirectURL("play", "upstream", method, file, source, nil, time.Time{}, transcodeNodeURL, 12.5, proxyNode) + withoutClaims, err := h.buildProxyRedirectURL("play", "upstream", method, file, source, nil, time.Time{}, transcodeNodeURL, 12.5, proxyNode, netaccess.Path{}) if err != nil { t.Fatal(err) } diff --git a/internal/netaccess/broker.go b/internal/netaccess/broker.go new file mode 100644 index 0000000000..42426dbe00 --- /dev/null +++ b/internal/netaccess/broker.go @@ -0,0 +1,91 @@ +package netaccess + +import ( + "crypto/subtle" + "sync" +) + +// Broker ties the token registry and the status cache to resident plugin +// process lifetimes: a start issues a fresh ingress token, a stop revokes it +// and forgets the status so no stale overlay origin survives the process. +// The resident supervisor in internal/plugins drives it. +type Broker struct { + Registry *Registry + Status *StatusCache + + // mu serializes Issue and Revoke. Registry.Revoke and Status.Forget are + // two operations; without this a replacement's Issue (and the status its + // process then pushes) could land between them and be forgotten by the + // old process's revoke. A process cannot report before it was issued a + // token, so ordering Issue after Revoke completes is enough. + mu sync.Mutex +} + +// NewBroker returns a broker over a fresh registry and status cache. +func NewBroker() *Broker { + return &Broker{Registry: NewRegistry(), Status: NewStatusCache()} +} + +// Issue mints the installation's ingress token for a starting process. +func (b *Broker) Issue(installationID int, provider string) (string, error) { + if b == nil { + return "", nil + } + b.mu.Lock() + defer b.mu.Unlock() + token, err := b.Registry.Issue(installationID, provider) + if err == nil { + b.Status.Forget(installationID) + } + return token, err +} + +// Revoke forgets the installation's token and last reported status, provided +// token is still the current one; a stale token (the process was already +// replaced) changes nothing. +func (b *Broker) Revoke(installationID int, token string) { + if b == nil { + return + } + b.mu.Lock() + defer b.mu.Unlock() + if b.Registry.Revoke(installationID, token) { + b.Status.Forget(installationID) + } +} + +// IngressToken returns the installation's current token. +func (b *Broker) IngressToken(installationID int) (string, bool) { + if b == nil { + return "", false + } + return b.Registry.IngressToken(installationID) +} + +// Report records a provider status push from this process's own reads and +// commands (see plugins.Service.applyNetworkAccess), which run against the +// current instance by construction. +func (b *Broker) Report(status Status) (Status, bool) { + if b == nil { + return Status{}, false + } + return b.Status.Report(status) +} + +// ReportFor records a status push from the plugin process holding token. +// The registry check and the cache write happen under the same lock Revoke +// takes, so a push that was in flight while its process was revoked lands +// after the revoke and is refused rather than resurrecting the old origin. +func (b *Broker) ReportFor(installationID int, token string, status Status) (previous Status, changed bool, accepted bool) { + if b == nil { + return Status{}, false, false + } + b.mu.Lock() + defer b.mu.Unlock() + current, ok := b.Registry.IngressToken(installationID) + if !ok || subtle.ConstantTimeCompare([]byte(current), []byte(token)) != 1 { + return Status{}, false, false + } + previous, changed = b.Status.Report(status) + return previous, changed, true +} diff --git a/internal/netaccess/broker_test.go b/internal/netaccess/broker_test.go new file mode 100644 index 0000000000..6a36fb70e3 --- /dev/null +++ b/internal/netaccess/broker_test.go @@ -0,0 +1,38 @@ +package netaccess + +import "testing" + +func TestBrokerRejectsReportsFromRevokedAndReplacedProcesses(t *testing.T) { + b := NewBroker() + old, err := b.Issue(7, "stub") + if err != nil { + t.Fatal(err) + } + stale := Status{InstallationID: 7, Provider: "stub", State: StateConnected, Origin: "https://old.example.test"} + b.ReportFor(7, old, stale) + b.Revoke(7, old) + b.ReportFor(7, old, stale) + if _, ok := b.Status.Get(7); ok { + t.Fatal("late report restored a revoked process") + } + fresh, err := b.Issue(7, "stub") + if err != nil { + t.Fatal(err) + } + b.ReportFor(7, old, stale) + if _, ok := b.Status.Get(7); ok { + t.Fatal("old process reported under its replacement's token") + } + current := Status{InstallationID: 7, Provider: "stub", State: StateDisconnected} + b.ReportFor(7, fresh, current) + b.ReportFor(7, old, stale) + if got, ok := b.Status.Get(7); !ok || got.State != StateDisconnected { + t.Fatalf("late report overwrote replacement: %+v %v", got, ok) + } + if _, err := b.Issue(7, "stub"); err != nil { + t.Fatal(err) + } + if _, ok := b.Status.Get(7); ok { + t.Fatal("replacement inherited the previous process status") + } +} diff --git a/internal/netaccess/middleware.go b/internal/netaccess/middleware.go new file mode 100644 index 0000000000..7c3f312192 --- /dev/null +++ b/internal/netaccess/middleware.go @@ -0,0 +1,36 @@ +package netaccess + +import "net/http" + +// IngressTokenHeader carries the per-process ingress token a network access +// provider stamps on every request it proxies to a host listener. +const IngressTokenHeader = "X-Silo-Ingress-Token" + +// Middleware validates and strips the ingress token header on every request. +// A valid token records the provider's access path on the request context; +// an unknown token is rejected with 403 before any handler runs; a request +// without the header stays on the default path. The header is removed in all +// cases so it never reaches handlers, logs, or upstream proxies. A nil +// registry accepts no tokens. +func Middleware(registry *Registry) func(http.Handler) http.Handler { + return func(next http.Handler) http.Handler { + return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + values := r.Header.Values(IngressTokenHeader) + if len(values) == 0 { + next.ServeHTTP(w, r) + return + } + r.Header.Del(IngressTokenHeader) + if len(values) != 1 { + http.Error(w, "invalid ingress token", http.StatusForbidden) + return + } + ingress, ok := registry.Lookup(values[0]) + if !ok { + http.Error(w, "invalid ingress token", http.StatusForbidden) + return + } + next.ServeHTTP(w, r.WithContext(WithPath(r.Context(), Path{Provider: ingress.Provider}))) + }) + } +} diff --git a/internal/netaccess/netaccess_test.go b/internal/netaccess/netaccess_test.go new file mode 100644 index 0000000000..69be79e26d --- /dev/null +++ b/internal/netaccess/netaccess_test.go @@ -0,0 +1,267 @@ +package netaccess + +import ( + "net/http" + "net/http/httptest" + "testing" +) + +func TestMiddlewareValidTokenSetsPathAndStripsHeader(t *testing.T) { + registry := NewRegistry() + token, err := registry.Issue(7, "tailscale") + if err != nil { + t.Fatal(err) + } + var ( + gotPath Path + gotHeader []string + ) + handler := Middleware(registry)(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + gotPath = PathFromContext(r.Context()) + gotHeader = r.Header.Values(IngressTokenHeader) + w.WriteHeader(http.StatusNoContent) + })) + req := httptest.NewRequest(http.MethodGet, "/api/v2/system", nil) + req.Header.Set(IngressTokenHeader, token) + rec := httptest.NewRecorder() + handler.ServeHTTP(rec, req) + if rec.Code != http.StatusNoContent { + t.Fatalf("status = %d, want 204", rec.Code) + } + if gotPath.Provider != "tailscale" || gotPath.IsDefault() { + t.Fatalf("path = %+v, want provider tailscale", gotPath) + } + if len(gotHeader) != 0 { + t.Fatalf("ingress header reached the handler: %v", gotHeader) + } +} + +func TestMiddlewareUnknownTokenIsForbidden(t *testing.T) { + registry := NewRegistry() + if _, err := registry.Issue(7, "tailscale"); err != nil { + t.Fatal(err) + } + called := false + handler := Middleware(registry)(http.HandlerFunc(func(http.ResponseWriter, *http.Request) { called = true })) + for name, token := range map[string]string{ + "unknown": "not-a-token", + "empty": "", + } { + t.Run(name, func(t *testing.T) { + req := httptest.NewRequest(http.MethodGet, "/", nil) + req.Header.Set(IngressTokenHeader, token) + rec := httptest.NewRecorder() + handler.ServeHTTP(rec, req) + if rec.Code != http.StatusForbidden { + t.Fatalf("status = %d, want 403", rec.Code) + } + if called { + t.Fatal("handler ran for a rejected token") + } + }) + } + t.Run("duplicate header", func(t *testing.T) { + token, _ := registry.Issue(7, "tailscale") + req := httptest.NewRequest(http.MethodGet, "/", nil) + req.Header.Add(IngressTokenHeader, token) + req.Header.Add(IngressTokenHeader, token) + rec := httptest.NewRecorder() + handler.ServeHTTP(rec, req) + if rec.Code != http.StatusForbidden || called { + t.Fatalf("duplicate header: status=%d called=%v", rec.Code, called) + } + }) + t.Run("revoked", func(t *testing.T) { + token, _ := registry.Issue(7, "tailscale") + if !registry.Revoke(7, token) { + t.Fatal("current token not revoked") + } + req := httptest.NewRequest(http.MethodGet, "/", nil) + req.Header.Set(IngressTokenHeader, token) + rec := httptest.NewRecorder() + handler.ServeHTTP(rec, req) + if rec.Code != http.StatusForbidden || called { + t.Fatalf("revoked token: status=%d called=%v", rec.Code, called) + } + }) +} + +func TestMiddlewareMissingHeaderKeepsDefaultPath(t *testing.T) { + for name, registry := range map[string]*Registry{"registry": NewRegistry(), "nil registry": nil} { + t.Run(name, func(t *testing.T) { + var gotPath Path + handler := Middleware(registry)(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + gotPath = PathFromContext(r.Context()) + w.WriteHeader(http.StatusOK) + })) + rec := httptest.NewRecorder() + handler.ServeHTTP(rec, httptest.NewRequest(http.MethodGet, "/", nil)) + if rec.Code != http.StatusOK { + t.Fatalf("status = %d, want 200", rec.Code) + } + if !gotPath.IsDefault() { + t.Fatalf("path = %+v, want default", gotPath) + } + }) + } +} + +func TestRegistryIssueRotatesAndRejectsInvalidInput(t *testing.T) { + registry := NewRegistry() + first, err := registry.Issue(3, "tailscale") + if err != nil { + t.Fatal(err) + } + second, err := registry.Issue(3, "tailscale") + if err != nil { + t.Fatal(err) + } + if first == second { + t.Fatal("reissued token did not rotate") + } + if _, ok := registry.Lookup(first); ok { + t.Fatal("stale token still resolves") + } + if registry.Revoke(3, first) { + t.Fatal("stale token revoked the current one") + } + if _, ok := registry.Lookup(second); !ok { + t.Fatal("current token lost to a stale revoke") + } + if ingress, ok := registry.Lookup(second); !ok || ingress.InstallationID != 3 || ingress.Provider != "tailscale" { + t.Fatalf("lookup = %+v, %v", ingress, ok) + } + if got, ok := registry.IngressToken(3); !ok || got != second { + t.Fatalf("IngressToken = %q, %v", got, ok) + } + if _, err := registry.Issue(-1, "tailscale"); err == nil { + t.Fatal("negative installation id accepted") + } + if _, err := registry.Issue(4, ""); err == nil { + t.Fatal("empty provider accepted") + } +} + +func TestStatusCacheConnectedOrigins(t *testing.T) { + cache := NewStatusCache() + if _, changed := cache.Report(Status{InstallationID: 1, Provider: "tailscale", State: StateConnecting}); !changed { + t.Fatal("first report not a transition") + } + if got := cache.ConnectedOrigins(); len(got) != 0 { + t.Fatalf("connecting provider leaked origins: %v", got) + } + previous, changed := cache.Report(Status{ + InstallationID: 1, Provider: "tailscale", State: StateConnected, + Origin: "https://Silo.tail1234.ts.net/", + Listeners: []Listener{{Name: "api", Origin: "https://silo.tail1234.ts.net"}, {Name: "jellyfin", Origin: "https://silo.tail1234.ts.net:8096"}}, + }) + if !changed || previous.State != StateConnecting { + t.Fatalf("transition: previous=%+v changed=%v", previous, changed) + } + cache.Report(Status{InstallationID: 2, Provider: "netbird", State: StateError, Origin: "https://broken.example"}) + got := cache.ConnectedOrigins() + want := []string{"https://silo.tail1234.ts.net", "https://silo.tail1234.ts.net:8096"} + if len(got) != len(want) { + t.Fatalf("origins = %v, want %v", got, want) + } + for i := range want { + if got[i] != want[i] { + t.Fatalf("origins = %v, want %v", got, want) + } + } + cache.Forget(1) + if got := cache.ConnectedOrigins(); len(got) != 0 { + t.Fatalf("forgotten provider still listed: %v", got) + } +} + +func TestNodeNetworkAccessConnectedOrigin(t *testing.T) { + report := NodeNetworkAccess{ + "tailscale": {State: StateConnected, Origin: "https://Proxy-1.tail1234.ts.net/"}, + "netbird": {State: StateConnecting, Origin: "https://proxy-1.netbird.example"}, + "broken": {State: StateConnected, Origin: "not a url"}, + } + if got, ok := report.ConnectedOrigin("tailscale"); !ok || got != "https://proxy-1.tail1234.ts.net" { + t.Fatalf("tailscale origin = %q, %v", got, ok) + } + for _, provider := range []string{"netbird", "broken", "missing", ""} { + if got, ok := report.ConnectedOrigin(provider); ok || got != "" { + t.Fatalf("%q origin = %q, %v; want none", provider, got, ok) + } + } + if got, ok := NodeNetworkAccess(nil).ConnectedOrigin("tailscale"); ok || got != "" { + t.Fatalf("nil report origin = %q, %v", got, ok) + } +} + +func TestNodeNetworkAccessNormalized(t *testing.T) { + if got := (NodeNetworkAccess{" ": {State: StateConnected}}).Normalized(); got != nil { + t.Fatalf("blank-only report normalized to %v, want nil", got) + } + got := NodeNetworkAccess{" tailscale ": {State: " connected ", Origin: " https://a.example ", Hostname: " a "}}.Normalized() + want := NodeNetworkAccess{"tailscale": {State: StateConnected, Origin: "https://a.example", Hostname: "a"}} + if len(got) != 1 || got["tailscale"] != want["tailscale"] { + t.Fatalf("normalized = %#v, want %#v", got, want) + } +} + +func TestStatusCacheNodeNetworkAccess(t *testing.T) { + cache := NewStatusCache() + if got := cache.NodeNetworkAccess(); got != nil { + t.Fatalf("empty cache reported %v", got) + } + cache.Report(Status{InstallationID: 1, Provider: "tailscale", State: StateConnected, Origin: "https://p.tail.ts.net", Hostname: "p.tail.ts.net", AuthURL: "https://login.example/secret"}) + cache.Report(Status{InstallationID: 2, Provider: "netbird", State: StateAwaitingAuthorization}) + got := cache.NodeNetworkAccess() + if len(got) != 2 { + t.Fatalf("report = %#v, want two providers", got) + } + ts := got["tailscale"] + if ts.State != StateConnected || ts.Origin != "https://p.tail.ts.net" || ts.Hostname != "p.tail.ts.net" || ts.UpdatedAt.IsZero() { + t.Fatalf("tailscale entry = %#v", ts) + } + if got["netbird"].State != StateAwaitingAuthorization || got["netbird"].Origin != "" { + t.Fatalf("netbird entry = %#v", got["netbird"]) + } + if origin, ok := got.ConnectedOrigin("netbird"); ok || origin != "" { + t.Fatal("an unauthorized provider yielded an origin") + } +} + +func TestBrokerRevokeOfOldTokenKeepsReplacementStatus(t *testing.T) { + b := NewBroker() + old, err := b.Issue(7, "stub") + if err != nil { + t.Fatal(err) + } + b.ReportFor(7, old, Status{InstallationID: 7, Provider: "stub", State: StateConnected, Origin: "https://old.example.test"}) + // The replacement process is issued its token and pushes its status + // before the old process's revoke runs. + fresh, err := b.Issue(7, "stub") + if err != nil { + t.Fatal(err) + } + b.ReportFor(7, fresh, Status{InstallationID: 7, Provider: "stub", State: StateConnected, Origin: "https://new.example.test"}) + b.Revoke(7, old) + if got, ok := b.Status.Get(7); !ok || got.Origin != "https://new.example.test" { + t.Fatalf("stale revoke cleared the replacement's status: %+v %v", got, ok) + } + if tok, ok := b.IngressToken(7); !ok || tok != fresh { + t.Fatalf("stale revoke removed the replacement's token") + } + b.Revoke(7, fresh) + if _, ok := b.Status.Get(7); ok { + t.Fatal("current revoke left the status behind") + } +} + +func TestStatusCacheNodeNetworkAccessLowestInstallationOwnsDuplicateSlug(t *testing.T) { + c := NewStatusCache() + c.Report(Status{InstallationID: 9, Provider: "tailscale", State: StateConnected, Origin: "https://nine.example.test"}) + c.Report(Status{InstallationID: 4, Provider: "tailscale", State: StateConnected, Origin: "https://four.example.test"}) + for i := 0; i < 20; i++ { + if got := c.NodeNetworkAccess()["tailscale"].Origin; got != "https://four.example.test" { + t.Fatalf("iteration %d: origin = %q, want the lowest installation's", i, got) + } + } +} diff --git a/internal/netaccess/node.go b/internal/netaccess/node.go new file mode 100644 index 0000000000..692664084b --- /dev/null +++ b/internal/netaccess/node.go @@ -0,0 +1,126 @@ +package netaccess + +import ( + "maps" + "strings" + "time" +) + +// NodeProviderStatus is what a node reports about one network access provider +// running beside it: the subset of Status a remote reader needs to decide +// whether clients on that provider's overlay can reach the node, and where. +// It travels in the node's /health response and is stored verbatim on the +// node's row, so it carries only stable, non-secret fields — never the auth +// URL or the error text. +type NodeProviderStatus struct { + State string `json:"state"` + Origin string `json:"origin,omitempty"` + Hostname string `json:"hostname,omitempty"` + // UpdatedAt is when the node last heard from the provider, on the node's + // clock. Display data; nothing routes on it. + UpdatedAt time.Time `json:"updated_at,omitzero"` +} + +// NodeNetworkAccess is a node's last provider report keyed by provider slug. +// A nil or empty map means the node reports no providers. +type NodeNetworkAccess map[string]NodeProviderStatus + +// ConnectedOrigin returns the normalized scheme://host[:port] origin clients on +// the provider's overlay should use to reach the node, and whether there is +// one: the provider must be connected and must have reported a well-formed +// http(s) origin. Anything else is "not reachable this way", which callers +// turn into their API-relative fallback rather than into a guess. +func (m NodeNetworkAccess) ConnectedOrigin(provider string) (string, bool) { + if len(m) == 0 || provider == "" { + return "", false + } + status, ok := m[provider] + if !ok || !status.Connected() { + return "", false + } + return NormalizeOrigin(status.Origin) +} + +// Connected reports whether the provider currently serves overlay traffic. +func (s NodeProviderStatus) Connected() bool { return s.State == StateConnected } + +// Normalized returns a copy with provider keys and string fields trimmed and +// entries with an empty provider dropped, so a report is stored and compared +// in one canonical form regardless of how the node spelled it. nil in, nil +// out; an empty result is also nil so "reports none" has one representation. +func (m NodeNetworkAccess) Normalized() NodeNetworkAccess { + if len(m) == 0 { + return nil + } + out := make(NodeNetworkAccess, len(m)) + for provider, status := range m { + provider = strings.TrimSpace(provider) + if provider == "" { + continue + } + status.State = strings.TrimSpace(status.State) + status.Origin = strings.TrimSpace(status.Origin) + status.Hostname = strings.TrimSpace(status.Hostname) + out[provider] = status + } + if len(out) == 0 { + return nil + } + return out +} + +// Clone returns an independent copy, nil for nil. +func (m NodeNetworkAccess) Clone() NodeNetworkAccess { + if m == nil { + return nil + } + return maps.Clone(m) +} + +// NodeNetworkAccess renders the cache as the per-provider report a node puts +// on its /health response: one entry per provider instance running in this +// process, keyed by provider slug. This is the shape the API server stores on +// the node's row and hands stream URL selection. +func (c *StatusCache) NodeNetworkAccess() NodeNetworkAccess { + if c == nil { + return nil + } + c.mu.RLock() + defer c.mu.RUnlock() + if len(c.entries) == 0 { + return nil + } + out := make(NodeNetworkAccess, len(c.entries)) + owner := make(map[string]int, len(c.entries)) + for _, status := range c.entries { + if status.Provider == "" { + continue + } + // Two installations declaring one slug: the lowest installation id + // owns it, matching plugins.ListNetworkAccessProviders, instead of + // whichever map iteration happened to visit last. + if id, dup := owner[status.Provider]; dup && id < status.InstallationID { + continue + } + owner[status.Provider] = status.InstallationID + out[status.Provider] = NodeProviderStatus{ + State: status.State, + Origin: status.Origin, + Hostname: status.Hostname, + UpdatedAt: status.UpdatedAt, + } + } + if len(out) == 0 { + return nil + } + return out +} + +// HostStatusReport is what a proxy node answers the API's bearer-authenticated +// GET /network-access/status with: the full status (auth URL and error text +// included, since the caller holds the node bearer) of every provider +// instance running beside it. A provider whose process is not running is +// listed as StateUnavailable with the reason in Error. +type HostStatusReport struct { + Providers []Status `json:"providers"` +} diff --git a/internal/netaccess/path.go b/internal/netaccess/path.go new file mode 100644 index 0000000000..3b7e45f211 --- /dev/null +++ b/internal/netaccess/path.go @@ -0,0 +1,37 @@ +// Package netaccess owns the access path a request arrived on: the default +// path (LAN, public URL, reverse proxy) or an overlay network fronted by a +// network access provider plugin. The plugin stamps every request it proxies +// with a per-process ingress token; the middleware here validates the token, +// strips the header and records the path on the request context so later +// stages (stream URL selection, WebSocket origin checks) can act on it. +package netaccess + +import "context" + +// Path is the access path of one request. The zero value is the default +// path; Provider names the network access provider (e.g. "tailscale") whose +// listener the request came through. +type Path struct { + Provider string +} + +// IsDefault reports whether the request did not arrive through a provider. +func (p Path) IsDefault() bool { return p.Provider == "" } + +type pathKey struct{} + +// WithPath returns ctx carrying the access path. +func WithPath(ctx context.Context, path Path) context.Context { + return context.WithValue(ctx, pathKey{}, path) +} + +// PathFromContext returns the access path recorded on ctx, or the default +// path when the request did not pass the ingress middleware or carried no +// token. +func PathFromContext(ctx context.Context) Path { + if ctx == nil { + return Path{} + } + path, _ := ctx.Value(pathKey{}).(Path) + return path +} diff --git a/internal/netaccess/registry.go b/internal/netaccess/registry.go new file mode 100644 index 0000000000..bcd9a1bcfc --- /dev/null +++ b/internal/netaccess/registry.go @@ -0,0 +1,131 @@ +package netaccess + +import ( + "crypto/rand" + "crypto/subtle" + "encoding/base64" + "fmt" + "sync" +) + +// IngressTokenBytes is the entropy of one ingress token. +const IngressTokenBytes = 32 + +// Ingress identifies the plugin instance behind a validated ingress token. +type Ingress struct { + InstallationID int + Provider string +} + +type registryEntry struct { + token []byte + provider string +} + +// Registry maps ingress tokens to the resident plugin installation that +// received them. The resident supervisor issues a fresh token every time it +// starts a network access provider and revokes it when the process stops, so +// a token outlives neither the plugin process nor the host process (the +// registry is in memory only). Lookups compare in constant time. +type Registry struct { + mu sync.RWMutex + entries map[int]registryEntry +} + +// NewRegistry returns an empty registry. +func NewRegistry() *Registry { + return &Registry{entries: make(map[int]registryEntry)} +} + +// Issue generates a new token for the installation, replacing any previous +// one, and returns its wire form. provider is the provider slug the +// installation's capability declares; it becomes Path.Provider for requests +// carrying the token. +func (r *Registry) Issue(installationID int, provider string) (string, error) { + if r == nil { + return "", fmt.Errorf("netaccess: registry is nil") + } + if installationID <= 0 { + return "", fmt.Errorf("netaccess: installation id %d is not a persisted installation", installationID) + } + if provider == "" { + return "", fmt.Errorf("netaccess: provider is required") + } + raw := make([]byte, IngressTokenBytes) + if _, err := rand.Read(raw); err != nil { + return "", fmt.Errorf("netaccess: generate ingress token: %w", err) + } + token := base64.RawURLEncoding.EncodeToString(raw) + r.mu.Lock() + r.entries[installationID] = registryEntry{token: []byte(token), provider: provider} + r.mu.Unlock() + return token, nil +} + +// Revoke forgets the installation's token if it is still the one given, and +// reports whether it did. A stop that races a replacing start therefore never +// removes the token the new process was just issued. Requests still carrying +// a revoked token are rejected until a new one is issued. +func (r *Registry) Revoke(installationID int, token string) bool { + if r == nil { + return false + } + r.mu.Lock() + defer r.mu.Unlock() + entry, ok := r.entries[installationID] + if !ok || subtle.ConstantTimeCompare([]byte(token), entry.token) != 1 { + return false + } + delete(r.entries, installationID) + return true +} + +// IngressToken returns the installation's current token for GetHostInfo. +func (r *Registry) IngressToken(installationID int) (string, bool) { + if r == nil { + return "", false + } + r.mu.RLock() + defer r.mu.RUnlock() + entry, ok := r.entries[installationID] + if !ok { + return "", false + } + return string(entry.token), true +} + +// Provider returns the provider slug registered for the installation. +func (r *Registry) Provider(installationID int) (string, bool) { + if r == nil { + return "", false + } + r.mu.RLock() + defer r.mu.RUnlock() + entry, ok := r.entries[installationID] + if !ok { + return "", false + } + return entry.provider, true +} + +// Lookup resolves a presented token. Every registered token is compared in +// constant time so the response time does not reveal which one was close. +func (r *Registry) Lookup(token string) (Ingress, bool) { + if r == nil || token == "" { + return Ingress{}, false + } + presented := []byte(token) + r.mu.RLock() + defer r.mu.RUnlock() + var ( + match Ingress + found int + ) + for installationID, entry := range r.entries { + if subtle.ConstantTimeCompare(presented, entry.token) == 1 { + match = Ingress{InstallationID: installationID, Provider: entry.provider} + found = 1 + } + } + return match, found == 1 +} diff --git a/internal/netaccess/status.go b/internal/netaccess/status.go new file mode 100644 index 0000000000..b823f0eabd --- /dev/null +++ b/internal/netaccess/status.go @@ -0,0 +1,197 @@ +package netaccess + +import ( + "errors" + "net/url" + "sort" + "strings" + "sync" + "time" +) + +// ErrProviderNotFound reports a provider slug no enabled installation on this +// host declares. The plugin service and the proxy's bearer routes share it so +// the API can tell "unknown provider" from a failed call. +var ErrProviderNotFound = errors.New("network access provider not found") + +// Provider states as reported by network_access_provider.v1 plugins. +const ( + StateDisconnected = "disconnected" + StateAwaitingAuthorization = "awaiting_authorization" + StateConnecting = "connecting" + StateConnected = "connected" + StateError = "error" + // StateUnavailable is host-side: the provider's plugin process is not + // running on that host, so no status can be read from it. A plugin never + // reports it. + StateUnavailable = "unavailable" +) + +// Listener is one host listener a provider exposes on the overlay. +type Listener struct { + Name string `json:"name"` + Origin string `json:"origin"` +} + +// Status is the last status a provider instance on this host reported +// through RuntimeHost.ReportNetworkAccessStatus. It mirrors the SDK's +// NetworkAccessStatus without depending on the generated types so the +// packages that consume it (API handlers, stream URL selection) stay free of +// the plugin SDK. The JSON form is what a proxy node answers the API's +// bearer-authenticated network-access routes with; it carries auth_url and +// error because those routes take the node bearer, unlike the public /health +// report (NodeProviderStatus). +type Status struct { + InstallationID int `json:"installation_id"` + Provider string `json:"provider"` + State string `json:"state"` + Hostname string `json:"hostname,omitempty"` + Origin string `json:"origin,omitempty"` + Addresses []string `json:"addresses,omitempty"` + Listeners []Listener `json:"listeners,omitempty"` + AuthURL string `json:"auth_url,omitempty"` + Error string `json:"error,omitempty"` + ProviderVersion string `json:"provider_version,omitempty"` + DesiredConnected bool `json:"desired_connected,omitempty"` + UpdatedAt time.Time `json:"updated_at,omitzero"` +} + +// Connected reports whether the provider currently serves overlay traffic. +func (s Status) Connected() bool { return s.State == StateConnected } + +// StatusCache holds the most recent status of every provider instance +// running in this process. Plugins push on every change; the cache is the +// host's view between pushes and answers the WebSocket origin check with the +// overlay origins currently connected. +type StatusCache struct { + mu sync.RWMutex + entries map[int]Status + now func() time.Time +} + +// NewStatusCache returns an empty cache. +func NewStatusCache() *StatusCache { + return &StatusCache{entries: make(map[int]Status), now: time.Now} +} + +// Report records a status push. It returns the previous status for the +// installation and whether the state field changed, so the caller can log +// transitions without the cache knowing about logging. +func (c *StatusCache) Report(status Status) (previous Status, changed bool) { + if c == nil || status.InstallationID == 0 { + return Status{}, false + } + status.UpdatedAt = c.now() + c.mu.Lock() + defer c.mu.Unlock() + previous, had := c.entries[status.InstallationID] + c.entries[status.InstallationID] = status + return previous, !had || previous.State != status.State +} + +// Forget drops the installation's status, for a provider whose process +// stopped: an origin nobody serves must not stay in the allow list. +func (c *StatusCache) Forget(installationID int) { + if c == nil { + return + } + c.mu.Lock() + delete(c.entries, installationID) + c.mu.Unlock() +} + +// Get returns the status of one installation. +func (c *StatusCache) Get(installationID int) (Status, bool) { + if c == nil { + return Status{}, false + } + c.mu.RLock() + defer c.mu.RUnlock() + status, ok := c.entries[installationID] + return status, ok +} + +// ByProvider returns the status of the provider with the given slug. +func (c *StatusCache) ByProvider(provider string) (Status, bool) { + if c == nil { + return Status{}, false + } + c.mu.RLock() + defer c.mu.RUnlock() + for _, status := range c.entries { + if status.Provider == provider { + return status, true + } + } + return Status{}, false +} + +// List returns every cached status ordered by installation id. +func (c *StatusCache) List() []Status { + if c == nil { + return nil + } + c.mu.RLock() + out := make([]Status, 0, len(c.entries)) + for _, status := range c.entries { + out = append(out, status) + } + c.mu.RUnlock() + sort.Slice(out, func(i, j int) bool { return out[i].InstallationID < out[j].InstallationID }) + return out +} + +// ConnectedOrigins returns the normalized scheme://host[:port] origins of +// every connected provider on this host: the api origin plus each exposed +// listener's origin. Browsers reaching Silo over the overlay send one of +// these as Origin, so the WebSocket handshake accepts them next to the +// configured public URL. +func (c *StatusCache) ConnectedOrigins() []string { + if c == nil { + return nil + } + c.mu.RLock() + defer c.mu.RUnlock() + seen := make(map[string]struct{}) + var out []string + add := func(raw string) { + origin, ok := NormalizeOrigin(raw) + if !ok { + return + } + if _, dup := seen[origin]; dup { + return + } + seen[origin] = struct{}{} + out = append(out, origin) + } + for _, status := range c.entries { + if !status.Connected() { + continue + } + add(status.Origin) + for _, listener := range status.Listeners { + add(listener.Origin) + } + } + sort.Strings(out) + return out +} + +// NormalizeOrigin reduces raw to lowercase scheme://host[:port] and reports +// whether raw was an http(s) origin at all. +func NormalizeOrigin(raw string) (string, bool) { + raw = strings.TrimSpace(raw) + if raw == "" { + return "", false + } + u, err := url.Parse(raw) + if err != nil || u.Host == "" || u.User != nil { + return "", false + } + scheme := strings.ToLower(u.Scheme) + if scheme != "http" && scheme != "https" { + return "", false + } + return scheme + "://" + strings.ToLower(u.Host), true +} diff --git a/internal/nodepool/admin_configuration_test.go b/internal/nodepool/admin_configuration_test.go index 1176f5a1c7..1bd61e3871 100644 --- a/internal/nodepool/admin_configuration_test.go +++ b/internal/nodepool/admin_configuration_test.go @@ -97,7 +97,7 @@ func TestAdminNodeConfigurationTransactions(t *testing.T) { } }) t.Run("health sample preserves configuration validator", func(t *testing.T) { - if err := bridge.UpdateHealth(ctx, node.ID, node.URL, true, 3, 40, nil); err != nil { + if err := bridge.UpdateHealth(ctx, node.ID, node.URL, true, 3, 40, nil, nil); err != nil { t.Fatal(err) } rows, g, err := store.Snapshot(ctx) diff --git a/internal/nodepool/health.go b/internal/nodepool/health.go index 4395bc42f0..1a1c1a51f7 100644 --- a/internal/nodepool/health.go +++ b/internal/nodepool/health.go @@ -13,6 +13,8 @@ import ( "sync" "time" "unicode/utf8" + + "github.com/Silo-Server/silo-server/internal/netaccess" ) // healthResponse is the JSON response from a node's /health endpoint. @@ -36,6 +38,13 @@ type healthResponse struct { // Build is the node's own build identity, carried opaquely for the same // reason as the sample: it is display data for the nodes dashboard. Build json.RawMessage `json:"build"` + // NetworkAccess is the node's report about the network access provider + // plugins running beside it, keyed by provider slug. Unlike the sample it + // is decoded here, because stream URL selection routes on it: an overlay + // client is handed the origin the matching provider reports on the proxy. + // Absent on a node that predates the field or runs no providers, which + // reads as "no overlay origins" — see CheckNode. + NetworkAccess netaccess.NodeNetworkAccess `json:"network_access,omitempty"` } // maxHealthResponseBytes bounds a node's whole /health body. @@ -56,46 +65,54 @@ const maxHealthResponseBytes = 256 << 10 const maxLastStatsBytes = 32 << 10 // CheckNode pings a node's /health endpoint and returns its health status, -// active job count, reported egress bandwidth, capability hash, and the opaque -// resource-stats blob to persist (nil when the node reported none). -func CheckNode(ctx context.Context, n *Node) (healthy bool, activeJobs, egressKbps int, capabilitiesHash string, lastStats []byte) { +// active job count, reported egress bandwidth, capability hash, the opaque +// resource-stats blob to persist (nil when the node reported none), and the +// node's network access report (nil when it reported none). +// +// A missing network_access field is a report of no providers, not "keep what +// you had": the node either predates the field or runs no provider, and in +// both cases no overlay client can reach it through one. Treating absence as +// "unchanged" would let an origin from a previous build outlive the process +// that served it, and the cost of clearing is only a fallback to the API +// relay until the next 30 s sweep. +func CheckNode(ctx context.Context, n *Node) (healthy bool, activeJobs, egressKbps int, capabilitiesHash string, lastStats []byte, networkAccess netaccess.NodeNetworkAccess) { client := &http.Client{Timeout: 5 * time.Second} ctx, cancel := context.WithTimeout(ctx, 5*time.Second) defer cancel() req, err := http.NewRequestWithContext(ctx, http.MethodGet, NodeEndpoint(n.URL, "/api/v1/health"), nil) if err != nil { - return false, 0, 0, "", nil + return false, 0, 0, "", nil, nil } resp, err := client.Do(req) if err != nil { - return false, 0, 0, "", nil + return false, 0, 0, "", nil, nil } defer resp.Body.Close() if resp.StatusCode != http.StatusOK { - return false, 0, 0, "", nil + return false, 0, 0, "", nil, nil } body, err := io.ReadAll(io.LimitReader(resp.Body, maxHealthResponseBytes+1)) if err != nil { - return false, 0, 0, "", nil + return false, 0, 0, "", nil, nil } if len(body) > maxHealthResponseBytes { // Nothing in the body can be trusted to be well-formed at that size, so // the node is treated as not answering rather than partially believed. slog.WarnContext(ctx, "node health response too large to read", "component", "nodepool", "id", n.ID, "name", n.Name, "url", n.URL, "limit_bytes", maxHealthResponseBytes) - return false, 0, 0, "", nil + return false, 0, 0, "", nil, nil } var hr healthResponse if err := json.Unmarshal(body, &hr); err != nil { - return false, 0, 0, "", nil + return false, 0, 0, "", nil, nil } - return true, hr.ActiveJobs, hr.EgressKbps, hr.CapabilitiesHash, marshalLastStats(ctx, n, hr) + return true, hr.ActiveJobs, hr.EgressKbps, hr.CapabilitiesHash, marshalLastStats(ctx, n, hr), hr.NetworkAccess.Normalized() } // marshalLastStats packs a health response's resource fields and build @@ -315,7 +332,7 @@ func (hc *HealthChecker) Start(ctx context.Context) { } // applyHealthFunc is a pool's copy-on-write health writer. -type applyHealthFunc func(id int, checkedURL string, healthy bool, activeJobs, egressKbps int, advertisedHash string, lastStats []byte, checkedAt time.Time) +type applyHealthFunc func(id int, checkedURL string, healthy bool, activeJobs, egressKbps int, advertisedHash string, lastStats []byte, networkAccess netaccess.NodeNetworkAccess, checkedAt time.Time) // applyCapabilitiesFunc is a pool's copy-on-write capability writer. type applyCapabilitiesFunc func(id int, fetchedFrom string, capabilities []byte, hash string, refreshedAt time.Time, drift *string, driftBaseline []byte) @@ -324,13 +341,13 @@ func (hc *HealthChecker) checkAll(ctx context.Context) { var wg sync.WaitGroup check := func(n *Node, applyHealth applyHealthFunc, applyCapabilities applyCapabilitiesFunc) { wg.Go(func() { - healthy, activeJobs, egressKbps, capabilitiesHash, lastStats := CheckNode(ctx, n) + healthy, activeJobs, egressKbps, capabilitiesHash, lastStats, networkAccess := CheckNode(ctx, n) // Publish the result through the pool lock so readers never see // a Node struct mutated in place (the pool swaps in a copy). Fenced // on the checked URL, like the database write below: the pool can be // reloaded with a different worker on this id while the check runs. - applyHealth(n.ID, n.URL, healthy, activeJobs, egressKbps, capabilitiesHash, lastStats, time.Now()) + applyHealth(n.ID, n.URL, healthy, activeJobs, egressKbps, capabilitiesHash, lastStats, networkAccess, time.Now()) if n.Healthy && !healthy { slog.WarnContext(ctx, "stream node unhealthy", "component", "nodepool", "id", n.ID, "name", n.Name, "url", n.URL) @@ -342,7 +359,7 @@ func (hc *HealthChecker) checkAll(ctx context.Context) { // Fenced on the URL that was checked: last_stats feeds transcode // admission, so one worker's disk reading must never land on a // row an administrator has since repointed at another. - if err := hc.repo.UpdateHealth(ctx, n.ID, n.URL, healthy, activeJobs, egressKbps, lastStats); err != nil { + if err := hc.repo.UpdateHealth(ctx, n.ID, n.URL, healthy, activeJobs, egressKbps, lastStats, networkAccess); err != nil { if errors.Is(err, ErrNodeMoved) { slog.InfoContext(ctx, "discarded a health result for a node that changed identity mid-check", "component", "nodepool", "id", n.ID, "name", n.Name, "url", n.URL) diff --git a/internal/nodepool/health_stats_test.go b/internal/nodepool/health_stats_test.go index bd767c4c19..46fd8c9ac1 100644 --- a/internal/nodepool/health_stats_test.go +++ b/internal/nodepool/health_stats_test.go @@ -43,7 +43,7 @@ func TestCheckNodePassesResourceStatsThroughOpaquely(t *testing.T) { "future_field_we_do_not_know":true}] }`) - healthy, activeJobs, egressKbps, hash, lastStats := CheckNode(context.Background(), &Node{URL: url}) + healthy, activeJobs, egressKbps, hash, lastStats, _ := CheckNode(context.Background(), &Node{URL: url}) if !healthy || activeJobs != 2 || egressKbps != 17 || hash != "sha256:abc" { t.Fatalf("check = %v/%d/%d/%q, want the existing fields unchanged", healthy, activeJobs, egressKbps, hash) } @@ -84,7 +84,7 @@ func TestCheckNodeReportsNoStatsForOlderNodes(t *testing.T) { } { t.Run(name, func(t *testing.T) { url := newStatsHealthNode(t, body) - healthy, _, _, _, lastStats := CheckNode(context.Background(), &Node{URL: url}) + healthy, _, _, _, lastStats, _ := CheckNode(context.Background(), &Node{URL: url}) if !healthy { t.Fatal("node reported unhealthy") } @@ -100,7 +100,7 @@ func TestCheckNodeReportsNoStatsForOlderNodes(t *testing.T) { // revision it is running. func TestCheckNodeKeepsBuildIdentity(t *testing.T) { url := newStatsHealthNode(t, `{"status":"ok","build":{"display":"abcdef12","revision":"abcdef1234","build_number":42,"available":true}}`) - _, _, _, _, lastStats := CheckNode(context.Background(), &Node{URL: url}) + _, _, _, _, lastStats, _ := CheckNode(context.Background(), &Node{URL: url}) var decoded struct { Build struct { Display string `json:"display"` @@ -118,7 +118,7 @@ func TestCheckNodeKeepsBuildIdentity(t *testing.T) { // A node that reports only one half still persists that half. func TestCheckNodeKeepsAPartialSample(t *testing.T) { url := newStatsHealthNode(t, `{"status":"ok","gpu":[{"device":"cuda:0","source":"nvidia-smi"}]}`) - _, _, _, _, lastStats := CheckNode(context.Background(), &Node{URL: url}) + _, _, _, _, lastStats, _ := CheckNode(context.Background(), &Node{URL: url}) var decoded map[string]json.RawMessage if err := json.Unmarshal(lastStats, &decoded); err != nil { t.Fatalf("lastStats invalid: %v (%s)", err, lastStats) @@ -140,7 +140,7 @@ func TestCheckNodeDropsAnOversizedResourceSample(t *testing.T) { url := newStatsHealthNode(t, `{"status":"ok","active_jobs":3,"egress_kbps":9, "system":{"cpu_pct":41,"junk":"`+padding+`"}}`) - healthy, activeJobs, egressKbps, _, lastStats := CheckNode(context.Background(), &Node{URL: url}) + healthy, activeJobs, egressKbps, _, lastStats, _ := CheckNode(context.Background(), &Node{URL: url}) if !healthy || activeJobs != 3 || egressKbps != 9 { t.Fatalf("check = %v/%d/%d, want the health verdict kept", healthy, activeJobs, egressKbps) } @@ -155,7 +155,7 @@ func TestCheckNodeRejectsAnOversizedHealthBody(t *testing.T) { url := newStatsHealthNode(t, `{"status":"ok","active_jobs":3,"junk":"`+ strings.Repeat("x", maxHealthResponseBytes)+`"}`) - healthy, activeJobs, _, _, lastStats := CheckNode(context.Background(), &Node{URL: url}) + healthy, activeJobs, _, _, lastStats, _ := CheckNode(context.Background(), &Node{URL: url}) if healthy || activeJobs != 0 || lastStats != nil { t.Fatalf("check = %v/%d/%s, want an unreadable body treated as no answer", healthy, activeJobs, lastStats) } @@ -167,13 +167,13 @@ func TestApplyHealthClearsStatsWhenACheckCarriesNone(t *testing.T) { pool := NewTranscodePool() pool.SetNodes([]*Node{{ID: 1, URL: "http://node", Enabled: true}}) - pool.ApplyHealth(1, "http://node", true, 1, 0, "", []byte(`{"system":{"cpu_pct":41}}`), time.Now()) + pool.ApplyHealth(1, "http://node", true, 1, 0, "", []byte(`{"system":{"cpu_pct":41}}`), nil, time.Now()) stored := pool.Nodes()[0] if len(stored.LastStats) == 0 { t.Fatal("stats were not published to the pool") } - pool.ApplyHealth(1, "http://node", false, 0, 0, "", nil, time.Now()) + pool.ApplyHealth(1, "http://node", false, 0, 0, "", nil, nil, time.Now()) if got := pool.Nodes()[0].LastStats; got != nil { t.Fatalf("LastStats = %s after a failed check, want nil", got) } @@ -186,7 +186,7 @@ func TestApplyHealthClonesStats(t *testing.T) { pool.SetNodes([]*Node{{ID: 1, URL: "http://node", Enabled: true}}) buffer := []byte(`{"system":{"cpu_pct":41}}`) - pool.ApplyHealth(1, "http://node", true, 0, 0, "", buffer, time.Now()) + pool.ApplyHealth(1, "http://node", true, 0, 0, "", buffer, nil, time.Now()) copy(buffer, []byte(`{"system":{"cpu_pct":99}}`)) if got := string(pool.Nodes()[0].LastStats); got != `{"system":{"cpu_pct":41}}` { @@ -203,7 +203,7 @@ func TestApplyHealthIgnoresAResultForAReplacedWorker(t *testing.T) { pool := NewTranscodePool() pool.SetNodes([]*Node{{ID: 1, URL: "http://replacement", Enabled: true}}) - pool.ApplyHealth(1, "http://original", true, 7, 0, "", []byte(`{"system":{"cpu_pct":41}}`), time.Now()) + pool.ApplyHealth(1, "http://original", true, 7, 0, "", []byte(`{"system":{"cpu_pct":41}}`), nil, time.Now()) stored := pool.Nodes()[0] if stored.ActiveJobs != 0 || len(stored.LastStats) != 0 || stored.LastHealthCheck != nil { @@ -217,7 +217,7 @@ func TestApplyHealthAcceptsATrailingSlashDifference(t *testing.T) { pool := NewTranscodePool() pool.SetNodes([]*Node{{ID: 1, URL: "http://node/", Enabled: true}}) - pool.ApplyHealth(1, "http://node", true, 3, 0, "", nil, time.Now()) + pool.ApplyHealth(1, "http://node", true, 3, 0, "", nil, nil, time.Now()) if stored := pool.Nodes()[0]; stored.ActiveJobs != 3 { t.Fatalf("active jobs = %d, want the result applied despite the trailing slash", stored.ActiveJobs) diff --git a/internal/nodepool/health_test.go b/internal/nodepool/health_test.go index 42aae6b98a..12c6cf59a6 100644 --- a/internal/nodepool/health_test.go +++ b/internal/nodepool/health_test.go @@ -570,7 +570,7 @@ func TestCheckNodeReachesATrailingSlashBaseURL(t *testing.T) { })) defer node.Close() - healthy, activeJobs, _, hash, _ := CheckNode(context.Background(), &Node{ID: 1, URL: node.URL + "/"}) + healthy, activeJobs, _, hash, _, _ := CheckNode(context.Background(), &Node{ID: 1, URL: node.URL + "/"}) if !healthy { t.Fatal("a node stored with a trailing slash was reported unhealthy") } diff --git a/internal/nodepool/network_access_test.go b/internal/nodepool/network_access_test.go new file mode 100644 index 0000000000..f96025ce42 --- /dev/null +++ b/internal/nodepool/network_access_test.go @@ -0,0 +1,306 @@ +package nodepool + +import ( + "context" + "encoding/json" + "fmt" + "net/http" + "net/http/httptest" + "sync/atomic" + "testing" + "time" + + "github.com/Silo-Server/silo-server/internal/netaccess" +) + +// ClientURLFor is the one accessor every client-facing proxy URL is built on. +// The default path keeps ClientURL's public-over-backend rule; a provider path +// gets exactly that provider's connected origin, and nothing else — never the +// LAN address, never another provider's origin, never a non-connected one. +func TestNodeClientURLFor(t *testing.T) { + public := "https://cdn.example.com/" + node := &Node{ + URL: "http://10.0.0.9:8083/", PublicURL: &public, + NetworkAccess: netaccess.NodeNetworkAccess{ + "tailscale": {State: netaccess.StateConnected, Origin: "https://proxy-1.tail1234.ts.net/"}, + "netbird": {State: netaccess.StateConnecting, Origin: "https://proxy-1.netbird.example"}, + "garbage": {State: netaccess.StateConnected, Origin: "::not-an-origin"}, + }, + } + noPublic := &Node{URL: "http://10.0.0.9:8083/"} + for _, tc := range []struct { + name string + node *Node + path netaccess.Path + want string + }{ + {"default path uses the public URL", node, netaccess.Path{}, "https://cdn.example.com"}, + {"default path falls back to the backend URL", noPublic, netaccess.Path{}, "http://10.0.0.9:8083"}, + {"connected provider origin", node, netaccess.Path{Provider: "tailscale"}, "https://proxy-1.tail1234.ts.net"}, + {"connecting provider has no origin", node, netaccess.Path{Provider: "netbird"}, ""}, + {"malformed origin is no origin", node, netaccess.Path{Provider: "garbage"}, ""}, + {"unknown provider", node, netaccess.Path{Provider: "zerotier"}, ""}, + {"node without any report", noPublic, netaccess.Path{Provider: "tailscale"}, ""}, + {"nil node", nil, netaccess.Path{Provider: "tailscale"}, ""}, + {"nil node default path", nil, netaccess.Path{}, ""}, + } { + t.Run(tc.name, func(t *testing.T) { + if got := tc.node.ClientURLFor(tc.path); got != tc.want { + t.Fatalf("ClientURLFor(%+v) = %q, want %q", tc.path, got, tc.want) + } + }) + } + if got := node.ClientURLFor(netaccess.Path{}); got != node.ClientURL() { + t.Fatalf("default path = %q, ClientURL() = %q; they must agree", got, node.ClientURL()) + } +} + +func TestClientReachableVia(t *testing.T) { + reachable := &Node{ID: 1, URL: "http://lan-1", NetworkAccess: netaccess.NodeNetworkAccess{"tailscale": {State: netaccess.StateConnected, Origin: "https://p1.ts.net"}}} + lanOnly := &Node{ID: 2, URL: "http://lan-2"} + tailnet := netaccess.Path{Provider: "tailscale"} + + if got := ClientReachableVia(netaccess.Path{}, nil); got != nil { + t.Fatal("default path with no base predicate must stay nil so planners accept any healthy proxy") + } + calls := 0 + base := func(n *Node) bool { calls++; return n.ID == 2 } + wrapped := ClientReachableVia(tailnet, base) + if wrapped(reachable) { + t.Fatal("base rejected the reachable proxy but the wrapper accepted it") + } + if wrapped(lanOnly) { + t.Fatal("a LAN-only proxy passed on a provider path") + } + if wrapped(nil) { + t.Fatal("nil node accepted") + } + if calls != 1 { + t.Fatalf("base predicate ran %d times, want once (only for the reachable proxy)", calls) + } + if !ClientReachableVia(tailnet, nil)(reachable) { + t.Fatal("reachable proxy rejected with no base predicate") + } + if got := ClientReachableVia(netaccess.Path{}, base); got == nil || !got(lanOnly) { + t.Fatal("default path must hand back the base predicate unchanged") + } +} + +// newNetworkAccessHealthNode serves a health body whose network_access block +// the test can swap between sweeps. +func newNetworkAccessHealthNode(t *testing.T) (string, *atomic.Pointer[string]) { + t.Helper() + var block atomic.Pointer[string] + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + if r.URL.Path != "/api/v1/health" { + http.NotFound(w, r) + return + } + body := `{"status":"ok","active_jobs":1` + if extra := block.Load(); extra != nil && *extra != "" { + body += `,"network_access":` + *extra + } + body += `}` + w.Header().Set("Content-Type", "application/json") + _, _ = w.Write([]byte(body)) + })) + t.Cleanup(server.Close) + return server.URL, &block +} + +// The sweep is the only path that learns a proxy's overlay origins. It has to +// publish them to the pool copy the planner reads, and a later check that +// carries none has to clear them again — an origin nobody confirms any more +// must not keep being handed to overlay clients. +func TestHealthCheckerStoresNetworkAccessFromTheHealthPull(t *testing.T) { + url, block := newNetworkAccessHealthNode(t) + report := `{" tailscale ":{"state":"connected","origin":" https://proxy-1.tail1234.ts.net ","hostname":"proxy-1.tail1234.ts.net","updated_at":"2026-09-14T08:00:00Z"},"netbird":{"state":"awaiting_authorization"}}` + block.Store(&report) + pool := NewProxyPool() + pool.SetNodes([]*Node{{ID: 9, Name: "proxy-1", URL: url, Enabled: true}}) + checker := NewHealthChecker(pool, NewTranscodePool(), nil) + + checker.checkAll(context.Background()) + + stored := pool.Nodes()[0] + if !stored.Healthy { + t.Fatalf("node not healthy: %+v", stored) + } + if got := stored.ClientURLFor(netaccess.Path{Provider: "tailscale"}); got != "https://proxy-1.tail1234.ts.net" { + t.Fatalf("tailscale origin after sweep = %q (report %#v)", got, stored.NetworkAccess) + } + if got := stored.NetworkAccess["tailscale"]; got.Hostname != "proxy-1.tail1234.ts.net" || got.UpdatedAt.IsZero() { + t.Fatalf("stored entry lost fields: %#v", got) + } + if got := stored.ClientURLFor(netaccess.Path{Provider: "netbird"}); got != "" { + t.Fatalf("an unauthorized provider yielded %q", got) + } + if got := stored.ClientURLFor(netaccess.Path{}); got != NormalizeNodeURL(url) { + t.Fatalf("default path = %q, want the backend URL", got) + } + + // A build that predates the field, or one whose providers all stopped, + // answers without the block: the stored report is cleared, not kept. + empty := "" + block.Store(&empty) + checker.checkAll(context.Background()) + stored = pool.Nodes()[0] + if !stored.Healthy || stored.NetworkAccess != nil { + t.Fatalf("after a check without the field: healthy=%v report=%#v, want healthy and no report", stored.Healthy, stored.NetworkAccess) + } + if got := stored.ClientURLFor(netaccess.Path{Provider: "tailscale"}); got != "" { + t.Fatalf("stale origin survived a report-less check: %q", got) + } +} + +func TestCheckNodeIgnoresAMalformedNetworkAccessBlock(t *testing.T) { + url, block := newNetworkAccessHealthNode(t) + bad := `"connected"` + block.Store(&bad) + healthy, _, _, _, _, report := CheckNode(context.Background(), &Node{URL: url}) + // A body that does not decode is the existing "node did not answer" + // verdict; the point here is that the block cannot smuggle in a report. + if healthy || report != nil { + t.Fatalf("healthy=%v report=%#v, want the undecodable body rejected whole", healthy, report) + } +} + +// The health update is the only write path for network_access, and it has to +// round-trip through the ordinary node read the pools and the admin API use. +func TestRepositoryUpdateHealthPersistsNetworkAccess(t *testing.T) { + repo := newNetworkAccessTestRepo(t) + ctx := context.Background() + unique := time.Now().UnixNano() + node, err := repo.Create(ctx, CreateNodeInput{ + Name: fmt.Sprintf("network-access-test-%d", unique), + Type: NodeTypeProxy, + URL: fmt.Sprintf("http://network-access-test-%d", unique), + }) + if err != nil { + t.Fatalf("create node: %v", err) + } + t.Cleanup(func() { _ = repo.Delete(ctx, node.ID) }) + if node.NetworkAccess != nil { + t.Fatalf("new node already carries a report: %#v", node.NetworkAccess) + } + + at := time.Date(2026, 9, 14, 8, 0, 0, 0, time.UTC) + report := netaccess.NodeNetworkAccess{"tailscale": {State: netaccess.StateConnected, Origin: "https://proxy-1.tail1234.ts.net", Hostname: "proxy-1.tail1234.ts.net", UpdatedAt: at}} + if err := repo.UpdateHealth(ctx, node.ID, node.URL, true, 1, 0, nil, report); err != nil { + t.Fatalf("update health: %v", err) + } + reloaded, err := repo.GetByID(ctx, node.ID) + if err != nil { + t.Fatalf("reload: %v", err) + } + got, ok := reloaded.NetworkAccess["tailscale"] + if !ok || got.State != netaccess.StateConnected || got.Origin != "https://proxy-1.tail1234.ts.net" || got.Hostname != "proxy-1.tail1234.ts.net" || !got.UpdatedAt.Equal(at) { + t.Fatalf("stored report = %#v", reloaded.NetworkAccess) + } + if url := reloaded.ClientURLFor(netaccess.Path{Provider: "tailscale"}); url != "https://proxy-1.tail1234.ts.net" { + t.Fatalf("ClientURLFor from a stored row = %q", url) + } + + // The list read shares the scan path; make sure it decodes too. + listed, err := repo.List(ctx) + if err != nil { + t.Fatalf("list: %v", err) + } + found := false + for _, n := range listed { + if n.ID == node.ID { + found = true + if _, ok := n.NetworkAccess["tailscale"]; !ok { + t.Fatalf("listed row lost the report: %#v", n.NetworkAccess) + } + } + } + if !found { + t.Fatal("created node missing from list") + } + + // A check with no report clears the column back to its empty form. + if err := repo.UpdateHealth(ctx, node.ID, node.URL, true, 0, 0, nil, nil); err != nil { + t.Fatalf("clear: %v", err) + } + reloaded, err = repo.GetByID(ctx, node.ID) + if err != nil { + t.Fatalf("reload after clear: %v", err) + } + if reloaded.NetworkAccess != nil { + t.Fatalf("report survived a report-less health write: %#v", reloaded.NetworkAccess) + } + var raw string + if err := repo.pool.QueryRow(ctx, `SELECT network_access::text FROM stream_nodes WHERE id = $1`, node.ID).Scan(&raw); err != nil { + t.Fatalf("read column: %v", err) + } + if raw != "{}" { + t.Fatalf("cleared column = %s, want {} (NOT NULL column)", raw) + } +} + +// Repointing the row at another machine drops the report with the rest of +// the identity-bound state: the overlay origin described the old worker. +func TestRepositoryUpdateURLClearsNetworkAccess(t *testing.T) { + repo := newNetworkAccessTestRepo(t) + ctx := context.Background() + unique := time.Now().UnixNano() + node, err := repo.Create(ctx, CreateNodeInput{ + Name: fmt.Sprintf("network-access-move-%d", unique), + Type: NodeTypeProxy, + URL: fmt.Sprintf("http://network-access-move-%d", unique), + }) + if err != nil { + t.Fatalf("create node: %v", err) + } + t.Cleanup(func() { _ = repo.Delete(ctx, node.ID) }) + report := netaccess.NodeNetworkAccess{"tailscale": {State: netaccess.StateConnected, Origin: "https://old.ts.net"}} + if err := repo.UpdateHealth(ctx, node.ID, node.URL, true, 0, 0, nil, report); err != nil { + t.Fatalf("update health: %v", err) + } + sameName := node.Name + if kept, err := repo.Update(ctx, node.ID, UpdateNodeInput{Name: &sameName}); err != nil || kept.NetworkAccess == nil { + t.Fatalf("a non-URL edit dropped the report: %#v err=%v", kept, err) + } + moved := fmt.Sprintf("http://network-access-moved-%d", unique) + updated, err := repo.Update(ctx, node.ID, UpdateNodeInput{URL: &moved}) + if err != nil { + t.Fatalf("update url: %v", err) + } + if updated.NetworkAccess != nil { + t.Fatalf("report survived a URL change: %#v", updated.NetworkAccess) + } +} + +func newNetworkAccessTestRepo(t *testing.T) *Repository { + t.Helper() + pool := newNodeTestPool(t) + var column *string + if err := pool.QueryRow(context.Background(), + `SELECT column_name FROM information_schema.columns + WHERE table_name = 'stream_nodes' AND column_name = 'network_access'`).Scan(&column); err != nil { + t.Skip("test database has not applied the node network_access migration") + } + return NewRepository(pool) +} + +// The jsonb round trip must keep the wire keys the proxy sends, so a report a +// proxy produced with the netaccess types decodes identically after storage. +func TestNodeNetworkAccessJSONRoundTrip(t *testing.T) { + at := time.Date(2026, 9, 14, 8, 0, 0, 0, time.UTC) + in := netaccess.NodeNetworkAccess{"tailscale": {State: netaccess.StateConnected, Origin: "https://p.ts.net", Hostname: "p.ts.net", UpdatedAt: at}} + encoded, err := marshalNetworkAccess(in) + if err != nil { + t.Fatal(err) + } + var out netaccess.NodeNetworkAccess + if err := json.Unmarshal(encoded, &out); err != nil { + t.Fatal(err) + } + if out["tailscale"] != in["tailscale"] { + t.Fatalf("round trip = %#v, want %#v", out, in) + } + if empty, _ := marshalNetworkAccess(nil); string(empty) != "{}" { + t.Fatalf("nil report marshals to %s, want {}", empty) + } +} diff --git a/internal/nodepool/planner.go b/internal/nodepool/planner.go index 194a236f2e..1ef96f9939 100644 --- a/internal/nodepool/planner.go +++ b/internal/nodepool/planner.go @@ -45,7 +45,7 @@ type SessionPlanner interface { // have no predictable bitrate, so implementations must not admit them onto a // proxy with a configured bandwidth cap. type DownloadPlanner interface { - PlanDownload(sessionID string, preferredGroup ...string) Plan + PlanDownloadWith(sessionID string, eligible func(*Node) bool, preferredGroup ...string) Plan ReleaseSession(sessionID string) } @@ -205,6 +205,13 @@ func (p *Planner) TranscodeNodeHealthy(nodeURL string) bool { // transfer rate, so capped proxies are excluded instead of being oversubscribed // during the egress meter's convergence window. func (p *Planner) PlanDownload(sessionID string, preferredGroup ...string) Plan { + return p.PlanDownloadWith(sessionID, nil, preferredGroup...) +} + +// PlanDownloadWith restricts selection to eligible proxies before applying +// group preferences and reserving capacity. The predicate runs under the +// planner lock and must be cheap and non-blocking; nil accepts every proxy. +func (p *Planner) PlanDownloadWith(sessionID string, eligible func(*Node) bool, preferredGroup ...string) Plan { if p == nil || p.proxies == nil || sessionID == "" { return Plan{} } @@ -221,6 +228,9 @@ func (p *Planner) PlanDownload(sessionID string, preferredGroup ...string) Plan } var candidates, fallback []*Node for _, node := range p.proxies.Nodes() { + if node != nil && eligible != nil && !eligible(node) { + continue + } if node == nil || !node.Enabled || !node.Healthy || !p.underCap(node, now) { continue } diff --git a/internal/nodepool/planner_test.go b/internal/nodepool/planner_test.go index 42e79b4863..a7c34719b9 100644 --- a/internal/nodepool/planner_test.go +++ b/internal/nodepool/planner_test.go @@ -559,7 +559,7 @@ func TestBandwidthReservationsCountDuringBridge(t *testing.T) { // Unlike job reservations, bandwidth bridges ignore health freshness — // a report right after admission would not reflect the streams yet. newer := f.now.Add(5 * time.Second) - f.proxies.ApplyHealth(1, f.proxies.Nodes()[0].URL, true, 0, 0, "", nil, newer) + f.proxies.ApplyHealth(1, f.proxies.Nodes()[0].URL, true, 0, 0, "", nil, nil, newer) f.now = f.now.Add(10 * time.Second) if got := f.planner.PlanSession("s4", "", false, 4_000).ProxyNode; got != nil { t.Fatalf("stream should still be rejected during bridge window, got %+v", got) @@ -569,7 +569,7 @@ func TestBandwidthReservationsCountDuringBridge(t *testing.T) { // meter now reports 8 Mbps, so one more 4 Mbps stream still won't fit, // but a 2 Mbps one will. f.now = f.now.Add(bandwidthBridgeAge) - f.proxies.ApplyHealth(1, f.proxies.Nodes()[0].URL, true, 0, 8_000, "", nil, f.now) + f.proxies.ApplyHealth(1, f.proxies.Nodes()[0].URL, true, 0, 8_000, "", nil, nil, f.now) if got := f.planner.PlanSession("s5", "", false, 4_000).ProxyNode; got != nil { t.Fatalf("4 Mbps stream should not fit at 8/10 Mbps, got %+v", got) } @@ -621,7 +621,7 @@ func TestUnknownBitrateAdmittedBelowCap(t *testing.T) { t.Fatal("unknown-bitrate stream should be admitted below cap") } - f.proxies.ApplyHealth(1, f.proxies.Nodes()[0].URL, true, 0, 10_000, "", nil, f.now) + f.proxies.ApplyHealth(1, f.proxies.Nodes()[0].URL, true, 0, 10_000, "", nil, nil, f.now) if got := f.planner.PlanSession("s2", "", false, 0).ProxyNode; got != nil { t.Fatalf("unknown-bitrate stream should be rejected at cap, got %+v", got) } diff --git a/internal/nodepool/proxy_pool.go b/internal/nodepool/proxy_pool.go index 355e1d1686..a578d23f32 100644 --- a/internal/nodepool/proxy_pool.go +++ b/internal/nodepool/proxy_pool.go @@ -4,6 +4,8 @@ import ( "sync" "sync/atomic" "time" + + "github.com/Silo-Server/silo-server/internal/netaccess" ) // ProxyPool manages proxy nodes with round-robin selection. @@ -83,10 +85,10 @@ func (p *ProxyPool) Nodes() []*Node { // ApplyHealth records a health check result by swapping the node for an // updated copy, keeping published *Node values immutable. -func (p *ProxyPool) ApplyHealth(id int, checkedURL string, healthy bool, activeJobs, egressKbps int, advertisedHash string, lastStats []byte, checkedAt time.Time) { +func (p *ProxyPool) ApplyHealth(id int, checkedURL string, healthy bool, activeJobs, egressKbps int, advertisedHash string, lastStats []byte, networkAccess netaccess.NodeNetworkAccess, checkedAt time.Time) { p.mu.Lock() defer p.mu.Unlock() - applyNodeHealth(p.nodes, id, checkedURL, healthy, activeJobs, egressKbps, advertisedHash, lastStats, checkedAt) + applyNodeHealth(p.nodes, id, checkedURL, healthy, activeJobs, egressKbps, advertisedHash, lastStats, networkAccess, checkedAt) } // ApplyCapabilities records a freshly fetched capability report by swapping the diff --git a/internal/nodepool/repository.go b/internal/nodepool/repository.go index a83109ee2b..d95809587d 100644 --- a/internal/nodepool/repository.go +++ b/internal/nodepool/repository.go @@ -13,6 +13,8 @@ import ( "github.com/jackc/pgx/v5" "github.com/jackc/pgx/v5/pgconn" "github.com/jackc/pgx/v5/pgxpool" + + "github.com/Silo-Server/silo-server/internal/netaccess" ) const ( @@ -83,6 +85,16 @@ type Node struct { // for. Non-nil exactly when CapabilityDrift is, except on a note written // before this column existed. CapabilityDriftBaseline json.RawMessage `json:"capability_drift_baseline,omitempty"` + // NetworkAccess is the node's last report about the network access + // provider plugins running beside it, keyed by provider slug. It is + // written by the same health update that writes LastStats, so it is exactly + // as fresh as LastHealthCheck: a check that carries no report — a node that + // predates the field, runs no providers, or did not answer — clears it, for + // the same reason a check clears LastStats. A stale connected origin is + // worse than none: it would hand an overlay client a URL nobody serves, + // while an empty entry makes ClientURLFor fall back to the API relay. + // Only ever read for proxy nodes: clients never talk to transcode nodes. + NetworkAccess netaccess.NodeNetworkAccess `json:"network_access,omitempty"` // AdvertisedCapabilitiesHash is the hash the node named on its last health // check, which is not always the one stored beside it: the sweep refetches // on a mismatch, and a refetch that keeps failing leaves the two apart while @@ -143,6 +155,46 @@ func (n *Node) ClientURL() string { return normalizeNodeURL(n.URL) } +// ClientURLFor is the base URL to hand a streaming client that arrived on the +// given access path. The default path — LAN, public URL, reverse proxy — gets +// ClientURL. A client that came through a network access provider (an overlay +// such as a tailnet) cannot reach that address at all, so it gets the origin +// the same provider reported on this node, and an empty string when the node +// has no connected origin for that provider. Empty is a real answer, not a +// missing one: the caller must not use this node for that client and falls +// back to an API-relative URL, which the API server relays. +func (n *Node) ClientURLFor(path netaccess.Path) string { + if n == nil { + return "" + } + if path.IsDefault() { + return n.ClientURL() + } + origin, ok := n.NetworkAccess.ConnectedOrigin(path.Provider) + if !ok { + return "" + } + return normalizeNodeURL(origin) +} + +// ClientReachableVia narrows a proxy eligibility predicate to proxies that +// have a client origin for the request's access path, so the route resolver +// never reserves a proxy the client cannot reach and then falls back on the +// URL builder. The default path reaches every proxy, so base is returned as +// is — nil included, which the planners read as "any healthy proxy". The +// returned predicate runs under the planner lock: a map lookup only. +func ClientReachableVia(path netaccess.Path, base func(*Node) bool) func(*Node) bool { + if path.IsDefault() { + return base + } + return func(n *Node) bool { + if n == nil || n.ClientURLFor(path) == "" { + return false + } + return base == nil || base(n) + } +} + // StoredCapabilities returns this node's last stored capability report, nil-safe // like the Effective* accessors so a caller whose lookup came up empty prices a // missing node and a missing report through one path. @@ -351,13 +403,13 @@ func NewRepository(pool *pgxpool.Pool) *Repository { return &Repository{pool: pool} } -const nodeColumns = `id, name, type, url, public_url, enabled, healthy, active_jobs, node_group, max_jobs, max_bandwidth_kbps, egress_kbps, last_health_check, created_at, capabilities, capabilities_hash, capabilities_refreshed_at, last_stats, hw_accel_override, hw_device_override, capability_drift, capability_drift_baseline` +const nodeColumns = `id, name, type, url, public_url, enabled, healthy, active_jobs, node_group, max_jobs, max_bandwidth_kbps, egress_kbps, last_health_check, created_at, capabilities, capabilities_hash, capabilities_refreshed_at, last_stats, hw_accel_override, hw_device_override, capability_drift, capability_drift_baseline, network_access` func scanNode(row pgx.Row) (*Node, error) { var n Node // jsonb is scanned as raw bytes rather than into json.RawMessage directly so // a NULL column stays nil instead of decoding through the JSON codec. - var capabilities, lastStats, driftBaselineBytes []byte + var capabilities, lastStats, driftBaselineBytes, networkAccessBytes []byte err := row.Scan( &n.ID, &n.Name, &n.Type, &n.URL, &n.PublicURL, &n.Enabled, &n.Healthy, &n.ActiveJobs, @@ -368,10 +420,18 @@ func scanNode(row pgx.Row) (*Node, error) { &lastStats, &n.HWAccelOverride, &n.HWDeviceOverride, &n.CapabilityDrift, &driftBaselineBytes, + &networkAccessBytes, ) if err != nil { return nil, err } + if len(networkAccessBytes) > 0 { + var report netaccess.NodeNetworkAccess + if err := json.Unmarshal(networkAccessBytes, &report); err != nil { + return nil, fmt.Errorf("decode node %d network_access: %w", n.ID, err) + } + n.NetworkAccess = report.Normalized() + } if len(capabilities) > 0 { n.Capabilities = json.RawMessage(capabilities) } @@ -506,7 +566,8 @@ func (r *Repository) Update(ctx context.Context, id int, input UpdateNodeInput) capabilities_refreshed_at = CASE WHEN `+sameURL+` THEN capabilities_refreshed_at END, last_stats = CASE WHEN `+sameURL+` THEN last_stats END, capability_drift = CASE WHEN `+sameURL+` THEN capability_drift END, - capability_drift_baseline = CASE WHEN `+sameURL+` THEN capability_drift_baseline END + capability_drift_baseline = CASE WHEN `+sameURL+` THEN capability_drift_baseline END, + network_access = CASE WHEN `+sameURL+` THEN network_access ELSE '{}'::jsonb END WHERE id = $1 RETURNING `+nodeColumns, id, input.Name, input.URL, input.Enabled, @@ -539,22 +600,28 @@ func (r *Repository) Delete(ctx context.Context, id int) error { } // UpdateHealth updates a node's health status, active job count, reported -// egress bandwidth, and last resource sample. +// egress bandwidth, last resource sample, and last network access report. // // A nil lastStats writes NULL, which is what a node that reports no sample — // an older build, or a non-Linux host — must produce. Passing the previous // value through instead would leave a dead node's numbers on screen looking -// current. +// current. networkAccess follows the same rule with '{}' as its empty form: a +// check that carried no report clears the stored one, so an overlay origin +// never outlives the check that last confirmed it. // checkedURL fences the write the same way UpdateCapabilities does. The window // is smaller — a health request is bounded at five seconds — but the // consequence is not: last_stats carries the scratch fill that transcode // admission reads, so one worker's disk reading landing on a row that now // addresses another can exclude a healthy node or admit a full one. -func (r *Repository) UpdateHealth(ctx context.Context, id int, checkedURL string, healthy bool, activeJobs, egressKbps int, lastStats []byte) error { +func (r *Repository) UpdateHealth(ctx context.Context, id int, checkedURL string, healthy bool, activeJobs, egressKbps int, lastStats []byte, networkAccess netaccess.NodeNetworkAccess) error { + networkAccessJSON, err := marshalNetworkAccess(networkAccess) + if err != nil { + return fmt.Errorf("update node health: %w", err) + } tag, err := r.pool.Exec(ctx, - `UPDATE stream_nodes SET healthy = $2, active_jobs = $3, egress_kbps = $4, last_stats = $5, last_health_check = NOW() + `UPDATE stream_nodes SET healthy = $2, active_jobs = $3, egress_kbps = $4, last_stats = $5, network_access = $7, last_health_check = NOW() WHERE id = $1 AND rtrim(url, '/') = rtrim($6, '/')`, - id, healthy, activeJobs, egressKbps, lastStats, checkedURL) + id, healthy, activeJobs, egressKbps, lastStats, checkedURL, networkAccessJSON) if err != nil { return fmt.Errorf("update node health: %w", err) } @@ -564,6 +631,16 @@ func (r *Repository) UpdateHealth(ctx context.Context, id int, checkedURL string return nil } +// marshalNetworkAccess renders a node's network access report for its jsonb +// column. The column is NOT NULL, so "no report" is the empty object. +func marshalNetworkAccess(report netaccess.NodeNetworkAccess) ([]byte, error) { + report = report.Normalized() + if len(report) == 0 { + return []byte(`{}`), nil + } + return json.Marshal(report) +} + // UpdateCapabilities persists a freshly fetched capability report together with // the hash that identifies it and the drift note comparing it against the // previous one. The four columns are written in one statement so a reader never diff --git a/internal/nodepool/repository_capabilities_test.go b/internal/nodepool/repository_capabilities_test.go index d3922f0265..407680a04c 100644 --- a/internal/nodepool/repository_capabilities_test.go +++ b/internal/nodepool/repository_capabilities_test.go @@ -226,7 +226,7 @@ func TestRepositoryUpdateClearsWorkerStateWhenTheURLMoves(t *testing.T) { if err := repo.UpdateCapabilities(ctx, node.ID, node.URL, payload, "sha256:old", time.Now(), ¬e, []byte(`{"backends":["qsv"]}`), nil); err != nil { t.Fatalf("store capabilities: %v", err) } - if err := repo.UpdateHealth(ctx, node.ID, node.URL, true, 2, 0, []byte(`{"system":{"cpu_pct":41}}`)); err != nil { + if err := repo.UpdateHealth(ctx, node.ID, node.URL, true, 2, 0, []byte(`{"system":{"cpu_pct":41}}`), nil); err != nil { t.Fatalf("store health: %v", err) } diff --git a/internal/nodepool/repository_last_stats_test.go b/internal/nodepool/repository_last_stats_test.go index 5edf93f88e..c87e7e9962 100644 --- a/internal/nodepool/repository_last_stats_test.go +++ b/internal/nodepool/repository_last_stats_test.go @@ -53,7 +53,7 @@ func TestRepositoryUpdateHealthPersistsLastStats(t *testing.T) { } stats := []byte(`{"system":{"cpu_pct":41,"mem_used_mb":9011},"gpu":[{"device":"/dev/dri/renderD128","source":"fdinfo"}]}`) - if err := repo.UpdateHealth(ctx, node.ID, node.URL, true, 3, 17, stats); err != nil { + if err := repo.UpdateHealth(ctx, node.ID, node.URL, true, 3, 17, stats, nil); err != nil { t.Fatalf("update health: %v", err) } @@ -91,10 +91,10 @@ func TestRepositoryUpdateHealthWritesNullForNodesWithoutStats(t *testing.T) { ctx := context.Background() node := createLastStatsNode(t, repo) - if err := repo.UpdateHealth(ctx, node.ID, node.URL, true, 1, 0, []byte(`{"system":{"cpu_pct":41}}`)); err != nil { + if err := repo.UpdateHealth(ctx, node.ID, node.URL, true, 1, 0, []byte(`{"system":{"cpu_pct":41}}`), nil); err != nil { t.Fatalf("update health: %v", err) } - if err := repo.UpdateHealth(ctx, node.ID, node.URL, false, 0, 0, nil); err != nil { + if err := repo.UpdateHealth(ctx, node.ID, node.URL, false, 0, 0, nil, nil); err != nil { t.Fatalf("update health without stats: %v", err) } @@ -142,7 +142,7 @@ func TestRepositoryUpdateHealthRefusesAfterAURLEdit(t *testing.T) { t.Fatalf("repoint node: %v", err) } - err = repo.UpdateHealth(ctx, node.ID, node.URL, true, 3, 17, []byte(`{"system":{"cpu_pct":41}}`)) + err = repo.UpdateHealth(ctx, node.ID, node.URL, true, 3, 17, []byte(`{"system":{"cpu_pct":41}}`), nil) if !errors.Is(err, ErrNodeMoved) { t.Fatalf("err = %v, want ErrNodeMoved after the row was repointed", err) } diff --git a/internal/nodepool/transcode_pool.go b/internal/nodepool/transcode_pool.go index 6fde483b8a..548db472a4 100644 --- a/internal/nodepool/transcode_pool.go +++ b/internal/nodepool/transcode_pool.go @@ -5,6 +5,8 @@ import ( "strings" "sync" "time" + + "github.com/Silo-Server/silo-server/internal/netaccess" ) // TranscodePool manages transcode nodes with least-connections selection. @@ -86,10 +88,10 @@ func (p *TranscodePool) Nodes() []*Node { // ApplyHealth records a health check result by swapping the node for an // updated copy, keeping published *Node values immutable. -func (p *TranscodePool) ApplyHealth(id int, checkedURL string, healthy bool, activeJobs, egressKbps int, advertisedHash string, lastStats []byte, checkedAt time.Time) { +func (p *TranscodePool) ApplyHealth(id int, checkedURL string, healthy bool, activeJobs, egressKbps int, advertisedHash string, lastStats []byte, networkAccess netaccess.NodeNetworkAccess, checkedAt time.Time) { p.mu.Lock() defer p.mu.Unlock() - applyNodeHealth(p.nodes, id, checkedURL, healthy, activeJobs, egressKbps, advertisedHash, lastStats, checkedAt) + applyNodeHealth(p.nodes, id, checkedURL, healthy, activeJobs, egressKbps, advertisedHash, lastStats, networkAccess, checkedAt) } // ApplyCapabilities records a freshly fetched capability report by swapping the @@ -136,7 +138,7 @@ func sameNodeURL(a, b string) bool { // by id alone would then write one worker's health — and the scratch fill // transcode admission reads — onto the replacement, and the database fence // downstream cannot undo that. The pool would stay wrong until a later sweep. -func applyNodeHealth(nodes []*Node, id int, checkedURL string, healthy bool, activeJobs, egressKbps int, advertisedHash string, lastStats []byte, checkedAt time.Time) { +func applyNodeHealth(nodes []*Node, id int, checkedURL string, healthy bool, activeJobs, egressKbps int, advertisedHash string, lastStats []byte, networkAccess netaccess.NodeNetworkAccess, checkedAt time.Time) { for i, n := range nodes { if n.ID != id || !sameNodeURL(n.URL, checkedURL) { continue @@ -159,6 +161,10 @@ func applyNodeHealth(nodes []*Node, id int, checkedURL string, healthy bool, act } else { clone.LastStats = nil } + // Same rule for the network access report: a check that carried none + // clears it, so ClientURLFor never hands out an origin the node has + // stopped confirming. Cloned so the published copy is immutable. + clone.NetworkAccess = networkAccess.Normalized().Clone() nodes[i] = &clone return } diff --git a/internal/playback/recipecard.go b/internal/playback/recipecard.go index 55f8ea2dc3..fdb9a88189 100644 --- a/internal/playback/recipecard.go +++ b/internal/playback/recipecard.go @@ -32,11 +32,12 @@ type RecipeCard struct { // session on another process. Stable execution and egress identities bind // the artifact to the nodes whose capacity the planner reserved; internal // URLs stay out of the portable recipe. - RoutingWorkload string `json:"routing_workload,omitempty"` - RoutingExecution string `json:"routing_execution,omitempty"` - RoutingExecutionNodeID int `json:"routing_execution_node_id,omitzero"` - RoutingEgress string `json:"routing_egress,omitempty"` - RoutingEgressNodeID int `json:"routing_egress_node_id,omitempty"` + RoutingNetworkProvider *string `json:"routing_network_provider,omitempty"` + RoutingWorkload string `json:"routing_workload,omitempty"` + RoutingExecution string `json:"routing_execution,omitempty"` + RoutingExecutionNodeID int `json:"routing_execution_node_id,omitzero"` + RoutingEgress string `json:"routing_egress,omitempty"` + RoutingEgressNodeID int `json:"routing_egress_node_id,omitempty"` // PlayMethod discriminates which serve path reconstructs this session // (direct / remux / transcode). Empty decodes as PlayTranscode for @@ -359,6 +360,7 @@ func (c RecipeCard) ToClaims() streamtoken.Claims { RemuxDVMode: string(c.RemuxDVMode), TranscodeNode: c.TranscodeNodeURL, TranscodeTransportID: c.TranscodeTransportID, + RoutingNetworkProvider: c.RoutingNetworkProvider, RoutingWorkload: c.RoutingWorkload, RoutingExecution: c.RoutingExecution, RoutingExecutionNodeID: c.RoutingExecutionNodeID, @@ -447,6 +449,7 @@ func RecipeCardFromClaims(c *streamtoken.Claims) RecipeCard { MediaFileID: c.MediaFileID, TranscodeNodeURL: c.TranscodeNode, TranscodeTransportID: c.TranscodeTransportID, + RoutingNetworkProvider: c.RoutingNetworkProvider, RoutingWorkload: c.RoutingWorkload, RoutingExecution: c.RoutingExecution, RoutingExecutionNodeID: c.RoutingExecutionNodeID, diff --git a/internal/playback/recipecard_test.go b/internal/playback/recipecard_test.go index 1054ba991d..51222df9d4 100644 --- a/internal/playback/recipecard_test.go +++ b/internal/playback/recipecard_test.go @@ -179,6 +179,50 @@ func TestRecipeCardPreservesRoutingNodeIDs(t *testing.T) { } } +func TestRecipeCardNetworkRouteSurvivesRecovery(t *testing.T) { + for _, provider := range []*string{nil, new(""), new("tailscale")} { + card := NewDirectRecipeCard("network-route", 42, "profile-1", 77) + card.RoutingNetworkProvider = provider + card.RoutingWorkload = "remux" + card.RoutingExecution = "transcode" + card.RoutingExecutionNodeID = 7 + card.RoutingEgress = "proxy" + card.RoutingEgressNodeID = 11 + for _, token := range []bool{false, true} { + var recovered RecipeCard + if token { + wire, err := json.Marshal(card.ToClaims()) + if err != nil { + t.Fatal(err) + } + var claims streamtoken.Claims + if err := json.Unmarshal(wire, &claims); err != nil { + t.Fatal(err) + } + recovered = RecipeCardFromClaims(&claims) + } else { + wire, err := json.Marshal(card) + if err != nil { + t.Fatal(err) + } + if err := json.Unmarshal(wire, &recovered); err != nil { + t.Fatal(err) + } + } + tm := NewTranscodeManager() + tm.Sessions = NewSessionManager(0, 0) + session := tm.ReconstructSession(t.Context(), card.SessionID, card.UserID, recovered) + if session == nil || session.RoutingExecutionNodeID != 7 || session.RoutingEgressNodeID != 11 { + t.Fatalf("recovered route: %#v", session) + } + got := session.RoutingNetworkProvider + if (got == nil) != (provider == nil) || (got != nil && *got != *provider) { + t.Fatalf("network provider changed through recovery: got %v want %v", got, provider) + } + } + } +} + func ptr[T any](value T) *T { return &value } func TestRecipeCardPlayMethodConstructors(t *testing.T) { diff --git a/internal/playback/session.go b/internal/playback/session.go index a32224fac9..d562232f72 100644 --- a/internal/playback/session.go +++ b/internal/playback/session.go @@ -45,10 +45,13 @@ type Session struct { TranscodeTransportID string // remote node process identity; empty means session ID AudioTrackIndex int + // RoutingNetworkProvider is the validated access path selected when preparing + // playback: nil means unknown, an empty value means the default network. // RoutingWorkload and the execution/egress fields describe the committed // node-routing assignment independently from the transcode process route. // Node URLs are internal identities used by the session sync layer to join // stable stream-node IDs; they are never returned as client media origins. + RoutingNetworkProvider *string RoutingWorkload string RoutingExecution string RoutingExecutionNodeID int @@ -120,6 +123,7 @@ type SessionStreamState struct { TranscodeNodeURL string TranscodeTransportID string TranscodeRouteSet bool + RoutingNetworkProvider *string RoutingWorkload string RoutingExecution string RoutingExecutionNodeID int @@ -153,6 +157,7 @@ type TranscodeRoute struct { // opaque session state to avoid coupling session lifetime management to route // selection. type NodeRoutingAssignment struct { + NetworkProvider *string Workload string Execution string ExecutionNodeID int @@ -1005,6 +1010,7 @@ func applySessionStreamStateLocked(s *Session, state SessionStreamState) { if state.TranscodeRouteSet { s.TranscodeNodeURL = state.TranscodeNodeURL s.TranscodeTransportID = state.TranscodeTransportID + s.RoutingNetworkProvider = state.RoutingNetworkProvider s.RoutingWorkload = state.RoutingWorkload s.RoutingExecution = state.RoutingExecution s.RoutingExecutionNodeID = state.RoutingExecutionNodeID @@ -1052,6 +1058,7 @@ func snapshotSessionStreamStateLocked(s *Session) SessionStreamState { TranscodeNodeURL: s.TranscodeNodeURL, TranscodeTransportID: s.TranscodeTransportID, TranscodeRouteSet: true, + RoutingNetworkProvider: s.RoutingNetworkProvider, RoutingWorkload: s.RoutingWorkload, RoutingExecution: s.RoutingExecution, RoutingExecutionNodeID: s.RoutingExecutionNodeID, @@ -1090,6 +1097,7 @@ func restoreSessionStreamStateLocked(s *Session, state SessionStreamState) { s.ToneMapMode = state.ToneMapMode s.TranscodeNodeURL = state.TranscodeNodeURL s.TranscodeTransportID = state.TranscodeTransportID + s.RoutingNetworkProvider = state.RoutingNetworkProvider s.RoutingWorkload = state.RoutingWorkload s.RoutingExecution = state.RoutingExecution s.RoutingExecutionNodeID = state.RoutingExecutionNodeID @@ -1250,6 +1258,7 @@ func (m *SessionManager) SetNodeRoutingAssignment(sessionID string, assignment N return ErrSessionNotFound } + s.RoutingNetworkProvider = assignment.NetworkProvider s.RoutingWorkload = assignment.Workload s.RoutingExecution = assignment.Execution s.RoutingExecutionNodeID = assignment.ExecutionNodeID diff --git a/internal/playback/session_test.go b/internal/playback/session_test.go index b49d77f78c..6890d32c19 100644 --- a/internal/playback/session_test.go +++ b/internal/playback/session_test.go @@ -766,17 +766,18 @@ func TestSessionReplacementAppliesAndRollsBackAtomically(t *testing.T) { t.Fatal(err) } if err := manager.UpdateStreamState(session.ID, playback.SessionStreamState{ - PlayMethod: playback.PlayDirect, - BasePlayMethod: playback.PlayDirect, - AudioTrackIndex: 0, - TranscodeRouteSet: true, - SubtitleTrackIndex: -1, - StreamBitrateKbps: 8_000, - SourceAudioChannels: 6, - TranscodeHWAccel: "qsv", - ToneMapMode: tonemap.ModeHardware, - TranscodeNodeURL: "http://old-node", - TranscodeTransportID: "old-transport", + PlayMethod: playback.PlayDirect, + BasePlayMethod: playback.PlayDirect, + AudioTrackIndex: 0, + TranscodeRouteSet: true, + SubtitleTrackIndex: -1, + StreamBitrateKbps: 8_000, + SourceAudioChannels: 6, + TranscodeHWAccel: "qsv", + ToneMapMode: tonemap.ModeHardware, + TranscodeNodeURL: "http://old-node", + TranscodeTransportID: "old-transport", + RoutingNetworkProvider: new("tailscale"), }); err != nil { t.Fatal(err) } @@ -784,18 +785,19 @@ func TestSessionReplacementAppliesAndRollsBackAtomically(t *testing.T) { rollback, err := manager.ApplyReplacement(session.ID, playback.SessionReplacement{ EffectiveMediaFileID: 84, StreamState: playback.SessionStreamState{ - PlayMethod: playback.PlayTranscode, - BasePlayMethod: playback.PlayTranscode, - AudioTrackIndex: 2, - TranscodeAudio: true, - TranscodeRouteSet: true, - SubtitleTrackIndex: 1, - StreamBitrateKbps: 3_500, - SourceAudioChannels: 8, - TranscodeHWAccel: "none", - ToneMapMode: tonemap.ModeSoftware, - TranscodeNodeURL: "http://new-node", - TranscodeTransportID: "new-transport", + PlayMethod: playback.PlayTranscode, + BasePlayMethod: playback.PlayTranscode, + AudioTrackIndex: 2, + TranscodeAudio: true, + TranscodeRouteSet: true, + SubtitleTrackIndex: 1, + StreamBitrateKbps: 3_500, + SourceAudioChannels: 8, + TranscodeHWAccel: "none", + ToneMapMode: tonemap.ModeSoftware, + TranscodeNodeURL: "http://new-node", + TranscodeTransportID: "new-transport", + RoutingNetworkProvider: new(""), }, PositionSeconds: &position, IsPaused: true, @@ -808,7 +810,7 @@ func TestSessionReplacementAppliesAndRollsBackAtomically(t *testing.T) { t.Fatal(err) } if replaced.MediaFileID != 84 || replaced.PlayMethod != playback.PlayTranscode || replaced.AudioTrackIndex != 2 || - replaced.SourceAudioChannels != 8 || + replaced.SourceAudioChannels != 8 || replaced.RoutingNetworkProvider == nil || *replaced.RoutingNetworkProvider != "" || replaced.TranscodeNodeURL != "http://new-node" || replaced.TranscodeHWAccel != "none" || replaced.ToneMapMode != "software" || replaced.Position != position || !replaced.IsPaused { t.Fatalf("replacement session = %#v", replaced) } @@ -820,7 +822,7 @@ func TestSessionReplacementAppliesAndRollsBackAtomically(t *testing.T) { t.Fatal(err) } if restored.MediaFileID != 42 || restored.PlayMethod != playback.PlayDirect || restored.AudioTrackIndex != 0 || - restored.SourceAudioChannels != 6 || + restored.SourceAudioChannels != 6 || restored.RoutingNetworkProvider == nil || *restored.RoutingNetworkProvider != "tailscale" || restored.TranscodeNodeURL != "http://old-node" || restored.TranscodeTransportID != "old-transport" || restored.TranscodeHWAccel != "qsv" || restored.ToneMapMode != "hardware" || restored.Position != 0 || restored.IsPaused { diff --git a/internal/playback/transcode_manager.go b/internal/playback/transcode_manager.go index 4ccc46dd5b..50a8855d5e 100644 --- a/internal/playback/transcode_manager.go +++ b/internal/playback/transcode_manager.go @@ -679,8 +679,10 @@ func (m *TranscodeManager) reconstructSession(ctx context.Context, sessionID str BasePlayMethod: method, TranscodeNodeURL: card.TranscodeNodeURL, TranscodeTransportID: card.TranscodeTransportID, + RoutingNetworkProvider: card.RoutingNetworkProvider, RoutingWorkload: card.RoutingWorkload, RoutingExecution: card.RoutingExecution, + RoutingExecutionNodeID: card.RoutingExecutionNodeID, RoutingEgress: card.RoutingEgress, RoutingEgressNodeID: card.RoutingEgressNodeID, AudioTrackIndex: card.AudioTrackIndex, diff --git a/internal/pluginhost/client.go b/internal/pluginhost/client.go index a7fc8aa6bb..cce02f13d1 100644 --- a/internal/pluginhost/client.go +++ b/internal/pluginhost/client.go @@ -21,6 +21,8 @@ var ( type Client struct { installationID int + startSeq uint64 + ingressToken string manifest *pluginv1.PluginManifest rpc *sdkruntime.Client capabilities map[string]*pluginv1.CapabilityDescriptor @@ -29,6 +31,8 @@ type Client struct { unhealthy bool } +func (c *Client) StartSeq() uint64 { return c.startSeq } + type MetadataProviderClient struct { client pluginv1.MetadataProviderClient timeout time.Duration @@ -85,7 +89,17 @@ type WatchSyncProviderClient struct { timeout time.Duration } -func newClient(installationID int, rpc *sdkruntime.Client, manifest *pluginv1.PluginManifest) *Client { +// NetworkAccessProviderClient drives one network_access_provider.v1 instance. +type NetworkAccessProviderClient struct { + ingressToken string + client pluginv1.NetworkAccessProviderClient + timeout time.Duration +} + +// IngressToken identifies the process that answers these provider RPCs. +func (c *NetworkAccessProviderClient) IngressToken() string { return c.ingressToken } + +func newClient(installationID int, rpc *sdkruntime.Client, manifest *pluginv1.PluginManifest, startSeq uint64) *Client { capabilities := make(map[string]*pluginv1.CapabilityDescriptor, len(manifest.GetCapabilities())) for _, capability := range manifest.GetCapabilities() { capabilities[capabilityKey(capability.GetType(), capability.GetId())] = capability @@ -93,6 +107,7 @@ func newClient(installationID int, rpc *sdkruntime.Client, manifest *pluginv1.Pl return &Client{ installationID: installationID, + startSeq: startSeq, manifest: proto.Clone(manifest).(*pluginv1.PluginManifest), rpc: rpc, capabilities: capabilities, @@ -225,6 +240,19 @@ func (c *Client) WatchSyncProvider(capabilityID string) (*WatchSyncProviderClien }, nil } +// NetworkAccessProvider returns the typed client for the plugin's +// network_access_provider.v1 capability. +func (c *Client) NetworkAccessProvider(capabilityID string) (*NetworkAccessProviderClient, error) { + if err := c.requireCapability("network_access_provider.v1", capabilityID); err != nil { + return nil, err + } + return &NetworkAccessProviderClient{ + client: c.rpc.NetworkAccessProvider(), + ingressToken: c.ingressToken, + timeout: DefaultNetworkAccessTimeout, + }, nil +} + func (c *Client) markUnhealthy() { c.mu.Lock() defer c.mu.Unlock() @@ -444,6 +472,24 @@ func (c *WatchSyncProviderClient) ListRemoteState(ctx context.Context, req *plug return c.client.ListRemoteState(callCtx, req) } +func (c *NetworkAccessProviderClient) Connect(ctx context.Context, req *pluginv1.NetworkAccessConnectRequest) (*pluginv1.NetworkAccessStatus, error) { + callCtx, cancel := ensureDeadline(ctx, c.timeout) + defer cancel() + return c.client.Connect(callCtx, req) +} + +func (c *NetworkAccessProviderClient) Disconnect(ctx context.Context, req *pluginv1.NetworkAccessDisconnectRequest) (*pluginv1.NetworkAccessStatus, error) { + callCtx, cancel := ensureDeadline(ctx, c.timeout) + defer cancel() + return c.client.Disconnect(callCtx, req) +} + +func (c *NetworkAccessProviderClient) GetStatus(ctx context.Context, req *pluginv1.NetworkAccessGetStatusRequest) (*pluginv1.NetworkAccessStatus, error) { + callCtx, cancel := ensureDeadline(ctx, c.timeout) + defer cancel() + return c.client.GetStatus(callCtx, req) +} + func ensureDeadline(ctx context.Context, timeout time.Duration) (context.Context, context.CancelFunc) { if _, ok := ctx.Deadline(); ok { return ctx, func() {} diff --git a/internal/pluginhost/client_test.go b/internal/pluginhost/client_test.go index 67fe80b21f..2daa240878 100644 --- a/internal/pluginhost/client_test.go +++ b/internal/pluginhost/client_test.go @@ -25,7 +25,7 @@ func makeTestClient(t *testing.T, capabilities []*pluginv1.CapabilityDescriptor) rpc := sdkruntime.NewClient(conn) manifest := &pluginv1.PluginManifest{Capabilities: capabilities} - return newClient(0, rpc, manifest) + return newClient(0, rpc, manifest, 0) } func TestClient_ScheduledTask_CapabilityGate(t *testing.T) { diff --git a/internal/pluginhost/exit_watcher_test.go b/internal/pluginhost/exit_watcher_test.go new file mode 100644 index 0000000000..2c5e6206f5 --- /dev/null +++ b/internal/pluginhost/exit_watcher_test.go @@ -0,0 +1,176 @@ +package pluginhost_test + +import ( + "context" + "errors" + "os" + "os/exec" + "path/filepath" + "sync" + "testing" + "time" + + "github.com/hashicorp/go-hclog" + "google.golang.org/grpc/codes" + "google.golang.org/grpc/status" + "google.golang.org/protobuf/encoding/protojson" + + pluginv1 "github.com/Silo-Server/silo-plugin-sdk/pkg/pluginproto/silo/plugin/v1" + + "github.com/Silo-Server/silo-server/internal/pluginhost" +) + +// buildExitingPlugin compiles testdata/exitingplugin into a temp dir and +// returns the binary path with the manifest the binary reports (checksum +// included), which is what Host.Start compares against the live manifest. +func buildExitingPlugin(t *testing.T) (string, *pluginv1.PluginManifest) { + t.Helper() + bin := filepath.Join(t.TempDir(), "exitingplugin") + build := exec.Command("go", "build", "-o", bin, "./testdata/exitingplugin") + if out, err := build.CombinedOutput(); err != nil { + t.Fatalf("build exitingplugin: %v\n%s", err, out) + } + raw, err := exec.Command(bin, "manifest").Output() + if err != nil { + t.Fatalf("read fixture manifest: %v", err) + } + manifest := &pluginv1.PluginManifest{} + if err := protojson.Unmarshal(raw, manifest); err != nil { + t.Fatalf("decode fixture manifest: %v", err) + } + return bin, manifest +} + +func TestHostExitWatcherReportsCrash(t *testing.T) { + bin, manifest := buildExitingPlugin(t) + exitFile := filepath.Join(t.TempDir(), "exit-now") + t.Setenv("SILO_TEST_PLUGIN_EXIT_FILE", exitFile) + + host := pluginhost.NewHost(pluginhost.Config{ + Logger: hclog.NewNullLogger(), + ExitCheckInterval: 20 * time.Millisecond, + }) + exited := make(chan int, 4) + host.SetExitHandler(func(id int) { exited <- id }) + + ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second) + defer cancel() + client, err := host.Start(ctx, pluginhost.StartRequest{InstallationID: 7, BinaryPath: bin, Manifest: manifest}) + if err != nil { + t.Fatalf("host.Start: %v", err) + } + t.Cleanup(func() { _ = host.Stop(7) }) + + if _, err := host.Client(7); err != nil { + t.Fatalf("Client before crash: %v", err) + } + if err := os.WriteFile(exitFile, []byte("x"), 0o644); err != nil { + t.Fatal(err) + } + + select { + case id := <-exited: + if id != 7 { + t.Fatalf("exit handler got installation %d, want 7", id) + } + case <-ctx.Done(): + t.Fatal("exit handler was not called after the plugin process exited") + } + if _, err := host.Client(7); !errors.Is(err, pluginhost.ErrClientNotFound) { + t.Fatalf("Client after crash = %v, want ErrClientNotFound", err) + } + if _, err := client.MetadataProvider("exiting"); !errors.Is(err, pluginhost.ErrPluginUnhealthy) { + t.Fatalf("retained client after crash = %v, want ErrPluginUnhealthy", err) + } +} + +func TestHostExitWatcherIgnoresDeliberateStop(t *testing.T) { + bin, manifest := buildExitingPlugin(t) + exitFile := filepath.Join(t.TempDir(), "exit-now") + t.Setenv("SILO_TEST_PLUGIN_EXIT_FILE", exitFile) + + host := pluginhost.NewHost(pluginhost.Config{ + Logger: hclog.NewNullLogger(), + ExitCheckInterval: 20 * time.Millisecond, + }) + exited := make(chan int, 4) + host.SetExitHandler(func(id int) { exited <- id }) + + ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second) + defer cancel() + if _, err := host.Start(ctx, pluginhost.StartRequest{InstallationID: 8, BinaryPath: bin, Manifest: manifest}); err != nil { + t.Fatalf("host.Start(8): %v", err) + } + if err := host.Stop(8); err != nil { + t.Fatalf("host.Stop: %v", err) + } + + // Stop reaps the process synchronously, so a report for 8 after this + // point would be a crash misclassification. Rather than waiting a fixed + // time for nothing to happen, crash a second instance on the same host + // and use its report as the clock: by the time the watcher has noticed + // 9's exit (written after Stop returned), 8's watcher has had at least + // as many ticks to misfire. + if _, err := host.Start(ctx, pluginhost.StartRequest{InstallationID: 9, BinaryPath: bin, Manifest: manifest}); err != nil { + t.Fatalf("host.Start(9): %v", err) + } + t.Cleanup(func() { _ = host.Stop(9) }) + if err := os.WriteFile(exitFile, []byte("x"), 0o644); err != nil { + t.Fatal(err) + } + + for { + select { + case id := <-exited: + switch id { + case 9: + return + case 8: + t.Fatal("exit handler reported installation 8 after a deliberate Stop") + default: + t.Fatalf("exit handler reported unexpected installation %d", id) + } + case <-ctx.Done(): + t.Fatal("exit handler was not called for the crashed instance 9") + } + } +} + +func TestHostConcurrentStartsRetireEveryProcess(t *testing.T) { + bin, manifest := buildExitingPlugin(t) + host := pluginhost.NewHost(pluginhost.Config{}) + t.Cleanup(func() { _ = host.Shutdown(context.Background()) }) + clients := make(chan *pluginhost.Client, 8) + start := make(chan struct{}) + var wg sync.WaitGroup + for range 8 { + wg.Go(func() { + <-start + client, err := host.Start(t.Context(), pluginhost.StartRequest{InstallationID: 7, BinaryPath: bin, Manifest: manifest}) + if err != nil { + t.Error(err) + return + } + clients <- client + }) + } + close(start) + wg.Wait() + close(clients) + if err := host.Shutdown(t.Context()); err != nil { + t.Fatal(err) + } + for client := range clients { + provider, err := client.MetadataProvider("exiting") + if errors.Is(err, pluginhost.ErrPluginUnhealthy) { + continue + } + if err != nil { + t.Fatal(err) + } + _, err = provider.Search(t.Context(), &pluginv1.SearchMetadataRequest{}) + if code := status.Code(err); code != codes.Unavailable && code != codes.Canceled { + t.Errorf("superseded process still answers RPCs: %v", err) + } + } +} diff --git a/internal/pluginhost/handshake.go b/internal/pluginhost/handshake.go index 2843e16824..11313b0fad 100644 --- a/internal/pluginhost/handshake.go +++ b/internal/pluginhost/handshake.go @@ -11,6 +11,7 @@ import ( const ( DefaultHealthCheckInterval = 30 * time.Second DefaultHealthFailureLimit = 3 + DefaultExitCheckInterval = time.Second DefaultMetadataTimeout = 30 * time.Second DefaultMarkerProviderTimeout = 30 * time.Second DefaultAnalyzerTimeout = 5 * time.Minute @@ -29,6 +30,11 @@ const ( // DefaultRequestRouterTimeout bounds a single request_router RPC. // Fulfillment hits remote arr instances, so allow generous headroom. DefaultRequestRouterTimeout = 60 * time.Second + // DefaultNetworkAccessTimeout bounds one network access provider RPC. + // Connect returns the state reached so far and continues enrollment in + // the background, so an admin request never waits longer than this per + // host, matching the node force-reload acknowledgement window. + DefaultNetworkAccessTimeout = 10 * time.Second ) func HandshakeConfig() plugin.HandshakeConfig { diff --git a/internal/pluginhost/host.go b/internal/pluginhost/host.go index 4180ff20d4..16227a9114 100644 --- a/internal/pluginhost/host.go +++ b/internal/pluginhost/host.go @@ -2,10 +2,12 @@ package pluginhost import ( "context" + "errors" "fmt" "os" "os/exec" "sync" + "sync/atomic" "time" "github.com/hashicorp/go-hclog" @@ -25,6 +27,9 @@ type Config struct { Logger hclog.Logger HealthCheckInterval time.Duration HealthFailureLimit int + // ExitCheckInterval is how often the exit watcher polls the plugin + // process for an exit. Zero takes DefaultExitCheckInterval. + ExitCheckInterval time.Duration // EventPublisher receives events that plugins publish via RuntimeHost. // Typically silo's *events.Hub. When nil, plugins that try to @@ -42,6 +47,17 @@ type Config struct { // GlobalConfigSetter persists SetGlobalConfigEntry calls from plugins. When // nil, SetGlobalConfigEntry returns an error. GlobalConfigSetter GlobalConfigSetter + // HostInfo answers GetHostInfo. When nil, GetHostInfo is Unimplemented. + HostInfo HostInfoFunc + // InstanceState persists ReadInstanceState / WriteInstanceState for this + // process's host scope. When nil, both RPCs fail with FailedPrecondition. + InstanceState InstanceStateStore + // RuntimeHostForStart binds host identity and state together for each launch. + RuntimeHostForStart func(context.Context) (HostInfoFunc, InstanceStateStore, error) + // NetworkAccess issues the ingress token for every network access provider + // the host starts, revokes it when the process stops, and receives status + // pushes. When nil, providers get no token and pushes are dropped. + NetworkAccess NetworkAccessBroker } type StartRequest struct { @@ -55,24 +71,49 @@ type Host struct { logger hclog.Logger healthCheckInterval time.Duration healthFailureLimit int - - eventPublisher EventPublisher - libraryLister LibraryLister - catalogPresence CatalogPresenceLookup - installedPlugins InstalledPluginLister - globalConfigSetter GlobalConfigSetter + exitCheckInterval time.Duration + + // exitHandler is told the installation id of a plugin whose process went + // away on its own (exit or failed health), never one the host stopped on + // purpose. The resident supervisor in internal/plugins uses it to + // schedule a restart. + exitMu sync.RWMutex + exitHandler func(installationID int) + + eventPublisher EventPublisher + libraryLister LibraryLister + catalogPresence CatalogPresenceLookup + installedPlugins InstalledPluginLister + globalConfigSetter GlobalConfigSetter + hostInfo HostInfoFunc + instanceState InstanceStateStore + runtimeHostForStart func(context.Context) (HostInfoFunc, InstanceStateStore, error) + networkAccess NetworkAccessBroker mu sync.RWMutex instances map[int]*instance + starting map[int]chan struct{} + startSeq atomic.Uint64 } type instance struct { - process *plugin.Client - command *exec.Cmd - usageOnce sync.Once - protocol plugin.ClientProtocol - client *Client - cancelHealth context.CancelFunc + process *plugin.Client + command *exec.Cmd + usageOnce sync.Once + protocol plugin.ClientProtocol + client *Client + // installationID lets stopInstance revoke the ingress token of a network + // access provider without a map lookup. + installationID int + // provider is the network_access_provider.v1 slug, empty otherwise, and + // ingressToken the token issued to this process instance. + provider string + ingressToken string + // cancelMonitors stops the health probe and the exit watcher. Stop, + // Shutdown and a replacing Start cancel it before tearing the process + // down, which is how the monitors tell a deliberate stop from a crash. + cancelMonitors context.CancelFunc + retireOnce sync.Once } func NewHost(cfg Config) *Host { @@ -88,21 +129,32 @@ func NewHost(cfg Config) *Host { if failureLimit <= 0 { failureLimit = DefaultHealthFailureLimit } + exitInterval := cfg.ExitCheckInterval + if exitInterval <= 0 { + exitInterval = DefaultExitCheckInterval + } return &Host{ logger: logger, healthCheckInterval: interval, healthFailureLimit: failureLimit, + exitCheckInterval: exitInterval, eventPublisher: cfg.EventPublisher, libraryLister: cfg.LibraryLister, catalogPresence: cfg.CatalogPresence, installedPlugins: cfg.InstalledPlugins, globalConfigSetter: cfg.GlobalConfigSetter, + hostInfo: cfg.HostInfo, + instanceState: cfg.InstanceState, + runtimeHostForStart: cfg.RuntimeHostForStart, + networkAccess: cfg.NetworkAccess, instances: make(map[int]*instance), + starting: make(map[int]chan struct{}), } } func (h *Host) Start(ctx context.Context, req StartRequest) (*Client, error) { + startSeq := h.startSeq.Add(1) if req.InstallationID == 0 { return nil, fmt.Errorf("installation id is required") } @@ -112,6 +164,22 @@ func (h *Host) Start(ctx context.Context, req StartRequest) (*Client, error) { if req.Manifest == nil { return nil, fmt.Errorf("plugin manifest is required") } + // Serialize replacements per installation while allowing unrelated plugins + // to launch independently. Failed-uninstall recovery can call Start + // directly while the resident supervisor is already launching it. + if err := h.acquireStart(ctx, req.InstallationID); err != nil { + return nil, err + } + defer h.releaseStart(req.InstallationID) + + hostInfo, instanceState := h.hostInfo, h.instanceState + if h.runtimeHostForStart != nil { + var err error + hostInfo, instanceState, err = h.runtimeHostForStart(ctx) + if err != nil { + return nil, fmt.Errorf("bind process host identity: %w", err) + } + } h.mu.Lock() if existing, ok := h.instances[req.InstallationID]; ok { @@ -122,6 +190,22 @@ func (h *Host) Start(ctx context.Context, req StartRequest) (*Client, error) { } h.mu.Unlock() + // A network access provider gets a fresh ingress token for this process + // instance before it can ask GetHostInfo for it. Config test-runs + // (negative installation ids) are not persisted installations and get + // none. + provider, isProvider := NetworkAccessProviderSlug(req.Manifest) + var ingressToken string + if !isProvider || req.InstallationID <= 0 || h.networkAccess == nil { + provider = "" + } else { + token, err := h.networkAccess.Issue(req.InstallationID, provider) + if err != nil { + return nil, fmt.Errorf("issue ingress token: %w", err) + } + ingressToken = token + } + command := exec.Command(req.BinaryPath) process := plugin.NewClient(&plugin.ClientConfig{ HandshakeConfig: HandshakeConfig(), @@ -143,6 +227,9 @@ func (h *Host) Start(ctx context.Context, req StartRequest) (*Client, error) { // returns. Only then is ProcessState safe to read. process.Kill() processmetrics.Record(processmetrics.Plugin, command.ProcessState, nil, nil) + if provider != "" { + h.networkAccess.Revoke(req.InstallationID, ingressToken) + } } }() @@ -166,7 +253,7 @@ func (h *Host) Start(ctx context.Context, req StartRequest) (*Client, error) { return nil, fmt.Errorf("unexpected plugin runtime client type %T", rawClient) } - if err := h.bindRuntimeHost(ctx, rpcClient, req.Manifest.GetPluginId(), req.InstallationID); err != nil { + if err := h.bindRuntimeHost(ctx, rpcClient, req.Manifest.GetPluginId(), req.InstallationID, provider, ingressToken, hostInfo, instanceState); err != nil { _ = protocol.Close() process.Kill() return nil, fmt.Errorf("bind runtime host: %w", err) @@ -203,15 +290,19 @@ func (h *Host) Start(ctx context.Context, req StartRequest) (*Client, error) { return nil, fmt.Errorf("configure plugin runtime: %w", err) } - client := newClient(req.InstallationID, rpcClient, liveManifestResponse.GetManifest()) + client := newClient(req.InstallationID, rpcClient, liveManifestResponse.GetManifest(), startSeq) + client.ingressToken = ingressToken - healthCtx, healthCancel := context.WithCancel(context.Background()) + monitorCtx, monitorCancel := context.WithCancel(context.Background()) instance := &instance{ - process: process, - command: command, - protocol: protocol, - client: client, - cancelHealth: healthCancel, + process: process, + command: command, + protocol: protocol, + client: client, + installationID: req.InstallationID, + provider: provider, + ingressToken: ingressToken, + cancelMonitors: monitorCancel, } h.mu.Lock() @@ -219,11 +310,50 @@ func (h *Host) Start(ctx context.Context, req StartRequest) (*Client, error) { h.mu.Unlock() retained = true - go h.monitorHealth(healthCtx, req.InstallationID, instance) + go h.monitorHealth(monitorCtx, req.InstallationID, instance) + go h.watchExit(monitorCtx, req.InstallationID, instance) return client, nil } +func (h *Host) acquireStart(ctx context.Context, id int) error { + for { + if err := ctx.Err(); err != nil { + return err + } + h.mu.Lock() + pending, busy := h.starting[id] + if !busy { + h.starting[id] = make(chan struct{}) + h.mu.Unlock() + return nil + } + h.mu.Unlock() + select { + case <-pending: + case <-ctx.Done(): + return ctx.Err() + } + } +} + +func (h *Host) releaseStart(id int) { + h.mu.Lock() + defer h.mu.Unlock() + close(h.starting[id]) + delete(h.starting, id) +} + +// SetExitHandler registers the callback told about plugin processes that +// stop on their own: the process exited, or the gRPC health probe gave up on +// it. Deliberate stops (Stop, Shutdown, a replacing Start) are not reported. +// The callback runs on the monitor goroutine and must not block. +func (h *Host) SetExitHandler(handler func(installationID int)) { + h.exitMu.Lock() + h.exitHandler = handler + h.exitMu.Unlock() +} + func (h *Host) Client(installationID int) (*Client, error) { h.mu.RLock() instance, ok := h.instances[installationID] @@ -242,6 +372,8 @@ func (h *Host) Client(installationID int) (*Client, error) { return instance.client, nil } +func (h *Host) NextStartSeq() uint64 { return h.startSeq.Load() } + func (h *Host) Stop(installationID int) error { h.mu.Lock() instance, ok := h.instances[installationID] @@ -307,20 +439,73 @@ func (h *Host) monitorHealth(ctx context.Context, installationID int, instance * continue } - instance.client.markUnhealthy() - h.logger.Error("plugin health check failed", "installation_id", installationID, "error", err) - h.stopInstance(instance) + h.retireInstance(ctx, installationID, instance, fmt.Errorf("plugin health check failed: %w", err)) return } } } +// watchExit polls the plugin process so a crash is noticed within one +// interval instead of at the next failed health probe (which needs several +// misses at DefaultHealthCheckInterval). +func (h *Host) watchExit(ctx context.Context, installationID int, instance *instance) { + ticker := time.NewTicker(h.exitCheckInterval) + defer ticker.Stop() + + for { + select { + case <-ctx.Done(): + return + case <-ticker.C: + if !instance.process.Exited() { + continue + } + h.retireInstance(ctx, installationID, instance, errors.New("plugin process exited")) + return + } + } +} + +// retireInstance marks a monitored instance dead: it is unhealthy for +// callers holding the *Client, is removed from the host so the next Client +// lookup reports ErrClientNotFound, and its process is reaped. The exit +// handler is told only when this instance is still the one the host maps +// for the installation: a canceled ctx or a different (or missing) mapped +// instance means Stop, Shutdown or a replacing Start owns the teardown, so +// a crash noticed while one of those is in flight is not reported twice. +func (h *Host) retireInstance(ctx context.Context, installationID int, instance *instance, cause error) { + instance.retireOnce.Do(func() { + if ctx.Err() != nil { + return + } + instance.client.markUnhealthy() + h.mu.Lock() + current, ok := h.instances[installationID] + if ok && current == instance { + delete(h.instances, installationID) + } + h.mu.Unlock() + if !ok || current != instance { + return + } + h.logger.Error("plugin instance retired", "installation_id", installationID, "error", cause) + h.stopInstance(instance) + + h.exitMu.RLock() + handler := h.exitHandler + h.exitMu.RUnlock() + if handler != nil { + handler(installationID) + } + }) +} + func (h *Host) stopInstance(instance *instance) { if instance == nil { return } - if instance.cancelHealth != nil { - instance.cancelHealth() + if instance.cancelMonitors != nil { + instance.cancelMonitors() } if instance.protocol != nil { _ = instance.protocol.Close() @@ -333,6 +518,11 @@ func (h *Host) stopInstance(instance *instance) { } }) } + // The token dies with the process: a stale one is refused until the next + // Start issues a replacement and the plugin re-reads GetHostInfo. + if instance.provider != "" && h.networkAccess != nil { + h.networkAccess.Revoke(instance.installationID, instance.ingressToken) + } } // bindRuntimeHost stands up a RuntimeHost gRPC server on a fresh broker @@ -341,8 +531,9 @@ func (h *Host) stopInstance(instance *instance) { // tears it down. // // Skipped when no RuntimeHost services are configured. -func (h *Host) bindRuntimeHost(ctx context.Context, sdkClient *sdkruntime.Client, pluginID string, installationID int) error { - if h.eventPublisher == nil && h.libraryLister == nil && h.catalogPresence == nil && h.installedPlugins == nil && h.globalConfigSetter == nil { +func (h *Host) bindRuntimeHost(ctx context.Context, sdkClient *sdkruntime.Client, pluginID string, installationID int, provider, ingressToken string, hostInfo HostInfoFunc, instanceState InstanceStateStore) error { + if h.eventPublisher == nil && h.libraryLister == nil && h.catalogPresence == nil && h.installedPlugins == nil && h.globalConfigSetter == nil && + hostInfo == nil && instanceState == nil && h.networkAccess == nil { return nil } @@ -356,15 +547,21 @@ func (h *Host) bindRuntimeHost(ctx context.Context, sdkClient *sdkruntime.Client streamID := broker.NextId() go broker.AcceptAndServe(streamID, func(opts []grpc.ServerOption) *grpc.Server { s := grpc.NewServer(append(opts, grpc.ChainUnaryInterceptor(observePluginCallback))...) - srv := NewRuntimeHostServerWithServices( - h.eventPublisher, - h.libraryLister, - h.catalogPresence, - h.installedPlugins, - h.globalConfigSetter, - pluginID, - installationID, - ) + srv := NewRuntimeHostServerWithOptions(RuntimeHostOptions{ + Publisher: h.eventPublisher, + Libraries: h.libraryLister, + Catalog: h.catalogPresence, + InstalledPlugins: h.installedPlugins, + GlobalConfigSetter: h.globalConfigSetter, + HostInfo: hostInfo, + InstanceState: instanceState, + NetworkAccess: h.networkAccess, + Logger: h.logger, + PluginID: pluginID, + InstallationID: installationID, + NetworkAccessProvider: provider, + IngressToken: ingressToken, + }) pluginv1.RegisterRuntimeHostServer(s, srv) return s }) diff --git a/internal/pluginhost/host_info.go b/internal/pluginhost/host_info.go new file mode 100644 index 0000000000..6c455e5876 --- /dev/null +++ b/internal/pluginhost/host_info.go @@ -0,0 +1,146 @@ +package pluginhost + +import ( + "context" + "net" + "strings" + + pluginv1 "github.com/Silo-Server/silo-plugin-sdk/pkg/pluginproto/silo/plugin/v1" + "github.com/Silo-Server/silo-plugin-sdk/pkg/pluginsdk/capability" + + "github.com/Silo-Server/silo-server/internal/netaccess" +) + +// Host roles reported in GetHostInfo.host_role. +const ( + HostRoleAPI = "api" + HostRoleProxy = "proxy" +) + +// Listener names reported in GetHostInfo.listeners. +const ( + ListenerAPI = "api" + ListenerJellyfin = "jellyfin" + ListenerABS = "abs" +) + +// Default overlay ports a network access provider exposes each listener on. +const ( + DefaultPortAPI = 443 + DefaultPortJellyfin = 8096 + DefaultPortABS = 13378 +) + +// HostListener is one local listener a network access provider should expose +// on the overlay: its loopback dial address and the port to expose it on. +type HostListener struct { + Name string + Address string + DefaultPort int +} + +// HostInfo is what GetHostInfo reports about the process hosting the plugin. +// Public URLs and listeners can change with a config reload, so the host +// supplies a HostInfoFunc that builds it per call. +type HostInfo struct { + // PublicBaseURL is server.public_url without a trailing slash; empty when + // unset. + PublicBaseURL string + // PluginContentPrefix is the API path under which plugin HTTP routes are + // proxied; GetHostInfo appends /plugins/ to it. + PluginContentPrefix string + Role string + Name string + NodeID int64 + Listeners []HostListener +} + +// HostInfoFunc answers GetHostInfo for the current process. +type HostInfoFunc func(ctx context.Context) (HostInfo, error) + +// NetworkAccessBroker is the netaccess side of a resident network access +// provider's lifetime: Host.Start issues the ingress token before the plugin +// can ask for it, stopping the process revokes it, and status pushes land in +// the host's status cache. *netaccess.Broker implements it. +type NetworkAccessBroker interface { + Issue(installationID int, provider string) (string, error) + // Revoke drops the token only while it is still the installation's + // current one, so a late stop cannot revoke a replacement's token. + Revoke(installationID int, token string) + IngressToken(installationID int) (string, bool) + // ReportFor records a status push from the process holding token. A push + // from a process whose token was already revoked (it crashed, was + // stopped, or was replaced while the RPC was in flight) is dropped, so a + // dead instance can never write a stale origin back over a fresh one. + ReportFor(installationID int, token string, status netaccess.Status) (previous netaccess.Status, changed bool, accepted bool) +} + +// NetworkAccessProviderSlug returns the provider slug a manifest declares +// through network_access_provider.v1 (the typed descriptor's provider, or the +// capability id when the descriptor is absent) and whether it declares one. +func NetworkAccessProviderSlug(manifest *pluginv1.PluginManifest) (string, bool) { + descriptor, slug := NetworkAccessProviderCapability(manifest) + return slug, descriptor != nil +} + +// NetworkAccessProviderCapability returns the manifest's +// network_access_provider.v1 descriptor and the provider slug it declares +// (the typed descriptor's provider, or the capability id when the descriptor +// is absent). The descriptor is nil when the manifest declares none; a +// manifest declares at most one, so the first wins. +func NetworkAccessProviderCapability(manifest *pluginv1.PluginManifest) (*pluginv1.CapabilityDescriptor, string) { + for _, descriptor := range manifest.GetCapabilities() { + if descriptor.GetType() != capability.NetworkAccessProvider { + continue + } + if slug := strings.TrimSpace(descriptor.GetNetworkAccessProvider().GetProvider()); slug != "" { + return descriptor, slug + } + if id := strings.TrimSpace(descriptor.GetId()); id != "" { + return descriptor, id + } + } + return nil, "" +} + +// NetworkAccessStatusFromProto converts a provider's reported status into the +// host's SDK-free form. Both the push (ReportNetworkAccessStatus) and the +// on-demand reads the admin API makes go through it so the status cache and +// the API agree on every field. +func NetworkAccessStatusFromProto(installationID int, provider string, reported *pluginv1.NetworkAccessStatus) netaccess.Status { + entry := netaccess.Status{ + InstallationID: installationID, + Provider: provider, + State: strings.TrimSpace(reported.GetState()), + Hostname: reported.GetHostname(), + Origin: reported.GetOrigin(), + Addresses: append([]string(nil), reported.GetAddresses()...), + AuthURL: reported.GetAuthUrl(), + Error: reported.GetError(), + ProviderVersion: reported.GetProviderVersion(), + DesiredConnected: reported.GetDesiredConnected(), + } + for _, listener := range reported.GetListeners() { + entry.Listeners = append(entry.Listeners, netaccess.Listener{Name: listener.GetName(), Origin: listener.GetOrigin()}) + } + return entry +} + +// LoopbackDialAddress turns a listen address into the host:port a plugin in +// the same process namespace dials: a wildcard or empty host becomes +// 127.0.0.1, a concrete host is kept. +func LoopbackDialAddress(listen string) string { + listen = strings.TrimSpace(listen) + if listen == "" { + return "" + } + host, port, err := net.SplitHostPort(listen) + if err != nil { + return listen + } + switch host { + case "", "0.0.0.0", "::", "[::]": + host = "127.0.0.1" + } + return net.JoinHostPort(host, port) +} diff --git a/internal/pluginhost/host_info_test.go b/internal/pluginhost/host_info_test.go new file mode 100644 index 0000000000..ec17f673ec --- /dev/null +++ b/internal/pluginhost/host_info_test.go @@ -0,0 +1,266 @@ +package pluginhost_test + +import ( + "context" + "errors" + "testing" + + "github.com/hashicorp/go-hclog" + "google.golang.org/grpc/codes" + "google.golang.org/grpc/status" + + pluginv1 "github.com/Silo-Server/silo-plugin-sdk/pkg/pluginproto/silo/plugin/v1" + "github.com/Silo-Server/silo-plugin-sdk/pkg/pluginsdk/capability" + + "github.com/Silo-Server/silo-server/internal/netaccess" + "github.com/Silo-Server/silo-server/internal/pluginhost" +) + +type fakeInstanceState struct { + values map[string][]byte + scopeOf map[string]int + err error +} + +func (f *fakeInstanceState) ReadInstanceState(_ context.Context, installationID int, key string) ([]byte, bool, error) { + if f.err != nil { + return nil, false, f.err + } + v, ok := f.values[key] + if ok && f.scopeOf[key] != installationID { + return nil, false, nil + } + return v, ok, nil +} + +func (f *fakeInstanceState) WriteInstanceState(_ context.Context, installationID int, key string, value []byte) error { + if f.err != nil { + return f.err + } + if f.values == nil { + f.values = map[string][]byte{} + f.scopeOf = map[string]int{} + } + f.values[key] = value + f.scopeOf[key] = installationID + return nil +} + +func apiHostInfo(listeners ...pluginhost.HostListener) pluginhost.HostInfoFunc { + return func(context.Context) (pluginhost.HostInfo, error) { + return pluginhost.HostInfo{ + PublicBaseURL: "https://silo.example/", + PluginContentPrefix: "/api/v2/plugin-content", + Role: pluginhost.HostRoleAPI, + Name: "Living Room", + Listeners: listeners, + }, nil + } +} + +func TestRuntimeHostServer_GetHostInfo_ReportsListenersAndIngressToken(t *testing.T) { + broker := netaccess.NewBroker() + token, err := broker.Issue(9, "tailscale") + if err != nil { + t.Fatal(err) + } + srv := pluginhost.NewRuntimeHostServerWithOptions(pluginhost.RuntimeHostOptions{ + HostInfo: apiHostInfo( + pluginhost.HostListener{Name: pluginhost.ListenerAPI, Address: "127.0.0.1:8080", DefaultPort: pluginhost.DefaultPortAPI}, + pluginhost.HostListener{Name: pluginhost.ListenerJellyfin, Address: "127.0.0.1:8096", DefaultPort: pluginhost.DefaultPortJellyfin}, + pluginhost.HostListener{Name: pluginhost.ListenerABS, Address: "127.0.0.1:13378", DefaultPort: pluginhost.DefaultPortABS}, + ), + NetworkAccess: broker, + Logger: hclog.NewNullLogger(), + PluginID: "silo.tailscale", + InstallationID: 9, + NetworkAccessProvider: "tailscale", + }) + + resp, err := srv.GetHostInfo(context.Background(), &pluginv1.GetHostInfoRequest{}) + if err != nil { + t.Fatalf("GetHostInfo: %v", err) + } + if resp.GetHostRole() != "api" || resp.GetHostName() != "Living Room" || resp.GetNodeId() != 0 { + t.Fatalf("role/name/node = %q/%q/%d", resp.GetHostRole(), resp.GetHostName(), resp.GetNodeId()) + } + if resp.GetPublicBaseUrl() != "https://silo.example" { + t.Fatalf("public_base_url = %q", resp.GetPublicBaseUrl()) + } + if resp.GetInternalBaseUrl() != "http://127.0.0.1:8080" { + t.Fatalf("internal_base_url = %q", resp.GetInternalBaseUrl()) + } + if resp.GetPluginProxyBaseUrl() != "https://silo.example/api/v2/plugin-content/plugins/9" { + t.Fatalf("plugin_proxy_base_url = %q", resp.GetPluginProxyBaseUrl()) + } + if resp.GetIngressToken() != token { + t.Fatalf("ingress_token = %q, want the registry's token", resp.GetIngressToken()) + } + want := []struct { + name, address string + port int32 + }{{"api", "127.0.0.1:8080", 443}, {"jellyfin", "127.0.0.1:8096", 8096}, {"abs", "127.0.0.1:13378", 13378}} + if len(resp.GetListeners()) != len(want) { + t.Fatalf("listeners = %v", resp.GetListeners()) + } + for i, w := range want { + got := resp.GetListeners()[i] + if got.GetName() != w.name || got.GetAddress() != w.address || got.GetDefaultPort() != w.port { + t.Fatalf("listener %d = %v, want %+v", i, got, w) + } + } + if ingress, ok := broker.Registry.Lookup(resp.GetIngressToken()); !ok || ingress.Provider != "tailscale" || ingress.InstallationID != 9 { + t.Fatalf("token does not resolve: %+v %v", ingress, ok) + } +} + +func TestRuntimeHostServer_GetHostInfo_NoTokenForNonProvidersOrTestRuns(t *testing.T) { + broker := netaccess.NewBroker() + if _, err := broker.Issue(9, "tailscale"); err != nil { + t.Fatal(err) + } + for name, opts := range map[string]pluginhost.RuntimeHostOptions{ + "non-provider": {HostInfo: apiHostInfo(), NetworkAccess: broker, InstallationID: 9}, + "config test-run": {HostInfo: apiHostInfo(), NetworkAccess: broker, InstallationID: -1, NetworkAccessProvider: "tailscale"}, + } { + t.Run(name, func(t *testing.T) { + resp, err := pluginhost.NewRuntimeHostServerWithOptions(opts).GetHostInfo(context.Background(), &pluginv1.GetHostInfoRequest{}) + if err != nil { + t.Fatal(err) + } + if resp.GetIngressToken() != "" { + t.Fatalf("ingress token leaked: %q", resp.GetIngressToken()) + } + }) + } + _, err := pluginhost.NewRuntimeHostServerWithOptions(pluginhost.RuntimeHostOptions{}).GetHostInfo(context.Background(), &pluginv1.GetHostInfoRequest{}) + if status.Code(err) != codes.Unimplemented { + t.Fatalf("no host info configured: %v", err) + } +} + +func TestRuntimeHostServer_InstanceState_RoundTripAndLimits(t *testing.T) { + state := &fakeInstanceState{} + srv := pluginhost.NewRuntimeHostServerWithOptions(pluginhost.RuntimeHostOptions{InstanceState: state, InstallationID: 4}) + ctx := context.Background() + + read, err := srv.ReadInstanceState(ctx, &pluginv1.ReadInstanceStateRequest{Key: "_machinekey"}) + if err != nil || read.GetFound() { + t.Fatalf("read before write = %v, %v", read, err) + } + if _, err := srv.WriteInstanceState(ctx, &pluginv1.WriteInstanceStateRequest{Key: "_machinekey", Value: []byte("k")}); err != nil { + t.Fatalf("write: %v", err) + } + if state.scopeOf["_machinekey"] != 4 { + t.Fatalf("store called with installation %d, want 4", state.scopeOf["_machinekey"]) + } + read, err = srv.ReadInstanceState(ctx, &pluginv1.ReadInstanceStateRequest{Key: "_machinekey"}) + if err != nil || !read.GetFound() || string(read.GetValue()) != "k" { + t.Fatalf("read = %v, %v", read, err) + } + + longKey := string(make([]byte, pluginhost.InstanceStateMaxKeyBytes+1)) + if _, err := srv.WriteInstanceState(ctx, &pluginv1.WriteInstanceStateRequest{Key: longKey, Value: []byte("k")}); status.Code(err) != codes.InvalidArgument { + t.Fatalf("long key: %v", err) + } + if _, err := srv.WriteInstanceState(ctx, &pluginv1.WriteInstanceStateRequest{Key: "big", Value: make([]byte, pluginhost.InstanceStateMaxValueBytes+1)}); status.Code(err) != codes.InvalidArgument { + t.Fatalf("big value: %v", err) + } + state.err = pluginhost.ErrInstanceStateTooManyKeys + if _, err := srv.WriteInstanceState(ctx, &pluginv1.WriteInstanceStateRequest{Key: "k", Value: []byte("v")}); status.Code(err) != codes.ResourceExhausted { + t.Fatalf("too many keys: %v", err) + } + state.err = errors.New("db down") + if _, err := srv.ReadInstanceState(ctx, &pluginv1.ReadInstanceStateRequest{Key: "k"}); err == nil || status.Code(err) != codes.Unknown { + t.Fatalf("store error: %v", err) + } + + testRun := pluginhost.NewRuntimeHostServerWithOptions(pluginhost.RuntimeHostOptions{InstanceState: &fakeInstanceState{}, InstallationID: -3}) + if _, err := testRun.WriteInstanceState(ctx, &pluginv1.WriteInstanceStateRequest{Key: "k", Value: []byte("v")}); status.Code(err) != codes.FailedPrecondition { + t.Fatalf("test-run write: %v", err) + } + if _, err := testRun.ReadInstanceState(ctx, &pluginv1.ReadInstanceStateRequest{Key: "k"}); status.Code(err) != codes.FailedPrecondition { + t.Fatalf("test-run read: %v", err) + } +} + +func TestRuntimeHostServer_ReportNetworkAccessStatus(t *testing.T) { + broker := netaccess.NewBroker() + token, err := broker.Issue(9, "tailscale") + if err != nil { + t.Fatal(err) + } + srv := pluginhost.NewRuntimeHostServerWithOptions(pluginhost.RuntimeHostOptions{ + NetworkAccess: broker, InstallationID: 9, NetworkAccessProvider: "tailscale", IngressToken: token, Logger: hclog.NewNullLogger(), + }) + ctx := context.Background() + _, err = srv.ReportNetworkAccessStatus(ctx, &pluginv1.ReportNetworkAccessStatusRequest{Status: &pluginv1.NetworkAccessStatus{ + State: "connected", Hostname: "silo.tail1234.ts.net", Origin: "https://silo.tail1234.ts.net", + Listeners: []*pluginv1.NetworkAccessListener{{Name: "api", Origin: "https://silo.tail1234.ts.net"}, {Name: "abs", Origin: "https://silo.tail1234.ts.net:13378"}}, + AuthUrl: "https://login.tailscale.com/a/secret", ProviderVersion: "tsnet 1.0", DesiredConnected: true, + }}) + if err != nil { + t.Fatalf("report: %v", err) + } + got, ok := broker.Status.Get(9) + if !ok || got.Provider != "tailscale" || !got.Connected() || got.AuthURL != "https://login.tailscale.com/a/secret" || len(got.Listeners) != 2 || !got.DesiredConnected { + t.Fatalf("cached status = %+v, %v", got, ok) + } + origins := broker.Status.ConnectedOrigins() + if len(origins) != 2 || origins[0] != "https://silo.tail1234.ts.net" || origins[1] != "https://silo.tail1234.ts.net:13378" { + t.Fatalf("connected origins = %v", origins) + } + + if _, err := srv.ReportNetworkAccessStatus(ctx, &pluginv1.ReportNetworkAccessStatusRequest{}); status.Code(err) != codes.InvalidArgument { + t.Fatalf("nil status: %v", err) + } + // A push from a process whose token was revoked (it was stopped or + // replaced while the RPC was in flight) is dropped, so it cannot write a + // stale origin over the replacement's. + broker.Revoke(9, token) + if _, err := srv.ReportNetworkAccessStatus(ctx, &pluginv1.ReportNetworkAccessStatusRequest{Status: &pluginv1.NetworkAccessStatus{State: "connected", Origin: "https://stale.tail1234.ts.net"}}); err != nil { + t.Fatalf("revoked push must be accepted quietly: %v", err) + } + if _, ok := broker.Status.Get(9); ok { + t.Fatal("a revoked process repopulated the status cache") + } + replacement, _ := broker.Issue(9, "tailscale") + broker.Report(netaccess.Status{InstallationID: 9, Provider: "tailscale", State: "connected", Origin: "https://fresh.tail1234.ts.net"}) + if _, err := srv.ReportNetworkAccessStatus(ctx, &pluginv1.ReportNetworkAccessStatusRequest{Status: &pluginv1.NetworkAccessStatus{State: "connected", Origin: "https://stale.tail1234.ts.net"}}); err != nil { + t.Fatal(err) + } + if got, _ := broker.Status.Get(9); got.Origin != "https://fresh.tail1234.ts.net" { + t.Fatalf("old process overwrote the replacement's status: %+v", got) + } + _ = replacement + + notProvider := pluginhost.NewRuntimeHostServerWithOptions(pluginhost.RuntimeHostOptions{NetworkAccess: broker, InstallationID: 10}) + if _, err := notProvider.ReportNetworkAccessStatus(ctx, &pluginv1.ReportNetworkAccessStatusRequest{Status: &pluginv1.NetworkAccessStatus{State: "connected"}}); status.Code(err) != codes.PermissionDenied { + t.Fatalf("non-provider push: %v", err) + } + if _, ok := broker.Status.Get(10); ok { + t.Fatal("non-provider status was cached") + } +} + +func TestNetworkAccessProviderSlug(t *testing.T) { + manifest := &pluginv1.PluginManifest{Capabilities: []*pluginv1.CapabilityDescriptor{ + {Type: capability.MetadataProvider, Id: "tmdb"}, + {Type: capability.NetworkAccessProvider, Id: "stub", NetworkAccessProvider: &pluginv1.NetworkAccessProviderDescriptor{Provider: "tailscale"}}, + }} + if slug, ok := pluginhost.NetworkAccessProviderSlug(manifest); !ok || slug != "tailscale" { + t.Fatalf("slug = %q, %v", slug, ok) + } + manifest.Capabilities[1].NetworkAccessProvider = nil + if slug, ok := pluginhost.NetworkAccessProviderSlug(manifest); !ok || slug != "stub" { + t.Fatalf("fallback slug = %q, %v", slug, ok) + } + if _, ok := pluginhost.NetworkAccessProviderSlug(&pluginv1.PluginManifest{Capabilities: manifest.Capabilities[:1]}); ok { + t.Fatal("non-provider manifest reported a slug") + } + for in, want := range map[string]string{":8080": "127.0.0.1:8080", "0.0.0.0:8080": "127.0.0.1:8080", "[::]:8080": "127.0.0.1:8080", "10.0.0.5:9000": "10.0.0.5:9000", "": ""} { + if got := pluginhost.LoopbackDialAddress(in); got != want { + t.Fatalf("LoopbackDialAddress(%q) = %q, want %q", in, got, want) + } + } +} diff --git a/internal/pluginhost/instance_state.go b/internal/pluginhost/instance_state.go new file mode 100644 index 0000000000..c7dc5b6ba6 --- /dev/null +++ b/internal/pluginhost/instance_state.go @@ -0,0 +1,50 @@ +package pluginhost + +import ( + "context" + "errors" +) + +// Limits of the per-instance state store exposed through +// RuntimeHost.ReadInstanceState / WriteInstanceState. They size the store for +// tsnet's ipn.StateStore (node keys, profile state) and stop a plugin from +// using it as general storage. The store in internal/plugins enforces the +// same limits at the row. +const ( + InstanceStateMaxKeyBytes = 256 + InstanceStateMaxValueBytes = 256 << 10 + InstanceStateMaxKeys = 256 +) + +var ( + // ErrInstanceStateKeyTooLong reports a key over InstanceStateMaxKeyBytes. + ErrInstanceStateKeyTooLong = errors.New("instance state key exceeds 256 bytes") + // ErrInstanceStateValueTooLarge reports a value over InstanceStateMaxValueBytes. + ErrInstanceStateValueTooLarge = errors.New("instance state value exceeds 256 KiB") + // ErrInstanceStateTooManyKeys reports a write that would put a scope over + // InstanceStateMaxKeys. + ErrInstanceStateTooManyKeys = errors.New("instance state scope holds the maximum of 256 keys") + // ErrInstanceStateUnavailable reports an instance that has no state + // scope: config test-runs use negative installation ids and get none. + ErrInstanceStateUnavailable = errors.New("instance state is not available for this plugin instance") +) + +// InstanceStateStore persists a plugin instance's private state. The host +// scope (api host or one proxy node) is fixed by the implementation, so a +// plugin only ever names installation-relative keys. +type InstanceStateStore interface { + ReadInstanceState(ctx context.Context, installationID int, key string) (value []byte, found bool, err error) + WriteInstanceState(ctx context.Context, installationID int, key string, value []byte) error +} + +// ValidateInstanceStateKey applies the key limit shared by the RPC layer and +// the store. +func ValidateInstanceStateKey(key string) error { + if key == "" { + return errors.New("instance state key is required") + } + if len(key) > InstanceStateMaxKeyBytes { + return ErrInstanceStateKeyTooLong + } + return nil +} diff --git a/internal/pluginhost/metrics.go b/internal/pluginhost/metrics.go index ac7cc2439c..93e684b8e3 100644 --- a/internal/pluginhost/metrics.go +++ b/internal/pluginhost/metrics.go @@ -26,6 +26,7 @@ var pluginOperations = func() map[string]string { &pluginv1.RequestRouter_ServiceDesc, &pluginv1.EventConsumer_ServiceDesc, &pluginv1.AuthProvider_ServiceDesc, &pluginv1.HttpRoutes_ServiceDesc, &pluginv1.WatchSyncProvider_ServiceDesc, &pluginv1.WatchSyncDeviceAuthorizationService_ServiceDesc, + &pluginv1.NetworkAccessProvider_ServiceDesc, } { for _, method := range service.Methods { ops["/"+service.ServiceName+"/"+method.MethodName] = strings.TrimPrefix(service.ServiceName, "silo.plugin.v1.") + "." + method.MethodName diff --git a/internal/pluginhost/runtime_host_server.go b/internal/pluginhost/runtime_host_server.go index 3755918f5d..6543ba814f 100644 --- a/internal/pluginhost/runtime_host_server.go +++ b/internal/pluginhost/runtime_host_server.go @@ -3,11 +3,16 @@ package pluginhost import ( "context" "encoding/json" + "errors" "fmt" + "strconv" "strings" pluginv1 "github.com/Silo-Server/silo-plugin-sdk/pkg/pluginproto/silo/plugin/v1" + "github.com/hashicorp/go-hclog" "golang.org/x/time/rate" + "google.golang.org/grpc/codes" + "google.golang.org/grpc/status" "github.com/Silo-Server/silo-server/internal/events" ) @@ -101,6 +106,62 @@ type RuntimeHostServer struct { installedPlugins InstalledPluginLister configSetter GlobalConfigSetter installationID int + + hostInfo HostInfoFunc + instanceState InstanceStateStore + networkAccess NetworkAccessBroker + // ingressToken is the token issued to this process instance; pushes are + // accepted only while it is current. + ingressToken string + // provider is the network_access_provider.v1 slug from the plugin's + // manifest; empty for plugins that are not providers. + provider string + logger hclog.Logger +} + +// RuntimeHostOptions configures a RuntimeHostServer for one plugin instance. +type RuntimeHostOptions struct { + Publisher EventPublisher + Libraries LibraryLister + Catalog CatalogPresenceLookup + InstalledPlugins InstalledPluginLister + GlobalConfigSetter GlobalConfigSetter + HostInfo HostInfoFunc + InstanceState InstanceStateStore + NetworkAccess NetworkAccessBroker + Logger hclog.Logger + // EventRatePerSec caps PublishEvent; <= 0 takes DefaultPublishEventRatePerSec. + EventRatePerSec int + + PluginID string + InstallationID int + // NetworkAccessProvider is the provider slug the manifest declares, or + // empty. Only providers may push network access status. + NetworkAccessProvider string + // IngressToken is the token issued to this process instance. Status + // pushes are accepted only while it is still the installation's current + // token. + IngressToken string +} + +// NewRuntimeHostServerWithOptions builds the server the host binds for one +// plugin instance. +func NewRuntimeHostServerWithOptions(opts RuntimeHostOptions) *RuntimeHostServer { + s := NewRuntimeHostServerWithRate(opts.Publisher, opts.Libraries, opts.PluginID, opts.EventRatePerSec) + s.catalog = opts.Catalog + s.installedPlugins = opts.InstalledPlugins + s.configSetter = opts.GlobalConfigSetter + s.installationID = opts.InstallationID + s.hostInfo = opts.HostInfo + s.instanceState = opts.InstanceState + s.networkAccess = opts.NetworkAccess + s.provider = opts.NetworkAccessProvider + s.ingressToken = opts.IngressToken + s.logger = opts.Logger + if s.logger == nil { + s.logger = hclog.NewNullLogger() + } + return s } // NewRuntimeHostServer constructs a RuntimeHostServer bound to the given @@ -332,3 +393,127 @@ func (s *RuntimeHostServer) CheckMediaPresence(ctx context.Context, req *pluginv } return resp, nil } + +// GetHostInfo reports the hosting process: public and loopback base URLs, +// role, name, node id, the listeners a network access provider should expose +// and, for providers, the ingress token issued for this process instance. +func (s *RuntimeHostServer) GetHostInfo(ctx context.Context, _ *pluginv1.GetHostInfoRequest) (*pluginv1.GetHostInfoResponse, error) { + if s.hostInfo == nil { + return nil, status.Error(codes.Unimplemented, "host info is not configured") + } + info, err := s.hostInfo(ctx) + if err != nil { + return nil, fmt.Errorf("host info: %w", err) + } + resp := &pluginv1.GetHostInfoResponse{ + PublicBaseUrl: strings.TrimRight(info.PublicBaseURL, "/"), + HostRole: info.Role, + HostName: info.Name, + NodeId: info.NodeID, + } + if resp.PublicBaseUrl != "" && info.PluginContentPrefix != "" && s.installationID > 0 { + resp.PluginProxyBaseUrl = resp.PublicBaseUrl + info.PluginContentPrefix + "/plugins/" + strconv.Itoa(s.installationID) + } + for _, listener := range info.Listeners { + if listener.Name == ListenerAPI && listener.Address != "" { + resp.InternalBaseUrl = "http://" + listener.Address + } + resp.Listeners = append(resp.Listeners, &pluginv1.HostListener{ + Name: listener.Name, + Address: listener.Address, + DefaultPort: int32(listener.DefaultPort), + }) + } + if s.provider != "" && s.networkAccess != nil && s.installationID > 0 { + if token, ok := s.networkAccess.IngressToken(s.installationID); ok { + resp.IngressToken = token + } + } + return resp, nil +} + +// ReadInstanceState returns one key of the instance's private state. The +// scope is fixed by the store the host configured for this process. +func (s *RuntimeHostServer) ReadInstanceState(ctx context.Context, req *pluginv1.ReadInstanceStateRequest) (*pluginv1.ReadInstanceStateResponse, error) { + if err := ValidateInstanceStateKey(req.GetKey()); err != nil { + return nil, status.Error(codes.InvalidArgument, err.Error()) + } + if s.instanceState == nil || s.installationID <= 0 { + return nil, status.Error(codes.FailedPrecondition, ErrInstanceStateUnavailable.Error()) + } + value, found, err := s.instanceState.ReadInstanceState(ctx, s.installationID, req.GetKey()) + if err != nil { + return nil, instanceStateError(err) + } + if !found { + return &pluginv1.ReadInstanceStateResponse{}, nil + } + return &pluginv1.ReadInstanceStateResponse{Value: value, Found: true}, nil +} + +// WriteInstanceState stores one key of the instance's private state. +func (s *RuntimeHostServer) WriteInstanceState(ctx context.Context, req *pluginv1.WriteInstanceStateRequest) (*pluginv1.WriteInstanceStateResponse, error) { + if err := ValidateInstanceStateKey(req.GetKey()); err != nil { + return nil, status.Error(codes.InvalidArgument, err.Error()) + } + if len(req.GetValue()) > InstanceStateMaxValueBytes { + return nil, status.Error(codes.InvalidArgument, ErrInstanceStateValueTooLarge.Error()) + } + if s.instanceState == nil || s.installationID <= 0 { + return nil, status.Error(codes.FailedPrecondition, ErrInstanceStateUnavailable.Error()) + } + if err := s.instanceState.WriteInstanceState(ctx, s.installationID, req.GetKey(), req.GetValue()); err != nil { + return nil, instanceStateError(err) + } + return &pluginv1.WriteInstanceStateResponse{}, nil +} + +func instanceStateError(err error) error { + switch { + case errors.Is(err, ErrInstanceStateKeyTooLong), errors.Is(err, ErrInstanceStateValueTooLarge): + return status.Error(codes.InvalidArgument, err.Error()) + case errors.Is(err, ErrInstanceStateTooManyKeys): + return status.Error(codes.ResourceExhausted, err.Error()) + case errors.Is(err, ErrInstanceStateUnavailable): + return status.Error(codes.FailedPrecondition, err.Error()) + } + return fmt.Errorf("instance state: %w", err) +} + +// ReportNetworkAccessStatus records a provider's status push in the host's +// status cache. Only state transitions are logged; auth_url never is. +func (s *RuntimeHostServer) ReportNetworkAccessStatus(_ context.Context, req *pluginv1.ReportNetworkAccessStatusRequest) (*pluginv1.ReportNetworkAccessStatusResponse, error) { + if s.provider == "" { + return nil, status.Error(codes.PermissionDenied, "plugin does not declare network_access_provider.v1") + } + if req.GetStatus() == nil { + return nil, status.Error(codes.InvalidArgument, "status is required") + } + if s.networkAccess == nil || s.installationID <= 0 { + // Config test-runs and hosts without netaccess wiring accept and drop + // the push so a provider does not fail on it. + return &pluginv1.ReportNetworkAccessStatusResponse{}, nil + } + entry := NetworkAccessStatusFromProto(s.installationID, s.provider, req.GetStatus()) + previous, changed, accepted := s.networkAccess.ReportFor(s.installationID, s.ingressToken, entry) + if !accepted { + // This process's token was revoked while the push was in flight: + // the host already stopped or replaced it. Its status must not + // overwrite whatever the replacement reports. + s.logger.Debug("network access status from a revoked process instance dropped", + "plugin_id", s.pluginID, "installation_id", s.installationID, "provider", s.provider) + return &pluginv1.ReportNetworkAccessStatusResponse{}, nil + } + if changed { + s.logger.Info("network access provider state changed", + "plugin_id", s.pluginID, + "installation_id", s.installationID, + "provider", s.provider, + "from", previous.State, + "to", entry.State, + "origin", entry.Origin, + "error", entry.Error, + ) + } + return &pluginv1.ReportNetworkAccessStatusResponse{}, nil +} diff --git a/internal/pluginhost/testdata/exitingplugin/main.go b/internal/pluginhost/testdata/exitingplugin/main.go new file mode 100644 index 0000000000..9a4c79f191 --- /dev/null +++ b/internal/pluginhost/testdata/exitingplugin/main.go @@ -0,0 +1,36 @@ +// Command exitingplugin is a test fixture for the host's exit watcher: a +// minimal metadata provider that exits with status 3 as soon as the file +// named by SILO_TEST_PLUGIN_EXIT_FILE exists. +package main + +import ( + _ "embed" + "os" + "time" + + pluginv1 "github.com/Silo-Server/silo-plugin-sdk/pkg/pluginproto/silo/plugin/v1" + sdkruntime "github.com/Silo-Server/silo-plugin-sdk/pkg/pluginsdk/runtime" +) + +//go:embed manifest.json +var manifestJSON []byte + +type metadataServer struct { + pluginv1.UnimplementedMetadataProviderServer +} + +func main() { + if exitFile := os.Getenv("SILO_TEST_PLUGIN_EXIT_FILE"); exitFile != "" { + go func() { + for { + if _, err := os.Stat(exitFile); err == nil { + os.Exit(3) + } + time.Sleep(25 * time.Millisecond) + } + }() + } + sdkruntime.ServeManifest(manifestJSON, "0.1.0", sdkruntime.CapabilityServers{ + MetadataProvider: &metadataServer{}, + }) +} diff --git a/internal/pluginhost/testdata/exitingplugin/manifest.json b/internal/pluginhost/testdata/exitingplugin/manifest.json new file mode 100644 index 0000000000..0a730eca98 --- /dev/null +++ b/internal/pluginhost/testdata/exitingplugin/manifest.json @@ -0,0 +1,20 @@ +{ + "plugin_id": "silo.test.exiting", + "version": "0.1.0", + "checksum": "__CHECKSUM__", + "silo_api_version": "v1", + "supported_platforms": [ + {"os": "linux", "arch": "amd64"}, + {"os": "linux", "arch": "arm64"}, + {"os": "darwin", "arch": "amd64"}, + {"os": "darwin", "arch": "arm64"} + ], + "capabilities": [ + { + "type": "metadata_provider.v1", + "id": "exiting", + "display_name": "Exiting Fixture", + "description": "Test fixture that exits on command." + } + ] +} diff --git a/internal/plugins/archive_cache.go b/internal/plugins/archive_cache.go index 8e8519a285..38181ac0a8 100644 --- a/internal/plugins/archive_cache.go +++ b/internal/plugins/archive_cache.go @@ -10,6 +10,7 @@ import ( "log/slog" "os" "path/filepath" + "sync" pluginv1 "github.com/Silo-Server/silo-plugin-sdk/pkg/pluginproto/silo/plugin/v1" ) @@ -20,7 +21,15 @@ type archiveStore interface { } type ArchiveCache struct { + // mu protects rehydration and pruning from concurrent status and + // reconcile calls, including cache-hit validation during extraction. + mu sync.Mutex archives archiveStore + // root, when set, is this host's own plugin cache dir. Installations are + // then rehydrated under it (see LocalInstallPath) instead of at the + // install path the API server recorded, which on a proxy node names a + // directory on another machine. + root string } func NewArchiveCache(archives archiveStore) *ArchiveCache { @@ -30,13 +39,69 @@ func NewArchiveCache(archives archiveStore) *ArchiveCache { return &ArchiveCache{archives: archives} } +// NewArchiveCacheAt returns a cache that keeps its copies of every +// installation under root, the host's own plugin cache dir. A proxy node uses +// it: the API server's install paths are release identities to it, never +// paths it reads or writes. +func NewArchiveCacheAt(archives archiveStore, root string) *ArchiveCache { + cache := NewArchiveCache(archives) + if cache != nil { + cache.root = filepath.Clean(root) + if root == "" { + cache.root = "" + } + } + return cache +} + +// LocalInstallPath is where this host keeps the installation's binary. Without +// a root it is the recorded install path. With one it is +// ////plugin, where the install +// dir name is the unique directory the API server's installer created +// (install-XXXX), so a replaced binary — even at the same version — lands in +// a fresh directory here too and the resident supervisor's install-path +// change detection stays meaningful on this host. +func (c *ArchiveCache) LocalInstallPath(installation *Installation) string { + if installation == nil { + return "" + } + if c == nil || c.root == "" { + return installation.InstallPath + } + release := filepath.Base(filepath.Dir(installation.InstallPath)) + if release == "." || release == string(filepath.Separator) || release == "" { + release = "install" + } + return filepath.Join( + c.root, + sanitizeFilesystemSegment(installation.PluginID), + sanitizeFilesystemSegment(installation.Version), + sanitizeFilesystemSegment(release), + "plugin", + ) +} + +// Ensure makes the installation's files present at LocalInstallPath, +// rehydrating them from plugin_archives when they are missing, incomplete, or corrupted, +// and returns the installed manifest. func (c *ArchiveCache) Ensure(ctx context.Context, installation *Installation) (*pluginv1.PluginManifest, error) { if installation == nil { return nil, fmt.Errorf("plugin installation is required") } + c.mu.Lock() + defer c.mu.Unlock() + if err := ctx.Err(); err != nil { + return nil, err + } + binaryPath := c.LocalInstallPath(installation) - if manifest, err := LoadManifestFile(InstalledManifestPath(installation.InstallPath)); err == nil { - if err := installedFilesPresent(installation.InstallPath, manifest); err == nil { + if manifest, err := LoadManifestFile(InstalledManifestPath(binaryPath)); err == nil { + if err := validateInstalledFiles(binaryPath, manifest); err == nil { + if c.root != "" { + if err := checkBinaryPlatform(binaryPath); err != nil { + return nil, fmt.Errorf("cached plugin for installation %d: %w", installation.ID, err) + } + } return manifest, nil } } @@ -74,7 +139,7 @@ func (c *ArchiveCache) Ensure(ctx context.Context, installation *Installation) ( ) } - installDir := filepath.Dir(installation.InstallPath) + installDir := filepath.Dir(binaryPath) if err := os.RemoveAll(installDir); err != nil { return nil, fmt.Errorf("clear plugin cache dir %q: %w", installDir, err) } @@ -85,14 +150,61 @@ func (c *ArchiveCache) Ensure(ctx context.Context, installation *Installation) ( _ = os.RemoveAll(installDir) return nil, fmt.Errorf("extract stored plugin archive for installation %d: %w", installation.ID, err) } - if err := validateInstalledFiles(installation.InstallPath, manifest); err != nil { + if err := validateInstalledFiles(binaryPath, manifest); err != nil { _ = os.RemoveAll(installDir) return nil, fmt.Errorf("validate rehydrated plugin cache for installation %d: %w", installation.ID, err) } + if c.root != "" { + // Proxy caches may contain a binary installed by an API host on a + // different platform. Apply the same check to fresh and cached files. + if err := checkBinaryPlatform(binaryPath); err != nil { + _ = os.RemoveAll(installDir) + return nil, fmt.Errorf("rehydrate plugin for installation %d: %w", installation.ID, err) + } + } + c.pruneStaleReleases(ctx, installation, installDir) return manifest, nil } +// pruneStaleReleases drops this host's copies of the plugin's other releases +// once a new one is in place, the way the API server's installer removes the +// previous install dir on replace. Only runs under an own root: without one +// the directories belong to the installer. Best effort; a failure is logged +// and costs disk, not correctness. +func (c *ArchiveCache) pruneStaleReleases(ctx context.Context, installation *Installation, keepDir string) { + if c == nil || c.root == "" { + return + } + pluginRoot := filepath.Join(c.root, sanitizeFilesystemSegment(installation.PluginID)) + versions, err := os.ReadDir(pluginRoot) + if err != nil { + return + } + for _, version := range versions { + if !version.IsDir() { + continue + } + versionDir := filepath.Join(pluginRoot, version.Name()) + releases, err := os.ReadDir(versionDir) + if err != nil { + continue + } + for _, release := range releases { + releaseDir := filepath.Join(versionDir, release.Name()) + if !release.IsDir() || releaseDir == keepDir { + continue + } + if err := os.RemoveAll(releaseDir); err != nil { + slog.WarnContext(ctx, "remove stale plugin release from cache", "component", "plugins", + "installation_id", installation.ID, "path", releaseDir, "error", err) + } + } + // Drop the version dir once it is empty; a non-empty one stays. + _ = os.Remove(versionDir) + } +} + func (c *ArchiveCache) recoverLegacyBinaryArchive( ctx context.Context, installationID int, diff --git a/internal/plugins/archive_cache_test.go b/internal/plugins/archive_cache_test.go index b054471b5f..9a4cb0c0fc 100644 --- a/internal/plugins/archive_cache_test.go +++ b/internal/plugins/archive_cache_test.go @@ -8,11 +8,100 @@ import ( "errors" "os" "path/filepath" + "runtime" + "strings" + "sync" + "sync/atomic" "testing" "google.golang.org/protobuf/encoding/protojson" ) +func TestArchiveCacheRejectsForeignBinaryOnCacheHit(t *testing.T) { + // Use an ELF architecture different from the running test so the cache + // represents a volume moved from another proxy platform. + binaryData := make([]byte, 64) + copy(binaryData, []byte{0x7f, 'E', 'L', 'F', 2, 1, 1}) + binaryData[16], binaryData[18], binaryData[20], binaryData[52] = 2, 0xB7, 1, 64 + if runtime.GOARCH == "arm64" { + binaryData[18] = 0x3E // EM_X86_64 + } + manifest := testPluginManifest(t, "silo.metadb", "0.0.19") + checksum := sha256.Sum256(binaryData) + manifest.Checksum = hex.EncodeToString(checksum[:]) + manifestBytes, err := protojson.Marshal(manifest) + if err != nil { + t.Fatal(err) + } + installation := &Installation{ID: 42, PluginID: manifest.PluginId, Version: manifest.Version, InstallPath: "/api/plugins/install-cached/plugin"} + cache := NewArchiveCacheAt(&legacyArchiveStore{}, t.TempDir()) + path := cache.LocalInstallPath(installation) + if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(path, binaryData, 0o755); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(InstalledManifestPath(path), manifestBytes, 0o644); err != nil { + t.Fatal(err) + } + if _, err := cache.Ensure(context.Background(), installation); err == nil || !strings.Contains(err.Error(), "plugin binary is built for") { + t.Fatalf("cached foreign binary was not rejected: %v", err) + } +} + +func TestArchiveCacheRepairsCorruptedBinaryOnCacheHit(t *testing.T) { + for _, proxy := range []bool{false, true} { + name := "local" + if proxy { + name = "proxy" + } + t.Run(name, func(t *testing.T) { + binaryData := []byte("#!/bin/sh\nexit 0\n") + checksum := sha256.Sum256(binaryData) + manifest := testPluginManifest(t, "silo.metadb", "0.0.19") + manifest.Checksum = hex.EncodeToString(checksum[:]) + manifestBytes, err := protojson.Marshal(manifest) + if err != nil { + t.Fatal(err) + } + archiveBytes, err := buildBinaryPluginArchive(manifestBytes, binaryData) + if err != nil { + t.Fatal(err) + } + store := &legacyArchiveStore{archive: &InstallationArchive{ + InstallationID: 42, ManifestJSON: manifestBytes, Checksum: manifest.GetChecksum(), Bytes: archiveBytes, + }} + installation := &Installation{ + ID: 42, PluginID: manifest.GetPluginId(), Version: manifest.GetVersion(), + InstallPath: filepath.Join(t.TempDir(), "install-cached", "plugin"), + } + cache := NewArchiveCache(store) + if proxy { + cache = NewArchiveCacheAt(store, t.TempDir()) + } + if _, err := cache.Ensure(t.Context(), installation); err != nil { + t.Fatalf("populate cache: %v", err) + } + path := cache.LocalInstallPath(installation) + // Keep the file executable and the same size, but change its contents. + if err := os.WriteFile(path, []byte("#!/bin/sh\nexit 1\n"), 0o755); err != nil { + t.Fatal(err) + } + if _, err := cache.Ensure(t.Context(), installation); err != nil { + t.Fatalf("repair cache: %v", err) + } + got, err := os.ReadFile(path) + if err != nil { + t.Fatal(err) + } + if !bytes.Equal(got, binaryData) { + t.Fatalf("cached binary = %q, want stored binary %q", got, binaryData) + } + }) + } +} + func TestArchiveCacheEnsureRecoversLegacyBinaryArchive(t *testing.T) { ctx := context.Background() @@ -148,3 +237,137 @@ func (s *legacyArchiveStore) SaveArchive( s.savedArchiveBytes = append([]byte(nil), archiveBytes...) return nil } + +// A cache with its own root (a proxy node's) never touches the install path +// the API server recorded: it rehydrates under /// +// /plugin, reuses that copy on the next Ensure, and drops the +// previous release once a replacement is in place. +func TestArchiveCacheAtOwnRootRehydratesUnderItAndPrunesOldReleases(t *testing.T) { + ctx := context.Background() + root := filepath.Join(t.TempDir(), "proxy-cache") + apiInstallRoot := filepath.Join(t.TempDir(), "api-machine", "plugins", "silo.metadb") + + release := func(version, dir, script string) (*Installation, *legacyArchiveStore) { + t.Helper() + binaryData := []byte(script) + checksum := sha256.Sum256(binaryData) + manifest := testPluginManifest(t, "silo.metadb", version) + manifest.Checksum = hex.EncodeToString(checksum[:]) + manifestBytes, err := protojson.Marshal(manifest) + if err != nil { + t.Fatal(err) + } + archiveBytes, err := buildBinaryPluginArchive(manifestBytes, binaryData) + if err != nil { + t.Fatal(err) + } + store := &legacyArchiveStore{archive: &InstallationArchive{ + InstallationID: 42, ManifestJSON: manifestBytes, Checksum: manifest.GetChecksum(), Bytes: archiveBytes, + }} + return &Installation{ + ID: 42, + PluginID: "silo.metadb", + Version: version, + InstallPath: filepath.Join(apiInstallRoot, version, dir, "plugin"), + }, store + } + + first, firstStore := release("0.0.19", "install-aaaa", "#!/bin/sh\nexit 0\n") + cache := NewArchiveCacheAt(firstStore, root) + wantFirst := filepath.Join(root, "silo.metadb", "0.0.19", "install-aaaa", "plugin") + if got := cache.LocalInstallPath(first); got != wantFirst { + t.Fatalf("LocalInstallPath = %q, want %q", got, wantFirst) + } + if got := NewArchiveCache(firstStore).LocalInstallPath(first); got != first.InstallPath { + t.Fatalf("LocalInstallPath without a root = %q, want the recorded path %q", got, first.InstallPath) + } + + manifest, err := cache.Ensure(ctx, first) + if err != nil { + t.Fatalf("Ensure(first) returned error: %v", err) + } + if manifest.GetVersion() != "0.0.19" { + t.Fatalf("manifest version = %q", manifest.GetVersion()) + } + if _, err := os.Stat(wantFirst); err != nil { + t.Fatalf("binary was not rehydrated under the cache root: %v", err) + } + if _, err := os.Stat(apiInstallRoot); !errors.Is(err, os.ErrNotExist) { + t.Fatalf("the API server's install root was touched: %v", err) + } + + // The second Ensure finds the copy and does not read the archive again. + firstStore.archive = nil + if _, err := cache.Ensure(ctx, first); err != nil { + t.Fatalf("Ensure(first) again returned error: %v", err) + } + + second, secondStore := release("0.0.20", "install-bbbb", "#!/bin/sh\nexit 1\n") + cache = NewArchiveCacheAt(secondStore, root) + if _, err := cache.Ensure(ctx, second); err != nil { + t.Fatalf("Ensure(second) returned error: %v", err) + } + wantSecond := filepath.Join(root, "silo.metadb", "0.0.20", "install-bbbb", "plugin") + if _, err := os.Stat(wantSecond); err != nil { + t.Fatalf("second release was not rehydrated: %v", err) + } + if _, err := os.Stat(filepath.Dir(wantFirst)); !errors.Is(err, os.ErrNotExist) { + t.Fatalf("previous release still present after the replacement: %v", err) + } +} + +type blockedArchiveStore struct { + *legacyArchiveStore + calls atomic.Int32 + release chan struct{} +} + +func (s *blockedArchiveStore) GetArchive(ctx context.Context, id int) (*InstallationArchive, error) { + s.calls.Add(1) + select { + case <-s.release: + return s.legacyArchiveStore.GetArchive(ctx, id) + case <-ctx.Done(): + return nil, ctx.Err() + } +} + +func TestArchiveCacheSerializesRehydration(t *testing.T) { + binary := []byte("#!/bin/sh\nexit 0\n") + manifest := testPluginManifest(t, "silo.metadb", "0.0.19") + sum := sha256.Sum256(binary) + manifest.Checksum = hex.EncodeToString(sum[:]) + raw, err := protojson.Marshal(manifest) + if err != nil { + t.Fatal(err) + } + archive, err := buildBinaryPluginArchive(raw, binary) + if err != nil { + t.Fatal(err) + } + store := &blockedArchiveStore{legacyArchiveStore: &legacyArchiveStore{archive: &InstallationArchive{InstallationID: 42, ManifestJSON: raw, Checksum: manifest.Checksum, Bytes: archive}}, release: make(chan struct{})} + cache := NewArchiveCacheAt(store, t.TempDir()) + installation := &Installation{ID: 42, PluginID: manifest.PluginId, Version: manifest.Version, InstallPath: "/api/plugins/install-release/plugin"} + var wg sync.WaitGroup + ready := make(chan struct{}, 32) + for range 32 { + wg.Go(func() { + ready <- struct{}{} + if _, err := cache.Ensure(t.Context(), installation); err != nil { + t.Error(err) + } + }) + } + for range 32 { + <-ready + } + close(store.release) + wg.Wait() + if got := store.calls.Load(); got != 1 { + t.Errorf("archive loads=%d, want 1", got) + } + + if err := validateInstalledFiles(cache.LocalInstallPath(installation), manifest); err != nil { + t.Fatal(err) + } +} diff --git a/internal/plugins/auto_update.go b/internal/plugins/auto_update.go index a273cd15c5..4b782c4830 100644 --- a/internal/plugins/auto_update.go +++ b/internal/plugins/auto_update.go @@ -401,6 +401,12 @@ func (s *AutoUpdateService) autoUpdatePlugin(ctx context.Context, existing *Inst } } if err != nil { + // The old process was stopped above and the row still names the old + // release. A lazily started plugin comes back on its next RPC, but a + // resident only restarts on a lifecycle reconcile, and the run's + // end-of-pass notification is skipped when nothing was applied; fire + // it here so the supervisor relaunches the old release now. + s.notifyChanged(ctx) return fmt.Errorf("install updated plugin %s from %s to %s: %w", pluginID, oldVersion, newVersion, err) } diff --git a/internal/plugins/binary_platform.go b/internal/plugins/binary_platform.go new file mode 100644 index 0000000000..ba8cd01f95 --- /dev/null +++ b/internal/plugins/binary_platform.go @@ -0,0 +1,84 @@ +package plugins + +import ( + "debug/elf" + "debug/macho" + "debug/pe" + "fmt" + "runtime" +) + +const ( + archAMD64 = "amd64" + archARM64 = "arm64" + osLinux = "linux" +) + +// binaryPlatform reports the GOOS/GOARCH a plugin binary was built for, read +// from its executable header. ok is false for formats this host does not +// recognize; callers treat that as "unknown" rather than as a mismatch. +func binaryPlatform(path string) (goos, goarch string, ok bool) { + if f, err := elf.Open(path); err == nil { + defer func() { _ = f.Close() }() + switch f.Machine { + case elf.EM_X86_64: + goarch = archAMD64 + case elf.EM_AARCH64: + goarch = archARM64 + case elf.EM_386: + goarch = "386" + case elf.EM_ARM: + goarch = "arm" + case elf.EM_RISCV: + goarch = "riscv64" + default: + return "", "", false + } + // Go tags ELF binaries for linux and freebsd alike; linux is the + // only ELF platform Silo ships plugins for. + return osLinux, goarch, true + } + if f, err := macho.Open(path); err == nil { + defer func() { _ = f.Close() }() + switch f.Cpu { + case macho.CpuAmd64: + goarch = archAMD64 + case macho.CpuArm64: + goarch = archARM64 + default: + return "", "", false + } + return "darwin", goarch, true + } + if f, err := pe.Open(path); err == nil { + defer func() { _ = f.Close() }() + switch f.Machine { + case pe.IMAGE_FILE_MACHINE_AMD64: + goarch = archAMD64 + case pe.IMAGE_FILE_MACHINE_ARM64: + goarch = archARM64 + case pe.IMAGE_FILE_MACHINE_I386: + goarch = "386" + default: + return "", "", false + } + return "windows", goarch, true + } + return "", "", false +} + +// checkBinaryPlatform fails when the binary at path was built for another +// platform than this process runs on. The archive in plugin_archives was +// resolved for the API server's platform at install time, so a proxy node +// on a different platform cannot run it; a clear error here beats the +// supervisor discovering an exec failure on every restart. +func checkBinaryPlatform(path string) error { + goos, goarch, ok := binaryPlatform(path) + if !ok { + return nil + } + if goos != runtime.GOOS || goarch != runtime.GOARCH { + return fmt.Errorf("plugin binary is built for %s/%s but this host is %s/%s; the stored archive was resolved for the API server's platform, so every proxy node must run the same platform in this release", goos, goarch, runtime.GOOS, runtime.GOARCH) + } + return nil +} diff --git a/internal/plugins/binary_platform_test.go b/internal/plugins/binary_platform_test.go new file mode 100644 index 0000000000..69e4242be6 --- /dev/null +++ b/internal/plugins/binary_platform_test.go @@ -0,0 +1,59 @@ +package plugins + +import ( + "os" + "path/filepath" + "runtime" + "testing" +) + +func TestBinaryPlatformReadsThisHostsBinaries(t *testing.T) { + self, err := os.Executable() + if err != nil { + t.Skip("no executable path") + } + goos, goarch, ok := binaryPlatform(self) + if !ok { + t.Skipf("test binary format not recognized on %s/%s", runtime.GOOS, runtime.GOARCH) + } + if goos != runtime.GOOS || goarch != runtime.GOARCH { + t.Fatalf("binaryPlatform(self) = %s/%s, want %s/%s", goos, goarch, runtime.GOOS, runtime.GOARCH) + } + if err := checkBinaryPlatform(self); err != nil { + t.Fatalf("checkBinaryPlatform(self) = %v", err) + } +} + +func TestCheckBinaryPlatformIgnoresUnknownFormats(t *testing.T) { + path := filepath.Join(t.TempDir(), "plugin") + if err := os.WriteFile(path, []byte("#!/bin/sh\nexit 0\n"), 0o755); err != nil { + t.Fatal(err) + } + if err := checkBinaryPlatform(path); err != nil { + t.Fatalf("script treated as a platform mismatch: %v", err) + } +} + +func TestCheckBinaryPlatformRejectsForeignELF(t *testing.T) { + if runtime.GOOS == "linux" && runtime.GOARCH == "arm64" { + t.Skip("fixture is linux/arm64") + } + // Minimal ELF64 header: little-endian, EM_AARCH64 (0xB7). + hdr := make([]byte, 64) + copy(hdr, []byte{0x7f, 'E', 'L', 'F', 2, 1, 1}) + hdr[16] = 2 // ET_EXEC + hdr[18] = 0xB7 // EM_AARCH64 + hdr[20] = 1 // EV_CURRENT + hdr[52] = 64 // e_ehsize + path := filepath.Join(t.TempDir(), "plugin") + if err := os.WriteFile(path, hdr, 0o755); err != nil { + t.Fatal(err) + } + goos, goarch, ok := binaryPlatform(path) + if !ok || goos != "linux" || goarch != "arm64" { + t.Fatalf("binaryPlatform = %s/%s ok=%v, want linux/arm64", goos, goarch, ok) + } + if err := checkBinaryPlatform(path); err == nil { + t.Fatal("foreign ELF accepted") + } +} diff --git a/internal/plugins/host_adapter.go b/internal/plugins/host_adapter.go index c349617813..3cec17061e 100644 --- a/internal/plugins/host_adapter.go +++ b/internal/plugins/host_adapter.go @@ -32,3 +32,10 @@ func (h *hostAdapter) Stop(installationID int) error { func (h *hostAdapter) Shutdown(ctx context.Context) error { return h.host.Shutdown(ctx) } + +// NextStartSeq forwards the host's start counter so the resident +// supervisor can tell a launch it issued from one it joined through the +// singleflight (see ResidentSupervisor.runStart). +func (h *hostAdapter) NextStartSeq() uint64 { + return h.host.NextStartSeq() +} diff --git a/internal/plugins/installation.go b/internal/plugins/installation.go index 1a8481e0f7..7635e74035 100644 --- a/internal/plugins/installation.go +++ b/internal/plugins/installation.go @@ -34,17 +34,18 @@ const ( ) type Installation struct { - ID int - RepositoryID *int - PluginID string - Version string - InstallPath string - Enabled bool - Kind string `json:"kind"` - UpdatePolicy string `json:"update_policy"` - AvailableVersion *string `json:"available_version,omitempty"` - CreatedAt time.Time - UpdatedAt time.Time + ID int + RepositoryID *int + PluginID string + Version string + InstallPath string + Enabled bool + Kind string `json:"kind"` + UpdatePolicy string `json:"update_policy"` + AvailableVersion *string `json:"available_version,omitempty"` + CreatedAt time.Time + UpdatedAt time.Time + RuntimeGeneration int64 `json:"-"` } // IsBuiltin reports whether this is the reserved builtin-host installation. @@ -87,6 +88,8 @@ type UpdateInstallationInput struct { UpdatePolicy *string AvailableVersion *string Capabilities []Capability + // Restart durably requests a new process on every resident host. + Restart bool } type InstallationStore struct { @@ -97,7 +100,7 @@ func NewInstallationStore(pool *pgxpool.Pool) *InstallationStore { return &InstallationStore{pool: pool} } -const installationColumns = `id, repository_id, plugin_id, version, install_path, enabled, kind, update_policy, available_version, created_at, updated_at` +const installationColumns = `id, repository_id, plugin_id, version, install_path, enabled, kind, update_policy, available_version, created_at, updated_at, runtime_generation` const capabilityColumns = `plugin_installation_id, capability_type, capability_id, metadata, created_at, updated_at` const archiveColumns = `plugin_installation_id, manifest_json, checksum, archive_bytes, created_at, updated_at` @@ -116,6 +119,7 @@ func scanInstallation(row pgx.Row) (*Installation, error) { &installation.AvailableVersion, &installation.CreatedAt, &installation.UpdatedAt, + &installation.RuntimeGeneration, ); err != nil { if errors.Is(err, pgx.ErrNoRows) { return nil, ErrInstallationNotFound @@ -330,6 +334,9 @@ func (s *InstallationStore) Update(ctx context.Context, id int, input UpdateInst args = append(args, *input.AvailableVersion) argIndex++ } + if input.Restart { + setClauses = append(setClauses, "runtime_generation = runtime_generation + 1") + } if len(setClauses) > 0 { setClauses = append(setClauses, "updated_at = NOW()") @@ -360,6 +367,42 @@ func (s *InstallationStore) Update(ctx context.Context, id int, input UpdateInst return nil } +// ListEnabledWithCapabilityTypes returns the enabled installations that +// declare at least one capability of the given types, in id order. One +// query replaces a ListEnabled + per-installation ListCapabilities fan-out +// for callers that only need membership, such as the resident supervisor. +func (s *InstallationStore) ListEnabledWithCapabilityTypes(ctx context.Context, capabilityTypes []string) ([]*Installation, error) { + if len(capabilityTypes) == 0 { + return nil, nil + } + query := `SELECT ` + installationColumns + ` FROM plugin_installations + WHERE enabled = true + AND EXISTS ( + SELECT 1 FROM plugin_capabilities + WHERE plugin_capabilities.plugin_installation_id = plugin_installations.id + AND plugin_capabilities.capability_type = ANY($1) + ) + ORDER BY id ASC` + rows, err := s.pool.Query(ctx, query, capabilityTypes) + if err != nil { + return nil, fmt.Errorf("listing enabled plugin installations by capability type: %w", err) + } + defer rows.Close() + + var installations []*Installation + for rows.Next() { + installation, err := scanInstallation(rows) + if err != nil { + return nil, err + } + installations = append(installations, installation) + } + if err := rows.Err(); err != nil { + return nil, fmt.Errorf("iterating enabled plugin installations by capability type: %w", err) + } + return installations, nil +} + func (s *InstallationStore) ListCapabilities(ctx context.Context, installationID int) ([]*Capability, error) { query := `SELECT ` + capabilityColumns + ` FROM plugin_capabilities WHERE plugin_installation_id = $1 diff --git a/internal/plugins/installation_resident_query_test.go b/internal/plugins/installation_resident_query_test.go new file mode 100644 index 0000000000..2595c5ab84 --- /dev/null +++ b/internal/plugins/installation_resident_query_test.go @@ -0,0 +1,82 @@ +package plugins + +import ( + "context" + "testing" + "time" + + "github.com/jackc/pgx/v5/pgxpool" + + "github.com/Silo-Server/silo-plugin-sdk/pkg/pluginsdk/capability" +) + +// ListEnabledWithCapabilityTypes backs the resident supervisor's reconcile: +// it must return exactly the enabled installations declaring one of the +// requested capability types, once each, and skip disabled rows and rows +// whose capabilities are all of other types. +func TestInstallationStoreListEnabledWithCapabilityTypes(t *testing.T) { + pool := builtinGuardTestPool(t) + store := NewInstallationStore(pool) + ctx := context.Background() + + resident := seedResidentQueryInstallation(t, pool, true) + seedResidentQueryCapability(t, pool, resident, capability.NetworkAccessProvider, "a") + seedResidentQueryCapability(t, pool, resident, capability.NetworkAccessProvider, "b") + seedResidentQueryCapability(t, pool, resident, capability.MetadataProvider, "meta") + + disabled := seedResidentQueryInstallation(t, pool, false) + seedResidentQueryCapability(t, pool, disabled, capability.NetworkAccessProvider, "off") + + lazy := seedResidentQueryInstallation(t, pool, true) + seedResidentQueryCapability(t, pool, lazy, capability.MetadataProvider, "meta") + + got, err := store.ListEnabledWithCapabilityTypes(ctx, []string{capability.NetworkAccessProvider}) + if err != nil { + t.Fatalf("ListEnabledWithCapabilityTypes: %v", err) + } + seen := map[int]int{} + for _, installation := range got { + seen[installation.ID]++ + } + if seen[resident] != 1 { + t.Fatalf("resident installation %d returned %d times, want 1 (got %+v)", resident, seen[resident], seen) + } + if seen[disabled] != 0 || seen[lazy] != 0 { + t.Fatalf("disabled (%d) or lazy (%d) installation returned: %+v", disabled, lazy, seen) + } + + none, err := store.ListEnabledWithCapabilityTypes(ctx, nil) + if err != nil { + t.Fatalf("ListEnabledWithCapabilityTypes(nil): %v", err) + } + if len(none) != 0 { + t.Fatalf("empty type list returned %d installations, want 0", len(none)) + } +} + +func seedResidentQueryInstallation(t *testing.T, pool *pgxpool.Pool, enabled bool) int { + t.Helper() + var id int + pluginID := "silo.test.resident-query" + time.Now().UTC().Format("-20060102150405.000000000") + err := pool.QueryRow(context.Background(), + `INSERT INTO plugin_installations (plugin_id, version, install_path, enabled, update_policy, kind) + VALUES ($1, '0', '/nonexistent/resident-query-test', $2, 'manual', 'plugin') + RETURNING id`, pluginID, enabled).Scan(&id) + if err != nil { + t.Fatalf("seed installation: %v", err) + } + t.Cleanup(func() { + _, _ = pool.Exec(context.Background(), `DELETE FROM plugin_installations WHERE id = $1`, id) + }) + return id +} + +func seedResidentQueryCapability(t *testing.T, pool *pgxpool.Pool, installationID int, capabilityType, capabilityID string) { + t.Helper() + _, err := pool.Exec(context.Background(), + `INSERT INTO plugin_capabilities (plugin_installation_id, capability_type, capability_id, metadata) + VALUES ($1, $2, $3, '{}'::jsonb)`, installationID, capabilityType, capabilityID) + if err != nil { + t.Fatalf("seed capability: %v", err) + } +} diff --git a/internal/plugins/instance_state.go b/internal/plugins/instance_state.go new file mode 100644 index 0000000000..dc2aa73a9c --- /dev/null +++ b/internal/plugins/instance_state.go @@ -0,0 +1,253 @@ +package plugins + +import ( + "context" + "errors" + "fmt" + "strconv" + "strings" + + "github.com/jackc/pgx/v5" + "github.com/jackc/pgx/v5/pgxpool" + + "github.com/Silo-Server/silo-server/internal/pluginhost" + "github.com/Silo-Server/silo-server/internal/secret" +) + +// HostScopeAPI is the instance-state scope of the api host. Proxy nodes use +// NodeHostScope. The scope is derived by the host, never by the plugin. +const HostScopeAPI = "api" + +// NodeHostScope returns the instance-state scope of one proxy node, +// "node:". +func NodeHostScope(nodeRowID int64) string { + return "node:" + strconv.FormatInt(nodeRowID, 10) +} + +// InstanceStateStore stores plugin_instance_state rows: one encrypted value +// per installation, host scope and key. Values are sealed with the server +// data cipher under row AAD so a blob cannot be moved to another installation, +// host or key, and nothing in the API returns them. +type InstanceStateStore struct { + pool *pgxpool.Pool + cipher *secret.Cipher +} + +// NewInstanceStateStore creates the store. The cipher is required: instance +// state holds overlay node keys and is never written in plaintext. +func NewInstanceStateStore(pool *pgxpool.Pool, cipher *secret.Cipher) *InstanceStateStore { + return &InstanceStateStore{pool: pool, cipher: cipher} +} + +// ForScope binds the store to one host scope, giving the pluginhost +// RuntimeHost server an InstanceStateStore that only names installations +// and keys. +func (s *InstanceStateStore) ForScope(scope string) *ScopedInstanceStateStore { + return &ScopedInstanceStateStore{store: s, scope: scope} +} + +func instanceStateAAD(installationID int, scope, key string) string { + return secret.RowAAD( + "plugin_instance_state", + "state_value", + strconv.Itoa(installationID)+":"+scope+":"+key, + ) +} + +func validateInstanceStateScope(installationID int, scope string) error { + if installationID <= 0 { + return pluginhost.ErrInstanceStateUnavailable + } + if strings.TrimSpace(scope) == "" { + return errors.New("instance state host scope is required") + } + return nil +} + +// Read returns the value stored under key for the installation in scope. +func (s *InstanceStateStore) Read(ctx context.Context, installationID int, scope, key string) ([]byte, bool, error) { + if err := validateInstanceStateScope(installationID, scope); err != nil { + return nil, false, err + } + if err := pluginhost.ValidateInstanceStateKey(key); err != nil { + return nil, false, err + } + if s == nil || s.pool == nil { + return nil, false, errors.New("instance state store is not configured") + } + var stored []byte + err := s.pool.QueryRow(ctx, ` + SELECT state_value + FROM plugin_instance_state + WHERE plugin_installation_id = $1 AND host_scope = $2 AND state_key = $3 + `, installationID, scope, key).Scan(&stored) + if err != nil { + if errors.Is(err, pgx.ErrNoRows) { + return nil, false, nil + } + return nil, false, fmt.Errorf("reading plugin instance state: %w", err) + } + value, err := s.open(installationID, scope, key, stored) + if err != nil { + return nil, false, err + } + return value, true, nil +} + +// Write stores value under key for the installation in scope, replacing any +// previous value. It fails with pluginhost.ErrInstanceStateTooManyKeys when +// the scope already holds InstanceStateMaxKeys other keys. +func (s *InstanceStateStore) Write(ctx context.Context, installationID int, scope, key string, value []byte) error { + if err := validateInstanceStateScope(installationID, scope); err != nil { + return err + } + if err := pluginhost.ValidateInstanceStateKey(key); err != nil { + return err + } + if len(value) > pluginhost.InstanceStateMaxValueBytes { + return pluginhost.ErrInstanceStateValueTooLarge + } + if s == nil || s.pool == nil { + return errors.New("instance state store is not configured") + } + sealed, err := s.seal(installationID, scope, key, value) + if err != nil { + return err + } + // The key budget is checked in the same statement as the upsert: an + // existing key always updates, a new key only lands while the scope has + // room. Zero affected rows means the budget refused it. Writers to one + // scope are serialized by a transaction-scoped advisory lock so two + // concurrent new keys cannot both observe the same count and overshoot. + tx, err := s.pool.Begin(ctx) + if err != nil { + return fmt.Errorf("writing plugin instance state: %w", err) + } + defer func() { _ = tx.Rollback(ctx) }() + if _, err := tx.Exec(ctx, `SELECT pg_advisory_xact_lock(hashtext('plugin_instance_state'), hashtext($1::text || ':' || $2))`, strconv.Itoa(installationID), scope); err != nil { + return fmt.Errorf("writing plugin instance state: %w", err) + } + tag, err := tx.Exec(ctx, ` + INSERT INTO plugin_instance_state (plugin_installation_id, host_scope, state_key, state_value) + SELECT $1, $2, $3, $4 + WHERE ( + SELECT count(*) + FROM plugin_instance_state + WHERE plugin_installation_id = $1 AND host_scope = $2 AND state_key <> $3 + ) < $5 + ON CONFLICT (plugin_installation_id, host_scope, state_key) DO UPDATE SET + state_value = EXCLUDED.state_value, + updated_at = NOW() + `, installationID, scope, key, sealed, pluginhost.InstanceStateMaxKeys) + if err != nil { + return fmt.Errorf("writing plugin instance state: %w", err) + } + if tag.RowsAffected() == 0 { + return pluginhost.ErrInstanceStateTooManyKeys + } + if err := tx.Commit(ctx); err != nil { + return fmt.Errorf("writing plugin instance state: %w", err) + } + return nil +} + +// Keys lists the keys stored for the installation in scope, in key order. +func (s *InstanceStateStore) Keys(ctx context.Context, installationID int, scope string) ([]string, error) { + if err := validateInstanceStateScope(installationID, scope); err != nil { + return nil, err + } + if s == nil || s.pool == nil { + return nil, errors.New("instance state store is not configured") + } + rows, err := s.pool.Query(ctx, ` + SELECT state_key + FROM plugin_instance_state + WHERE plugin_installation_id = $1 AND host_scope = $2 + ORDER BY state_key ASC + `, installationID, scope) + if err != nil { + return nil, fmt.Errorf("listing plugin instance state keys: %w", err) + } + defer rows.Close() + var keys []string + for rows.Next() { + var key string + if err := rows.Scan(&key); err != nil { + return nil, fmt.Errorf("scanning plugin instance state key: %w", err) + } + keys = append(keys, key) + } + if err := rows.Err(); err != nil { + return nil, fmt.Errorf("iterating plugin instance state keys: %w", err) + } + return keys, nil +} + +// Empty values need an authenticated marker because Cipher.Encrypt leaves +// empty plaintext unwrapped. A separate AAD domain prevents adding or removing +// the marker from turning an ordinary ciphertext into an empty value. +const instanceStateEmptyMarker = "empty:v1:" + +func (s *InstanceStateStore) seal(installationID int, scope, key string, value []byte) ([]byte, error) { + if s.cipher == nil { + return nil, errors.New("plugin instance state requires the server data cipher") + } + if len(value) == 0 { + ciphertext, err := s.cipher.Encrypt("empty", instanceStateEmptyMarker+instanceStateAAD(installationID, scope, key)) + if err != nil { + return nil, fmt.Errorf("encrypting empty plugin instance state: %w", err) + } + return []byte(instanceStateEmptyMarker + ciphertext), nil + } + ciphertext, err := s.cipher.Encrypt(string(value), instanceStateAAD(installationID, scope, key)) + if err != nil { + return nil, fmt.Errorf("encrypting plugin instance state: %w", err) + } + return []byte(ciphertext), nil +} + +func (s *InstanceStateStore) open(installationID int, scope, key string, stored []byte) ([]byte, error) { + if s.cipher == nil { + return nil, errors.New("plugin instance state requires the server data cipher") + } + if ciphertext, empty := strings.CutPrefix(string(stored), instanceStateEmptyMarker); empty { + plaintext, err := s.cipher.Decrypt(ciphertext, instanceStateEmptyMarker+instanceStateAAD(installationID, scope, key)) + if err != nil { + return nil, fmt.Errorf("decrypting empty plugin instance state: %w", err) + } + if plaintext != "empty" { + return nil, errors.New("invalid empty plugin instance state marker") + } + return []byte{}, nil + } + if !secret.IsEncrypted(string(stored)) { + return nil, errors.New("plugin instance state row is not an encrypted envelope") + } + plaintext, err := s.cipher.Decrypt(string(stored), instanceStateAAD(installationID, scope, key)) + if err != nil { + return nil, fmt.Errorf("decrypting plugin instance state: %w", err) + } + return []byte(plaintext), nil +} + +// ScopedInstanceStateStore is an InstanceStateStore bound to one fixed host +// scope for a plugin process. It satisfies pluginhost.InstanceStateStore. +type ScopedInstanceStateStore struct { + store *InstanceStateStore + scope string +} + +// Scope returns the bound host scope. +func (s *ScopedInstanceStateStore) Scope() string { return s.scope } + +// ReadInstanceState implements pluginhost.InstanceStateStore. +func (s *ScopedInstanceStateStore) ReadInstanceState(ctx context.Context, installationID int, key string) ([]byte, bool, error) { + return s.store.Read(ctx, installationID, s.scope, key) +} + +// WriteInstanceState implements pluginhost.InstanceStateStore. +func (s *ScopedInstanceStateStore) WriteInstanceState(ctx context.Context, installationID int, key string, value []byte) error { + return s.store.Write(ctx, installationID, s.scope, key, value) +} + +var _ pluginhost.InstanceStateStore = (*ScopedInstanceStateStore)(nil) diff --git a/internal/plugins/instance_state_test.go b/internal/plugins/instance_state_test.go new file mode 100644 index 0000000000..d977f1f3b7 --- /dev/null +++ b/internal/plugins/instance_state_test.go @@ -0,0 +1,278 @@ +package plugins + +import ( + "bytes" + "context" + "errors" + "fmt" + "sort" + "strings" + "testing" + + "github.com/Silo-Server/silo-server/internal/pluginhost" + "github.com/Silo-Server/silo-server/internal/secret" +) + +func instanceStateTestCipher(t *testing.T) *secret.Cipher { + t.Helper() + cipher, err := secret.New(bytes.Repeat([]byte("k"), secret.MinMasterKeyLen)) + if err != nil { + t.Fatalf("secret.New: %v", err) + } + return cipher +} + +// The store round-trips arbitrary bytes, keeps the row encrypted with AAD +// bound to installation, scope and key, isolates scopes and installations, +// and distinguishes an empty value from an absent one. +func TestInstanceStateStoreRoundTripEncryptedAndScoped(t *testing.T) { + pool := builtinGuardTestPool(t) + ctx := context.Background() + cipher := instanceStateTestCipher(t) + store := NewInstanceStateStore(pool, cipher) + + installation := seedResidentQueryInstallation(t, pool, true) + other := seedResidentQueryInstallation(t, pool, true) + + value := []byte("node-key\x00\x01binary") + if err := store.Write(ctx, installation, HostScopeAPI, "_machinekey", value); err != nil { + t.Fatalf("Write: %v", err) + } + got, found, err := store.Read(ctx, installation, HostScopeAPI, "_machinekey") + if err != nil || !found || !bytes.Equal(got, value) { + t.Fatalf("Read = %q, %v, %v; want %q", got, found, err, value) + } + + // The row never holds the plaintext; it is an enc:v1 envelope bound to + // this exact (installation, scope, key). + var stored []byte + if err := pool.QueryRow(ctx, + `SELECT state_value FROM plugin_instance_state WHERE plugin_installation_id = $1 AND host_scope = $2 AND state_key = $3`, + installation, HostScopeAPI, "_machinekey").Scan(&stored); err != nil { + t.Fatalf("select row: %v", err) + } + if bytes.Contains(stored, value) || !secret.IsEncrypted(string(stored)) { + t.Fatalf("stored value is not an encrypted envelope: %q", stored) + } + if _, err := cipher.Decrypt(string(stored), instanceStateAAD(installation, NodeHostScope(3), "_machinekey")); err == nil { + t.Fatal("envelope decrypted under another scope's AAD") + } + if plain, err := cipher.Decrypt(string(stored), instanceStateAAD(installation, HostScopeAPI, "_machinekey")); err != nil || plain != string(value) { + t.Fatalf("envelope under the row AAD: %q, %v", plain, err) + } + + // Scope and installation isolation. + if _, found, err := store.Read(ctx, installation, NodeHostScope(3), "_machinekey"); err != nil || found { + t.Fatalf("node scope saw the api value: found=%v err=%v", found, err) + } + if _, found, err := store.Read(ctx, other, HostScopeAPI, "_machinekey"); err != nil || found { + t.Fatalf("other installation saw the value: found=%v err=%v", found, err) + } + if err := store.Write(ctx, installation, NodeHostScope(3), "_machinekey", []byte("proxy-key")); err != nil { + t.Fatalf("Write node scope: %v", err) + } + got, _, err = store.Read(ctx, installation, HostScopeAPI, "_machinekey") + if err != nil || !bytes.Equal(got, value) { + t.Fatalf("api scope changed by node write: %q, %v", got, err) + } + + // Overwrite and empty values. + if err := store.Write(ctx, installation, HostScopeAPI, "_machinekey", []byte("rotated")); err != nil { + t.Fatalf("overwrite: %v", err) + } + got, _, _ = store.Read(ctx, installation, HostScopeAPI, "_machinekey") + if string(got) != "rotated" { + t.Fatalf("after overwrite = %q", got) + } + if err := store.Write(ctx, installation, HostScopeAPI, "empty", nil); err != nil { + t.Fatalf("write empty: %v", err) + } + got, found, err = store.Read(ctx, installation, HostScopeAPI, "empty") + if err != nil || !found || len(got) != 0 { + t.Fatalf("empty value = %q, %v, %v; want found and empty", got, found, err) + } + if _, found, err := store.Read(ctx, installation, HostScopeAPI, "missing"); err != nil || found { + t.Fatalf("missing key: found=%v err=%v", found, err) + } + + // Key order follows the database collation, so compare as a set. + keys, err := store.Keys(ctx, installation, HostScopeAPI) + sort.Strings(keys) + if err != nil || strings.Join(keys, ",") != "_machinekey,empty" { + t.Fatalf("Keys = %v, %v", keys, err) + } + + // The scoped adapter satisfies the pluginhost contract with the scope fixed. + var scoped pluginhost.InstanceStateStore = store.ForScope(NodeHostScope(3)) + got, found, err = scoped.ReadInstanceState(ctx, installation, "_machinekey") + if err != nil || !found || string(got) != "proxy-key" { + t.Fatalf("scoped read = %q, %v, %v", got, found, err) + } + + // Uninstall cascades. + if _, err := pool.Exec(ctx, `DELETE FROM plugin_installations WHERE id = $1`, installation); err != nil { + t.Fatal(err) + } + var remaining int + if err := pool.QueryRow(ctx, `SELECT count(*) FROM plugin_instance_state WHERE plugin_installation_id = $1`, installation).Scan(&remaining); err != nil { + t.Fatal(err) + } + if remaining != 0 { + t.Fatalf("%d rows survived the installation delete", remaining) + } +} + +func TestInstanceStateStoreLimits(t *testing.T) { + pool := builtinGuardTestPool(t) + ctx := context.Background() + store := NewInstanceStateStore(pool, instanceStateTestCipher(t)) + installation := seedResidentQueryInstallation(t, pool, true) + + longKey := strings.Repeat("k", pluginhost.InstanceStateMaxKeyBytes+1) + if err := store.Write(ctx, installation, HostScopeAPI, longKey, []byte("x")); !errors.Is(err, pluginhost.ErrInstanceStateKeyTooLong) { + t.Fatalf("long key: %v", err) + } + if _, _, err := store.Read(ctx, installation, HostScopeAPI, longKey); !errors.Is(err, pluginhost.ErrInstanceStateKeyTooLong) { + t.Fatalf("long key read: %v", err) + } + if err := store.Write(ctx, installation, HostScopeAPI, "", []byte("x")); err == nil { + t.Fatal("empty key accepted") + } + maxKey := strings.Repeat("k", pluginhost.InstanceStateMaxKeyBytes) + if err := store.Write(ctx, installation, HostScopeAPI, maxKey, []byte("x")); err != nil { + t.Fatalf("256-byte key rejected: %v", err) + } + + big := bytes.Repeat([]byte("v"), pluginhost.InstanceStateMaxValueBytes+1) + if err := store.Write(ctx, installation, HostScopeAPI, "big", big); !errors.Is(err, pluginhost.ErrInstanceStateValueTooLarge) { + t.Fatalf("oversize value: %v", err) + } + if err := store.Write(ctx, installation, HostScopeAPI, "big", big[:pluginhost.InstanceStateMaxValueBytes]); err != nil { + t.Fatalf("256 KiB value rejected: %v", err) + } + + // Key budget per scope: the two keys above plus 254 more fill it; the + // 257th new key is refused while rewriting an existing key still works + // and another scope is unaffected. + for i := 0; i < pluginhost.InstanceStateMaxKeys-2; i++ { + if err := store.Write(ctx, installation, HostScopeAPI, fmt.Sprintf("k%03d", i), []byte("v")); err != nil { + t.Fatalf("fill key %d: %v", i, err) + } + } + if err := store.Write(ctx, installation, HostScopeAPI, "one-too-many", []byte("v")); !errors.Is(err, pluginhost.ErrInstanceStateTooManyKeys) { + t.Fatalf("257th key: %v", err) + } + if err := store.Write(ctx, installation, HostScopeAPI, "k000", []byte("rewritten")); err != nil { + t.Fatalf("rewrite at the budget: %v", err) + } + if err := store.Write(ctx, installation, NodeHostScope(9), "one-too-many", []byte("v")); err != nil { + t.Fatalf("other scope blocked by the api budget: %v", err) + } + keys, err := store.Keys(ctx, installation, HostScopeAPI) + if err != nil || len(keys) != pluginhost.InstanceStateMaxKeys { + t.Fatalf("Keys = %d, %v; want %d", len(keys), err, pluginhost.InstanceStateMaxKeys) + } + + // Config test-runs use negative installation ids and get no state. + for _, id := range []int{0, -1, -42} { + if err := store.Write(ctx, id, HostScopeAPI, "k", []byte("v")); !errors.Is(err, pluginhost.ErrInstanceStateUnavailable) { + t.Fatalf("installation %d write: %v", id, err) + } + if _, _, err := store.Read(ctx, id, HostScopeAPI, "k"); !errors.Is(err, pluginhost.ErrInstanceStateUnavailable) { + t.Fatalf("installation %d read: %v", id, err) + } + } + if err := store.Write(ctx, installation, "", "k", []byte("v")); err == nil { + t.Fatal("empty scope accepted") + } +} + +func TestInstanceStateStoreRequiresCipher(t *testing.T) { + pool := builtinGuardTestPool(t) + store := NewInstanceStateStore(pool, nil) + installation := seedResidentQueryInstallation(t, pool, true) + if err := store.Write(context.Background(), installation, HostScopeAPI, "k", []byte("v")); err == nil { + t.Fatal("plaintext write accepted without a cipher") + } +} + +// Concurrent first writes of distinct keys at the budget must not overshoot +// it: admission of a new key is serialized per scope. +func TestInstanceStateWriteKeyBudgetHoldsUnderConcurrency(t *testing.T) { + pool := builtinGuardTestPool(t) + store := NewInstanceStateStore(pool, instanceStateTestCipher(t)) + ctx := context.Background() + installation := seedResidentQueryInstallation(t, pool, true) + for i := 0; i < pluginhost.InstanceStateMaxKeys-1; i++ { + if err := store.Write(ctx, installation, HostScopeAPI, fmt.Sprintf("k%03d", i), []byte("v")); err != nil { + t.Fatalf("fill key %d: %v", i, err) + } + } + const racers = 16 + errs := make(chan error, racers) + start := make(chan struct{}) + for i := 0; i < racers; i++ { + go func(i int) { + <-start + errs <- store.Write(ctx, installation, HostScopeAPI, fmt.Sprintf("race%02d", i), []byte("v")) + }(i) + } + close(start) + admitted := 0 + for i := 0; i < racers; i++ { + switch err := <-errs; { + case err == nil: + admitted++ + case errors.Is(err, pluginhost.ErrInstanceStateTooManyKeys): + default: + t.Fatalf("racing write: %v", err) + } + } + keys, err := store.Keys(ctx, installation, HostScopeAPI) + if err != nil { + t.Fatal(err) + } + if admitted != 1 || len(keys) != pluginhost.InstanceStateMaxKeys { + t.Fatalf("admitted %d racers, scope holds %d keys; want exactly 1 and %d", admitted, len(keys), pluginhost.InstanceStateMaxKeys) + } +} + +func TestInstanceStateEmptyValueAuthenticatesRow(t *testing.T) { + store := NewInstanceStateStore(nil, instanceStateTestCipher(t)) + stored, err := store.seal(5, "api", "key", nil) + if err != nil { + t.Fatal(err) + } + if value, err := store.open(5, "api", "key", stored); err != nil || len(value) != 0 { + t.Fatalf("roundtrip=%q, %v", value, err) + } + for _, row := range []struct { + id int + scope, key string + }{{6, "api", "key"}, {5, "node:1", "key"}, {5, "api", "other"}} { + if _, err := store.open(row.id, row.scope, row.key, stored); err == nil { + t.Errorf("empty value accepted under another row: %+v", row) + } + } + if _, err := store.open(5, "api", "key", []byte("empty:v1")); err == nil { + t.Error("plaintext empty marker accepted") + } + if _, err := store.open(5, "api", "key", bytes.TrimPrefix(stored, []byte(instanceStateEmptyMarker))); err == nil { + t.Error("removing the empty marker changed an authenticated empty value into ordinary data") + } + ordinary, err := store.seal(5, "api", "key:empty", []byte("empty")) + if err != nil { + t.Fatal(err) + } + if _, err := store.open(5, "api", "key", append([]byte(instanceStateEmptyMarker), ordinary...)); err == nil { + t.Error("ordinary ciphertext became an authenticated empty value") + } + value := []byte("empty:v1") + stored, err = store.seal(5, "api", "key", value) + if err != nil { + t.Fatal(err) + } + if got, err := store.open(5, "api", "key", stored); err != nil || !bytes.Equal(got, value) { + t.Fatalf("marker literal=%q, %v", got, err) + } +} diff --git a/internal/plugins/lifecycle_events.go b/internal/plugins/lifecycle_events.go new file mode 100644 index 0000000000..a58ab5a3ac --- /dev/null +++ b/internal/plugins/lifecycle_events.go @@ -0,0 +1,158 @@ +package plugins + +import ( + "context" + "encoding/json" + "errors" + "log/slog" + "sort" + "sync" + "time" + + "github.com/Silo-Server/silo-server/internal/cache" +) + +// PluginsChangedEvent is the payload of cache.EventPluginsChanged. An empty +// payload means "something about the installed set changed; reconcile". +// Restart remains accepted from older API replicas. Current senders advance +// plugin_installations.runtime_generation on admin config saves and restarts, +// then send a plain reconcile event; the poll can recover missed events. +type PluginsChangedEvent struct { + InstallationID int `json:"installation_id,omitempty"` + Restart bool `json:"restart,omitempty"` +} + +// DefaultLifecyclePollInterval is how often a following host reconciles +// without an event, so a missed publish (Redis hiccup, restart) costs at most +// one interval. +const DefaultLifecyclePollInterval = time.Minute + +// PublishLifecycleChanges makes this service announce every lifecycle change +// on cache.ChannelAdmin as cache.EventPluginsChanged. The API server calls +// it; proxy nodes follow with FollowLifecycleChanges. Publishing is +// best-effort: a failed publish is logged and the followers' poll catches up. +func (s *Service) PublishLifecycleChanges(bus cache.EventBus) { + if s == nil || bus == nil { + return + } + s.lifecycleBus = bus + s.AddLifecycleHook(func(ctx context.Context) { s.publishPluginsChanged(ctx, PluginsChangedEvent{}) }) +} + +func (s *Service) publishPluginsChanged(ctx context.Context, event PluginsChangedEvent) { + if s == nil || s.lifecycleBus == nil { + return + } + payload := "" + if event != (PluginsChangedEvent{}) { + encoded, err := json.Marshal(event) + if err != nil { + slog.WarnContext(ctx, "encode plugins changed event", "component", "plugins", "error", err) + return + } + payload = string(encoded) + } + if err := s.lifecycleBus.Publish(ctx, cache.ChannelAdmin, cache.Event{Type: cache.EventPluginsChanged, Payload: payload}); err != nil { + slog.WarnContext(ctx, "publish plugins changed event", "component", "plugins", "error", err) + } +} + +// FollowLifecycleChanges makes this service (a proxy node's) reconcile on +// every cache.EventPluginsChanged published on cache.ChannelAdmin and, as a +// backstop, every poll interval (DefaultLifecyclePollInterval when poll is +// zero). Both run OnLifecycleChange, which drops the installation cache and +// reconciles the resident set; an event naming an installation to restart +// replaces that process first. Events are handled by one worker off the bus +// goroutine, so a slow plugin handshake never delays other subscribers, and +// a burst of admin actions collapses into one reconcile (plus one restart +// per named installation) rather than a reconcile per event. +// +// The poll is started even when the subscription fails (Redis down at boot, +// for example): the returned error then only says that changes arrive on +// the poll alone, and the caller logs it rather than giving up. +func (s *Service) FollowLifecycleChanges(ctx context.Context, bus cache.EventBus, poll time.Duration) error { + if s == nil { + return nil + } + if poll <= 0 { + poll = DefaultLifecyclePollInterval + } + follower := &lifecycleFollower{wake: make(chan struct{}, 1)} + var subscribeErr error + if bus != nil { + subscribeErr = bus.Subscribe(ctx, cache.ChannelAdmin, func(event cache.Event) { + if event.Type != cache.EventPluginsChanged { + return + } + var payload PluginsChangedEvent + if event.Payload != "" { + if err := json.Unmarshal([]byte(event.Payload), &payload); err != nil { + slog.WarnContext(ctx, "decode plugins changed event; reconciling anyway", "component", "plugins", "error", err) + payload = PluginsChangedEvent{} + } + } + follower.enqueue(payload) + }) + } + go func() { + ticker := time.NewTicker(poll) + defer ticker.Stop() + for { + select { + case <-ctx.Done(): + return + case <-ticker.C: + follower.enqueue(PluginsChangedEvent{}) + case <-follower.wake: + } + restarts := follower.drain() + for _, installationID := range restarts { + if err := s.resident.Restart(ctx, installationID); err != nil && !errors.Is(err, ErrNotResident) { + slog.WarnContext(ctx, "restart resident plugin after remote change", "component", "plugins", "installation_id", installationID, "error", err) + } + } + s.OnLifecycleChange(ctx) + } + }() + return subscribeErr +} + +// lifecycleFollower coalesces pending lifecycle events for the worker in +// FollowLifecycleChanges: any number of plain events become one reconcile, +// and restart events are kept per installation. +type lifecycleFollower struct { + mu sync.Mutex + restarts map[int]struct{} + wake chan struct{} +} + +func (f *lifecycleFollower) enqueue(event PluginsChangedEvent) { + f.mu.Lock() + if event.Restart && event.InstallationID > 0 { + if f.restarts == nil { + f.restarts = make(map[int]struct{}) + } + f.restarts[event.InstallationID] = struct{}{} + } + f.mu.Unlock() + select { + case f.wake <- struct{}{}: + default: + } +} + +// drain returns the installations to restart, in id order, and clears them. +func (f *lifecycleFollower) drain() []int { + f.mu.Lock() + defer f.mu.Unlock() + if len(f.restarts) == 0 { + return nil + } + ids := make([]int, 0, len(f.restarts)) + for id := range f.restarts { + ids = append(ids, id) + } + f.restarts = nil + sort.Ints(ids) + return ids +} diff --git a/internal/plugins/lifecycle_generation_test.go b/internal/plugins/lifecycle_generation_test.go new file mode 100644 index 0000000000..19e4f1d4bc --- /dev/null +++ b/internal/plugins/lifecycle_generation_test.go @@ -0,0 +1,181 @@ +package plugins + +import ( + "context" + "fmt" + "testing" + "time" + + "github.com/Silo-Server/silo-plugin-sdk/pkg/pluginsdk/capability" +) + +func TestRuntimeGenerationCountsConcurrentConfigSavesAndRestarts(t *testing.T) { + pool := builtinGuardTestPool(t) + store := NewInstallationStore(pool) + configs := NewRuntimeConfigStore(pool, instanceStateTestCipher(t)) + id := seedResidentQueryInstallation(t, pool, true) + ctx, cancel := context.WithTimeout(context.Background(), 10*time.Second) + defer cancel() + const requests = 16 + errs := make(chan error, requests) + start := make(chan struct{}) + for i := range requests { + go func() { + <-start + if i%2 == 0 { + saved, err := configs.CompareAndSwapGlobalConfig(ctx, id, fmt.Sprintf("config-%d", i), map[string]any{"enabled": true}, nil) + if err == nil && !saved { + err = fmt.Errorf("new config %d was not saved", i) + } + errs <- err + return + } + errs <- store.Update(ctx, id, UpdateInstallationInput{Restart: true}) + }() + } + close(start) + for range requests { + if err := <-errs; err != nil { + t.Fatalf("concurrent config save/restart: %v", err) + } + } + installation, err := store.GetByID(ctx, id) + if err != nil || installation.RuntimeGeneration != requests { + t.Fatalf("runtime generation after %d requests: %+v, %v", requests, installation, err) + } +} + +func newResidentDatabaseFixture(t *testing.T, opts ResidentOptions) (*residentFixture, *InstallationStore, *RuntimeConfigStore, int) { + t.Helper() + pool := builtinGuardTestPool(t) + f := newResidentFixture(t, opts) + store := NewInstallationStore(pool) + source := f.store.byID[5] + installation, err := store.Create(context.Background(), CreateInstallationInput{ + PluginID: source.PluginID, Version: source.Version, InstallPath: source.InstallPath, Enabled: true, + Capabilities: []Capability{{Type: capability.NetworkAccessProvider, ID: "stub"}}, + }) + if err != nil { + t.Fatal(err) + } + t.Cleanup(func() { + _, _ = pool.Exec(context.Background(), `DELETE FROM plugin_installations WHERE id = $1`, installation.ID) + }) + configs := NewRuntimeConfigStore(pool, instanceStateTestCipher(t)) + f.service.installations, f.service.configs = store, configs + return f, store, configs, installation.ID +} + +func TestResidentPollRecoversMissedConfigSave(t *testing.T) { + f, store, configs, id := newResidentDatabaseFixture(t, ResidentOptions{}) + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + f.service.StartResidents(ctx) + waitState(t, f.service, id, "initial running", running) + first, err := f.host.Client(id) + if err != nil { + t.Fatal(err) + } + // This is the admin save's durable transaction; no lifecycle event reaches + // the follower, as happens while its Redis subscription is disconnected. + if saved, err := configs.CompareAndSwapGlobalConfig(ctx, id, "connection", map[string]any{"token": "replacement"}, nil); err != nil || !saved { + t.Fatalf("save config: saved=%v err=%v", saved, err) + } + installation, err := store.GetByID(ctx, id) + if err != nil || installation.RuntimeGeneration != 1 { + t.Fatalf("saved generation: %+v, %v", installation, err) + } + if err := f.service.FollowLifecycleChanges(ctx, nil, 20*time.Millisecond); err != nil { + t.Fatal(err) + } + waitState(t, f.service, id, "running with the saved config", func(state RuntimeState, tracked bool) bool { + current, err := f.host.Client(id) + return tracked && state.State == ResidentRunning && err == nil && current != first + }) + current, _ := f.host.Client(id) + // Provider callbacks persist their own state without requesting a restart. + if err := configs.PutGlobalConfig(ctx, id, "connection", map[string]any{"token": "rotated"}); err != nil { + t.Fatal(err) + } + f.service.OnLifecycleChange(ctx) + if again, err := f.host.Client(id); err != nil || again != current { + t.Fatalf("provider config write restarted its process: %v", err) + } + // A lost compare-and-swap race must not request another restart. + if saved, err := configs.CompareAndSwapGlobalConfig(ctx, id, "connection", map[string]any{"token": "stale"}, nil); err != nil || saved { + t.Fatalf("stale config save: saved=%v err=%v", saved, err) + } + installation, err = store.GetByID(ctx, id) + if err != nil || installation.RuntimeGeneration != 1 { + t.Fatalf("generation after provider/stale writes: %+v, %v", installation, err) + } +} + +func TestResidentPollRecoversMissedRestartOfFailedProvider(t *testing.T) { + f, store, _, id := newResidentDatabaseFixture(t, ResidentOptions{MaxFailures: 1}) + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + f.service.StartResidents(ctx) + waitState(t, f.service, id, "initial running", running) + f.crash(t) + waitState(t, f.service, id, "failed", func(state RuntimeState, tracked bool) bool { + return tracked && state.State == ResidentFailed + }) + f.heal(t) + // The API persists the restart but has no event bus connected to this + // follower. The generation must clear the follower's failure budget. + publisher := &Service{installations: store} + if err := publisher.RestartInstallation(ctx, id); err != nil { + t.Fatal(err) + } + if err := f.service.FollowLifecycleChanges(ctx, nil, 20*time.Millisecond); err != nil { + t.Fatal(err) + } + state := waitState(t, f.service, id, "running after missed restart event", running) + if state.RestartCount != 0 || state.LastError != "" { + t.Fatalf("restart did not clear failure budget: %+v", state) + } +} + +// An admin restart on the API host replaces a following proxy's process +// exactly once. The generation is durable, so the published event is a +// plain reconcile; a restart event on top would restart and then replace +// the fresh process again for the generation the follower had not seen. +func TestAdminRestartReplacesFollowerProcessOnce(t *testing.T) { + bus := newFakeBus() + follower, store, _, id := newResidentDatabaseFixture(t, ResidentOptions{}) + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + follower.service.StartResidents(ctx) + waitState(t, follower.service, id, "follower running", running) + // This hook follows the real reconcile hook. Wait for its asynchronous + // launches too, so the assertion observes all work from the restart event. + reconciled := make(chan struct{}) + follower.service.AddLifecycleHook(func(context.Context) { + follower.service.resident.starts.Wait() + close(reconciled) + }) + if err := follower.service.FollowLifecycleChanges(ctx, bus, time.Hour); err != nil { + t.Fatal(err) + } + before := follower.host.NextStartSeq() + + publisher := &Service{installations: store} + publisher.resident = newResidentSupervisor(publisher, ResidentOptions{}) + publisher.PublishLifecycleChanges(bus) + if err := publisher.RestartInstallation(ctx, id); err != nil { + t.Fatal(err) + } + + select { + case <-reconciled: + case <-time.After(30 * time.Second): + t.Fatal("follower did not finish reconciling the restart event") + } + if state, tracked := follower.service.resident.State(id); !tracked || state.State != ResidentRunning { + t.Fatalf("follower not running after restart: %+v, tracked=%v", state, tracked) + } + if got := follower.host.NextStartSeq() - before; got != 1 { + t.Fatalf("follower start count after one admin restart = %d, want 1", got) + } +} diff --git a/internal/plugins/network_access.go b/internal/plugins/network_access.go new file mode 100644 index 0000000000..ca97254bfe --- /dev/null +++ b/internal/plugins/network_access.go @@ -0,0 +1,520 @@ +package plugins + +import ( + "context" + "errors" + "fmt" + "log/slog" + "sort" + "strings" + "sync" + "time" + + pluginv1 "github.com/Silo-Server/silo-plugin-sdk/pkg/pluginproto/silo/plugin/v1" + "github.com/Silo-Server/silo-plugin-sdk/pkg/pluginsdk/capability" + + "github.com/Silo-Server/silo-server/internal/netaccess" + "github.com/Silo-Server/silo-server/internal/pluginhost" +) + +// ErrNetworkAccessProviderNotFound reports a provider slug no enabled +// installation declares. It is netaccess.ErrProviderNotFound so the proxy +// routes and the API agree on it without the proxy importing this package. +var ErrNetworkAccessProviderNotFound = netaccess.ErrProviderNotFound + +// ErrNetworkAccessHostUnknown reports a host id a connect or disconnect named +// that this deployment does not run a provider instance on. +var ErrNetworkAccessHostUnknown = errors.New("network access host unknown") + +// NetworkAccessProvider is one enabled installation declaring +// network_access_provider.v1, as the capability endpoint lists it. It is read +// from the manifest descriptor; listing never launches the plugin. +type NetworkAccessProvider struct { + InstallationID int + CapabilityID string + // Provider is the stable slug (tailscale, netbird, ...) that keys the + // admin routes and names the access path. + Provider string + DisplayName string +} + +// NetworkAccessHost identifies one process running an instance of a +// provider installation: the api server or, later, a proxy node. ID is the +// instance-state scope ("api", "node:") so the two never disagree. +type NetworkAccessHost struct { + ID string + Role string + Name string +} + +// NetworkAccessHostStatus is one host's answer. Status.State is +// netaccess.StateUnavailable when the plugin process is not running on the +// host or did not answer; Status.Error then says why when known. +type NetworkAccessHostStatus struct { + Host NetworkAccessHost + Status netaccess.Status +} + +// NetworkAccessReport is the admin view of one provider across every host. +type NetworkAccessReport struct { + Provider NetworkAccessProvider + Hosts []NetworkAccessHostStatus +} + +// NetworkAccessStatusSink receives the status a provider answered to an +// on-demand read or command, so the host's cached view (WebSocket origin +// allow list, later stream URL selection) does not wait for the plugin's next +// push. *netaccess.Broker implements it. +type NetworkAccessStatusSink interface { + IngressToken(installationID int) (string, bool) + ReportFor(installationID int, token string, status netaccess.Status) (previous netaccess.Status, changed bool, accepted bool) +} + +// NetworkAccessHostTimeout bounds one host's answer to a status read or a +// command, local or remote, like the node force-reload. +const NetworkAccessHostTimeout = 10 * time.Second + +// NetworkAccessNode is one enabled proxy node the admin operations fan out +// to. URL is the backend address the API dials with the node bearer. +type NetworkAccessNode struct { + ID int + Name string + URL string +} + +// Host returns the node's host identity: the instance-state scope +// "node:" as the id, role proxy, the node's name. +func (n NetworkAccessNode) Host() NetworkAccessHost { + return NetworkAccessHost{ID: NodeHostScope(int64(n.ID)), Role: pluginhost.HostRoleProxy, Name: n.Name} +} + +// NetworkAccessNodes reaches the provider instances running on proxy nodes. +// The API's node handler implements it over each node's bearer routes; a +// proxy node's own service has none and answers for itself alone. +type NetworkAccessNodes interface { + // ListNetworkAccessNodes returns every enabled proxy node. + ListNetworkAccessNodes(ctx context.Context) ([]NetworkAccessNode, error) + NodeNetworkAccessStatus(ctx context.Context, node NetworkAccessNode, provider string) (netaccess.Status, error) + NodeNetworkAccessConnect(ctx context.Context, node NetworkAccessNode, provider string) (netaccess.Status, error) + NodeNetworkAccessDisconnect(ctx context.Context, node NetworkAccessNode, provider string) (netaccess.Status, error) +} + +// SetNetworkAccessNodes wires the proxy-node fan-out into the admin status, +// connect and disconnect operations. Without it only the local host is +// reported. +func (s *Service) SetNetworkAccessNodes(nodes NetworkAccessNodes) { + if s == nil { + return + } + s.networkAccessNodes = nodes +} + +// SetNetworkAccessHostInfo supplies the identity this process reports as a +// host; main wires the same HostInfoFunc the plugin host answers GetHostInfo +// with. Without it the host is reported as role api with an empty name. +func (s *Service) SetNetworkAccessHostInfo(info pluginhost.HostInfoFunc) { + if s == nil { + return + } + s.networkAccessHostInfo = info +} + +// SetNetworkAccessStatusSink wires the status cache on-demand reads refresh. +func (s *Service) SetNetworkAccessStatusSink(sink NetworkAccessStatusSink) { + if s == nil { + return + } + s.networkAccessStatus = sink +} + +// ListNetworkAccessProviders returns every enabled installation declaring +// network_access_provider.v1, in installation id order. An installation whose +// manifest cannot be read is logged and skipped rather than hiding the rest. +func (s *Service) ListNetworkAccessProviders(ctx context.Context) ([]NetworkAccessProvider, error) { + if s == nil || s.installations == nil { + return nil, nil + } + installations, err := s.installations.ListEnabledWithCapabilityTypes(ctx, []string{capability.NetworkAccessProvider}) + if err != nil { + return nil, fmt.Errorf("list network access providers: %w", err) + } + providers := make([]NetworkAccessProvider, 0, len(installations)) + seen := make(map[string]int, len(installations)) + for _, installation := range installations { + if installation == nil || installation.IsBuiltin() { + continue + } + manifest, err := s.networkAccessManifest(ctx, installation) + if err != nil { + slog.WarnContext(ctx, "network access provider manifest unavailable; skipping", "component", "plugins", + "installation_id", installation.ID, "plugin_id", installation.PluginID, "error", err) + continue + } + descriptor, slug := pluginhost.NetworkAccessProviderCapability(manifest) + if descriptor == nil { + continue + } + // A slug names one provider per deployment. Installations are listed + // in id order, so the oldest enabled one owns the slug and later + // duplicates are skipped everywhere the slug is resolved and are + // excluded from the resident set. + if first, dup := seen[slug]; dup { + slog.WarnContext(ctx, "network access provider slug is declared by more than one enabled installation; using the first", "component", "plugins", + "provider", slug, "installation_id", first, "skipped_installation_id", installation.ID, "plugin_id", installation.PluginID) + continue + } + seen[slug] = installation.ID + providers = append(providers, NetworkAccessProvider{ + InstallationID: installation.ID, + CapabilityID: descriptor.GetId(), + Provider: slug, + DisplayName: networkAccessDisplayName(descriptor, slug), + }) + } + return providers, nil +} + +// networkAccessManifest keeps a running provider discoverable and commandable +// during a transient manifest read failure. Enabled rows still control membership; +// a process from an older version cannot stand in for a replacement release. +func (s *Service) networkAccessManifest(ctx context.Context, installation *Installation) (*pluginv1.PluginManifest, error) { + manifest, err := s.ensureLoadedInstallation(ctx, installation) + if err == nil || s.host == nil { + return manifest, err + } + client, clientErr := s.host.Client(installation.ID) + if clientErr == nil && manifestVersion(client.Manifest()) == installation.Version { + return client.Manifest(), nil + } + return nil, err +} + +func networkAccessDisplayName(descriptor *pluginv1.CapabilityDescriptor, slug string) string { + if name := strings.TrimSpace(descriptor.GetNetworkAccessProvider().GetDisplayName()); name != "" { + return name + } + if name := strings.TrimSpace(descriptor.GetDisplayName()); name != "" { + return name + } + return slug +} + +// networkAccessProvider resolves a slug to its installation. +func (s *Service) networkAccessProvider(ctx context.Context, slug string) (NetworkAccessProvider, error) { + slug = strings.TrimSpace(slug) + if slug == "" { + return NetworkAccessProvider{}, ErrNetworkAccessProviderNotFound + } + providers, err := s.ListNetworkAccessProviders(ctx) + if err != nil { + return NetworkAccessProvider{}, err + } + for _, provider := range providers { + if provider.Provider == slug { + return provider, nil + } + } + return NetworkAccessProvider{}, ErrNetworkAccessProviderNotFound +} + +// localNetworkAccessHost identifies this process as a host: the api host by +// default, or whatever HostInfo says (a proxy names itself node:). +func (s *Service) localNetworkAccessHost(ctx context.Context) NetworkAccessHost { + host := NetworkAccessHost{ID: HostScopeAPI, Role: pluginhost.HostRoleAPI} + if s.networkAccessHostInfo != nil { + if info, err := s.networkAccessHostInfo(ctx); err == nil { + if info.Role != "" { + host.Role = info.Role + } + host.Name = info.Name + if info.Role == pluginhost.HostRoleProxy && info.NodeID > 0 { + host.ID = NodeHostScope(info.NodeID) + } + } + } + return host +} + +// networkAccessTarget is one host an admin operation reaches: the local +// process, or a proxy node over its bearer routes. +type networkAccessTarget struct { + host NetworkAccessHost + node *NetworkAccessNode +} + +// networkAccessTargets lists the hosts this deployment runs provider +// instances on: this process first, then every enabled proxy node in id +// order. A node listing failure is reported, not hidden: an admin who reads +// "connected" on the api host alone must not conclude the proxies are too. +func (s *Service) networkAccessTargets(ctx context.Context) ([]networkAccessTarget, error) { + targets := []networkAccessTarget{{host: s.localNetworkAccessHost(ctx)}} + if s.networkAccessNodes == nil { + return targets, nil + } + nodes, err := s.networkAccessNodes.ListNetworkAccessNodes(ctx) + if err != nil { + return nil, fmt.Errorf("list network access proxy nodes: %w", err) + } + sort.Slice(nodes, func(i, j int) bool { return nodes[i].ID < nodes[j].ID }) + for _, node := range nodes { + node := node + targets = append(targets, networkAccessTarget{host: node.Host(), node: &node}) + } + return targets, nil +} + +// networkAccessCommand is one provider RPC applied to a host. +type networkAccessCommand func(ctx context.Context, client *pluginhost.NetworkAccessProviderClient) (*pluginv1.NetworkAccessStatus, error) + +// networkAccessOperation names what a report applies to each targeted host. +type networkAccessOperation int + +const ( + networkAccessRead networkAccessOperation = iota + networkAccessConnect + networkAccessDisconnect +) + +func (op networkAccessOperation) command() networkAccessCommand { + switch op { + case networkAccessConnect: + return func(ctx context.Context, client *pluginhost.NetworkAccessProviderClient) (*pluginv1.NetworkAccessStatus, error) { + return client.Connect(ctx, &pluginv1.NetworkAccessConnectRequest{}) + } + case networkAccessDisconnect: + return func(ctx context.Context, client *pluginhost.NetworkAccessProviderClient) (*pluginv1.NetworkAccessStatus, error) { + return client.Disconnect(ctx, &pluginv1.NetworkAccessDisconnectRequest{}) + } + } + return func(ctx context.Context, client *pluginhost.NetworkAccessProviderClient) (*pluginv1.NetworkAccessStatus, error) { + return client.GetStatus(ctx, &pluginv1.NetworkAccessGetStatusRequest{}) + } +} + +// NetworkAccessStatus reads the provider's live status on every host. +func (s *Service) NetworkAccessStatus(ctx context.Context, provider string) (NetworkAccessReport, error) { + return s.networkAccessReport(ctx, provider, nil, networkAccessRead) +} + +// ConnectNetworkAccess asks the provider on the named hosts (nil = every +// host) to bring the overlay up, and reports the status each host reached. +// Hosts not named answer their current status. +func (s *Service) ConnectNetworkAccess(ctx context.Context, provider string, hosts []string) (NetworkAccessReport, error) { + return s.networkAccessReport(ctx, provider, hosts, networkAccessConnect) +} + +// DisconnectNetworkAccess is the counterpart of ConnectNetworkAccess. +func (s *Service) DisconnectNetworkAccess(ctx context.Context, provider string, hosts []string) (NetworkAccessReport, error) { + return s.networkAccessReport(ctx, provider, hosts, networkAccessDisconnect) +} + +// networkAccessReport resolves the provider, checks the requested host ids, +// applies the operation to the targeted hosts (every host when targets is +// nil) and reads status from the rest. Hosts are contacted in parallel, each +// bounded by NetworkAccessHostTimeout, so one hung proxy delays the answer by +// ten seconds and hides nothing about the others. +func (s *Service) networkAccessReport(ctx context.Context, slug string, targets []string, op networkAccessOperation) (NetworkAccessReport, error) { + if s == nil { + return NetworkAccessReport{}, ErrNetworkAccessProviderNotFound + } + provider, err := s.networkAccessProvider(ctx, slug) + if err != nil { + return NetworkAccessReport{}, err + } + hosts, err := s.networkAccessTargets(ctx) + if err != nil { + return NetworkAccessReport{}, err + } + targeted := make(map[string]bool, len(targets)) + for _, id := range targets { + id = strings.TrimSpace(id) + known := false + for _, target := range hosts { + if target.host.ID == id { + known = true + break + } + } + if !known { + return NetworkAccessReport{}, fmt.Errorf("%w: %q", ErrNetworkAccessHostUnknown, id) + } + targeted[id] = true + } + report := NetworkAccessReport{Provider: provider, Hosts: make([]NetworkAccessHostStatus, len(hosts))} + var wg sync.WaitGroup + for i, target := range hosts { + apply := networkAccessRead + if targets == nil || targeted[target.host.ID] { + apply = op + } + wg.Add(1) + go func(i int, target networkAccessTarget, apply networkAccessOperation) { + defer wg.Done() + hostCtx, cancel := context.WithTimeout(ctx, NetworkAccessHostTimeout) + defer cancel() + var status netaccess.Status + if target.node == nil { + status = s.applyNetworkAccess(hostCtx, provider, apply.command()) + } else { + status = s.applyNodeNetworkAccess(hostCtx, *target.node, provider, apply) + } + report.Hosts[i] = NetworkAccessHostStatus{Host: target.host, Status: status} + }(i, target, apply) + } + wg.Wait() + return report, nil +} + +// applyNodeNetworkAccess runs one operation against the provider on a proxy +// node. The node answers for its own instance; a node that cannot be +// reached, refuses the bearer, or runs a build without the routes reports +// unavailable with the reason. +func (s *Service) applyNodeNetworkAccess(ctx context.Context, node NetworkAccessNode, provider NetworkAccessProvider, op networkAccessOperation) netaccess.Status { + unavailable := netaccess.Status{InstallationID: provider.InstallationID, Provider: provider.Provider, State: netaccess.StateUnavailable} + if s.networkAccessNodes == nil { + unavailable.Error = "proxy node fan-out is not configured" + return unavailable + } + var ( + status netaccess.Status + err error + ) + switch op { + case networkAccessConnect: + status, err = s.networkAccessNodes.NodeNetworkAccessConnect(ctx, node, provider.Provider) + case networkAccessDisconnect: + status, err = s.networkAccessNodes.NodeNetworkAccessDisconnect(ctx, node, provider.Provider) + default: + status, err = s.networkAccessNodes.NodeNetworkAccessStatus(ctx, node, provider.Provider) + } + if err != nil { + if errors.Is(err, ErrNetworkAccessProviderNotFound) { + unavailable.Error = "provider is not installed on this node yet" + } else { + unavailable.Error = err.Error() + } + return unavailable + } + if status.InstallationID == 0 { + status.InstallationID = provider.InstallationID + } + if status.Provider == "" { + status.Provider = provider.Provider + } + if status.State == "" { + status.State = netaccess.StateUnavailable + } + return status +} + +// HostNetworkAccessStatus lists the live status of every provider instance +// on this host alone; a proxy node answers the API's fan-out with it. +func (s *Service) HostNetworkAccessStatus(ctx context.Context) (netaccess.HostStatusReport, error) { + report := netaccess.HostStatusReport{Providers: []netaccess.Status{}} + if s == nil { + return report, nil + } + providers, err := s.ListNetworkAccessProviders(ctx) + if err != nil { + return report, err + } + for _, provider := range providers { + hostCtx, cancel := context.WithTimeout(ctx, NetworkAccessHostTimeout) + report.Providers = append(report.Providers, s.applyNetworkAccess(hostCtx, provider, networkAccessRead.command())) + cancel() + } + return report, nil +} + +// HostNetworkAccessProviderStatus reads one provider on this host without +// waiting for unrelated providers. The API uses this for its per-provider +// status fan-out to proxy nodes. +func (s *Service) HostNetworkAccessProviderStatus(ctx context.Context, slug string) (netaccess.Status, error) { + return s.hostNetworkAccess(ctx, slug, networkAccessRead) +} + +// HostNetworkAccessConnect brings the provider up on this host alone. +func (s *Service) HostNetworkAccessConnect(ctx context.Context, slug string) (netaccess.Status, error) { + return s.hostNetworkAccess(ctx, slug, networkAccessConnect) +} + +// HostNetworkAccessDisconnect tears the provider down on this host alone. +func (s *Service) HostNetworkAccessDisconnect(ctx context.Context, slug string) (netaccess.Status, error) { + return s.hostNetworkAccess(ctx, slug, networkAccessDisconnect) +} + +func (s *Service) hostNetworkAccess(ctx context.Context, slug string, op networkAccessOperation) (netaccess.Status, error) { + if s == nil { + return netaccess.Status{}, ErrNetworkAccessProviderNotFound + } + provider, err := s.networkAccessProvider(ctx, slug) + if err != nil { + return netaccess.Status{}, err + } + hostCtx, cancel := context.WithTimeout(ctx, NetworkAccessHostTimeout) + defer cancel() + return s.applyNetworkAccess(hostCtx, provider, op.command()), nil +} + +// applyNetworkAccess runs one RPC against the provider instance in this +// process. The process is never launched here: the resident supervisor owns +// that, and a host whose instance is not running answers unavailable with +// the supervisor's last error, so an admin sees why without a second read. +func (s *Service) applyNetworkAccess(ctx context.Context, provider NetworkAccessProvider, apply networkAccessCommand) netaccess.Status { + unavailable := netaccess.Status{InstallationID: provider.InstallationID, Provider: provider.Provider, State: netaccess.StateUnavailable} + // Every unavailable answer also replaces the cached status: whatever + // origin the instance last pushed is not being served by a process this + // host can reach, so the origin check and the node health report must + // stop advertising it. The plugin's next push restores it. + // The returned status carries no updated_at: the contract defines it as + // when the host last heard from the provider, which an unavailable + // answer is not. The cache stamps its own copy on Report. + var ingressToken string + if s.networkAccessStatus != nil { + ingressToken, _ = s.networkAccessStatus.IngressToken(provider.InstallationID) + } + fail := func(reason string) netaccess.Status { + unavailable.Error = reason + if s.networkAccessStatus != nil { + s.networkAccessStatus.ReportFor(provider.InstallationID, ingressToken, unavailable) + } + return unavailable + } + if s.host == nil { + return fail("plugin host is not running") + } + if reason := s.resident.GateError(); reason != "" { + return fail(reason) + } + pc, err := s.host.Client(provider.InstallationID) + if err != nil { + return fail(networkAccessUnavailableReason(err, s.RuntimeState(provider.InstallationID))) + } + client, err := pc.NetworkAccessProvider(provider.CapabilityID) + if err != nil { + return fail(err.Error()) + } + ingressToken = client.IngressToken() + reported, err := apply(ctx, client) + if err != nil { + return fail(err.Error()) + } + status := pluginhost.NetworkAccessStatusFromProto(provider.InstallationID, provider.Provider, reported) + status.UpdatedAt = time.Now() + if s.networkAccessStatus != nil { + s.networkAccessStatus.ReportFor(provider.InstallationID, ingressToken, status) + } + return status +} + +func networkAccessUnavailableReason(err error, state RuntimeState) string { + switch { + case errors.Is(err, pluginhost.ErrPluginUnhealthy): + return "plugin process is unhealthy" + case state.LastError != "": + return fmt.Sprintf("plugin process is %s: %s", state.State, state.LastError) + case state.State != "" && state.State != ResidentRunning: + return "plugin process is " + string(state.State) + } + return "plugin process is not running" +} diff --git a/internal/plugins/network_access_broker_test.go b/internal/plugins/network_access_broker_test.go new file mode 100644 index 0000000000..821ac5dfe0 --- /dev/null +++ b/internal/plugins/network_access_broker_test.go @@ -0,0 +1,89 @@ +package plugins + +import ( + "context" + "net" + "os/exec" + "strings" + "sync" + "testing" + "time" + + "github.com/hashicorp/go-hclog" + "github.com/hashicorp/go-plugin" + "google.golang.org/grpc" + + pluginv1 "github.com/Silo-Server/silo-plugin-sdk/pkg/pluginproto/silo/plugin/v1" + sdkruntime "github.com/Silo-Server/silo-plugin-sdk/pkg/pluginsdk/runtime" + "github.com/Silo-Server/silo-server/internal/pluginhost" +) + +type callbackListener struct { + net.Listener + accepted chan struct{} + once sync.Once +} + +func (l *callbackListener) Accept() (net.Conn, error) { + conn, err := l.Listener.Accept() + if err == nil { + l.once.Do(func() { close(l.accepted) }) + } + return conn, err +} + +// The plugin must dial the callback stream during BindHostBroker, before any +// capability RPC needs Host(). Observing that connection directly catches the +// lazy-dial regression without waiting out go-plugin's pending-stream timeout. +func TestNetworkAccessHostCallbackConnectsAtBind(t *testing.T) { + process := plugin.NewClient(&plugin.ClientConfig{ + HandshakeConfig: pluginhost.HandshakeConfig(), + AllowedProtocols: []plugin.Protocol{plugin.ProtocolGRPC}, + Cmd: exec.Command(buildResidentFixture(t)), + Plugins: sdkruntime.DefaultPluginSet(sdkruntime.CapabilityServers{}), + Logger: hclog.NewNullLogger(), + }) + t.Cleanup(process.Kill) + protocol, err := process.Client() + if err != nil { + t.Fatal(err) + } + raw, err := protocol.Dispense(sdkruntime.PluginSetName) + if err != nil { + t.Fatal(err) + } + client := raw.(*sdkruntime.Client) + broker := client.Broker() + id := broker.NextId() + listener, err := broker.Accept(id) + if err != nil { + t.Fatal(err) + } + t.Cleanup(func() { _ = listener.Close() }) + observed := &callbackListener{Listener: listener, accepted: make(chan struct{})} + server := grpc.NewServer() + pluginv1.RegisterRuntimeHostServer(server, pluginhost.NewRuntimeHostServerWithOptions(pluginhost.RuntimeHostOptions{ + HostInfo: func(context.Context) (pluginhost.HostInfo, error) { + return pluginhost.HostInfo{Role: pluginhost.HostRoleAPI, Name: "api"}, nil + }, + })) + t.Cleanup(server.Stop) + go func() { _ = server.Serve(observed) }() + ctx, cancel := context.WithTimeout(t.Context(), 5*time.Second) + defer cancel() + if _, err := client.Runtime().BindHostBroker(ctx, &pluginv1.BindHostBrokerRequest{BrokerId: id}); err != nil { + t.Fatal(err) + } + select { + case <-observed.accepted: + case <-ctx.Done(): + t.Fatal("plugin did not open its callback connection during binding") + } + status, err := client.NetworkAccessProvider().GetStatus(ctx, &pluginv1.NetworkAccessGetStatusRequest{}) + if err != nil { + t.Fatal(err) + } + if status.Error != "" || !strings.HasSuffix(status.ProviderVersion, "host-ok") { + t.Fatalf("host callback failed: %+v", status) + } +} diff --git a/internal/plugins/network_access_generation_test.go b/internal/plugins/network_access_generation_test.go new file mode 100644 index 0000000000..4fdf8ffc0c --- /dev/null +++ b/internal/plugins/network_access_generation_test.go @@ -0,0 +1,46 @@ +package plugins + +import ( + "context" + "testing" + + pluginv1 "github.com/Silo-Server/silo-plugin-sdk/pkg/pluginproto/silo/plugin/v1" + "github.com/Silo-Server/silo-server/internal/netaccess" + "github.com/Silo-Server/silo-server/internal/pluginhost" +) + +func TestNetworkAccessLateRPCDoesNotRestoreRevokedOrReplacedStatus(t *testing.T) { + f := newResidentFixture(t, ResidentOptions{}) + f.service.SetNetworkAccessStatusSink(f.broker) + ctx := context.Background() + f.service.StartResidents(ctx) + waitState(t, f.service, 5, "running", running) + provider, err := f.service.networkAccessProvider(ctx, "stub") + if err != nil { + t.Fatal(err) + } + token, ok := f.broker.IngressToken(5) + if !ok { + t.Fatal("no token issued") + } + f.service.applyNetworkAccess(ctx, provider, func(context.Context, *pluginhost.NetworkAccessProviderClient) (*pluginv1.NetworkAccessStatus, error) { + // A successful RPC response arrives, then the process stops before + // the caller can publish that response to the shared cache. + f.broker.Revoke(5, token) + return &pluginv1.NetworkAccessStatus{State: netaccess.StateConnected, Origin: "https://old.example.test"}, nil + }) + if _, ok := f.broker.Status.Get(5); ok { + t.Fatal("late RPC response restored a revoked origin") + } + fresh, err := f.broker.Issue(5, "stub") + if err != nil { + t.Fatal(err) + } + f.broker.ReportFor(5, fresh, netaccess.Status{InstallationID: 5, Provider: "stub", State: netaccess.StateDisconnected}) + f.service.applyNetworkAccess(ctx, provider, func(context.Context, *pluginhost.NetworkAccessProviderClient) (*pluginv1.NetworkAccessStatus, error) { + return &pluginv1.NetworkAccessStatus{State: netaccess.StateConnected, Origin: "https://old.example.test"}, nil + }) + if got, ok := f.broker.Status.Get(5); !ok || got.State != netaccess.StateDisconnected { + t.Fatalf("old client overwrote replacement status: %+v %v", got, ok) + } +} diff --git a/internal/plugins/network_access_nodes_test.go b/internal/plugins/network_access_nodes_test.go new file mode 100644 index 0000000000..1127e9a6e6 --- /dev/null +++ b/internal/plugins/network_access_nodes_test.go @@ -0,0 +1,392 @@ +package plugins + +import ( + "context" + "errors" + "sync" + "testing" + "time" + + "github.com/Silo-Server/silo-server/internal/cache" + "github.com/Silo-Server/silo-server/internal/netaccess" + "github.com/Silo-Server/silo-server/internal/pluginhost" +) + +// fakeNetworkAccessNodes stands in for the API's node handler: two proxies, +// one answering, one whose plugin is not installed yet, plus an unreachable +// one when set. +type fakeNetworkAccessNodes struct { + mu sync.Mutex + listErr error + unreachable bool + connected map[int]bool + calls []string +} + +func (f *fakeNetworkAccessNodes) ListNetworkAccessNodes(context.Context) ([]NetworkAccessNode, error) { + if f.listErr != nil { + return nil, f.listErr + } + return []NetworkAccessNode{{ID: 7, Name: "proxy-b", URL: "http://proxy-b"}, {ID: 3, Name: "proxy-a", URL: "http://proxy-a"}, {ID: 9, Name: "proxy-new", URL: "http://proxy-new"}}, nil +} + +func (f *fakeNetworkAccessNodes) answer(node NetworkAccessNode, provider, verb string) (netaccess.Status, error) { + f.mu.Lock() + defer f.mu.Unlock() + f.calls = append(f.calls, verb+":"+node.Host().ID) + if f.unreachable { + return netaccess.Status{}, errors.New("proxy node unreachable: dial tcp: connection refused") + } + if node.ID == 9 { + return netaccess.Status{}, ErrNetworkAccessProviderNotFound + } + if f.connected == nil { + f.connected = map[int]bool{} + } + switch verb { + case "connect": + f.connected[node.ID] = true + case "disconnect": + f.connected[node.ID] = false + } + status := netaccess.Status{InstallationID: 5, Provider: provider, State: netaccess.StateDisconnected, UpdatedAt: time.Now()} + if f.connected[node.ID] { + status.State = netaccess.StateConnected + status.Origin = "https://" + node.Name + ".stub.test" + } + return status, nil +} + +func (f *fakeNetworkAccessNodes) NodeNetworkAccessStatus(_ context.Context, node NetworkAccessNode, provider string) (netaccess.Status, error) { + return f.answer(node, provider, "status") +} + +func (f *fakeNetworkAccessNodes) NodeNetworkAccessConnect(_ context.Context, node NetworkAccessNode, provider string) (netaccess.Status, error) { + return f.answer(node, provider, "connect") +} + +func (f *fakeNetworkAccessNodes) NodeNetworkAccessDisconnect(_ context.Context, node NetworkAccessNode, provider string) (netaccess.Status, error) { + return f.answer(node, provider, "disconnect") +} + +func hostByID(t *testing.T, report NetworkAccessReport, id string) NetworkAccessHostStatus { + t.Helper() + for _, host := range report.Hosts { + if host.Host.ID == id { + return host + } + } + t.Fatalf("host %s missing from %+v", id, report.Hosts) + return NetworkAccessHostStatus{} +} + +// With proxy nodes wired, every admin operation reports the api host first +// and then each enabled proxy in id order, applies the command only to the +// named hosts, and turns a node's failure into that host's unavailable row +// without hiding the others. +func TestNetworkAccessFansOutToProxyNodes(t *testing.T) { + f := newResidentFixture(t, ResidentOptions{}) + ctx := context.Background() + f.service.SetNetworkAccessStatusSink(f.broker) + f.service.SetNetworkAccessHostInfo(func(context.Context) (pluginhost.HostInfo, error) { + return pluginhost.HostInfo{Role: pluginhost.HostRoleAPI, Name: "Living Room"}, nil + }) + nodes := &fakeNetworkAccessNodes{} + f.service.SetNetworkAccessNodes(nodes) + f.service.StartResidents(ctx) + waitState(t, f.service, 5, "running", running) + + report, err := f.service.NetworkAccessStatus(ctx, "stub") + if err != nil { + t.Fatal(err) + } + ids := make([]string, 0, len(report.Hosts)) + for _, host := range report.Hosts { + ids = append(ids, host.Host.ID) + } + if len(ids) != 4 || ids[0] != HostScopeAPI || ids[1] != "node:3" || ids[2] != "node:7" || ids[3] != "node:9" { + t.Fatalf("host order = %v", ids) + } + if host := hostByID(t, report, "node:3"); host.Host.Role != "proxy" || host.Host.Name != "proxy-a" || host.Status.State != netaccess.StateDisconnected { + t.Fatalf("node:3 = %+v", host) + } + if host := hostByID(t, report, "node:9"); host.Status.State != netaccess.StateUnavailable || host.Status.Error != "provider is not installed on this node yet" || host.Status.InstallationID != 5 { + t.Fatalf("node:9 = %+v", host) + } + + // Connect one proxy only: the api host and the other proxies are read, + // not commanded. + report, err = f.service.ConnectNetworkAccess(ctx, "stub", []string{"node:7"}) + if err != nil { + t.Fatal(err) + } + if host := hostByID(t, report, "node:7"); host.Status.State != netaccess.StateConnected || host.Status.Origin != "https://proxy-b.stub.test" { + t.Fatalf("node:7 after connect = %+v", host) + } + if host := hostByID(t, report, "node:3"); host.Status.State != netaccess.StateDisconnected { + t.Fatalf("node:3 after targeted connect = %+v", host) + } + if host := hostByID(t, report, HostScopeAPI); host.Status.State != netaccess.StateDisconnected { + t.Fatalf("api after targeted connect = %+v", host) + } + nodes.mu.Lock() + var connects []string + for _, call := range nodes.calls { + if call == "connect:node:7" || call == "connect:node:3" || call == "connect:node:9" { + connects = append(connects, call) + } + } + nodes.mu.Unlock() + if len(connects) != 1 || connects[0] != "connect:node:7" { + t.Fatalf("connect calls = %v", connects) + } + + // Connect everywhere, then disconnect everywhere. + report, err = f.service.ConnectNetworkAccess(ctx, "stub", nil) + if err != nil { + t.Fatal(err) + } + for _, id := range []string{HostScopeAPI, "node:3", "node:7"} { + if host := hostByID(t, report, id); host.Status.State != netaccess.StateConnected { + t.Fatalf("%s after connect all = %+v", id, host) + } + } + report, err = f.service.DisconnectNetworkAccess(ctx, "stub", nil) + if err != nil { + t.Fatal(err) + } + for _, id := range []string{HostScopeAPI, "node:3", "node:7"} { + if host := hostByID(t, report, id); host.Status.State != netaccess.StateDisconnected { + t.Fatalf("%s after disconnect all = %+v", id, host) + } + } + + if _, err := f.service.ConnectNetworkAccess(ctx, "stub", []string{"node:42"}); !errors.Is(err, ErrNetworkAccessHostUnknown) { + t.Fatalf("unknown node err = %v", err) + } + + // An unreachable proxy is that host's problem alone. + nodes.unreachable = true + report, err = f.service.NetworkAccessStatus(ctx, "stub") + if err != nil { + t.Fatal(err) + } + if host := hostByID(t, report, "node:3"); host.Status.State != netaccess.StateUnavailable || host.Status.Error == "" { + t.Fatalf("unreachable node:3 = %+v", host) + } + if host := hostByID(t, report, HostScopeAPI); host.Status.State != netaccess.StateDisconnected { + t.Fatalf("api beside unreachable node = %+v", host) + } + + // A node listing failure is an error: an admin must not read the api host + // alone as the whole deployment. + nodes.listErr = errors.New("database is away") + if _, err := f.service.NetworkAccessStatus(ctx, "stub"); err == nil { + t.Fatal("node listing failure was hidden") + } +} + +// A proxy's own service answers for itself: HostNetworkAccessStatus lists +// each provider instance here, connect and disconnect act on this host, and +// an unknown slug is ErrNetworkAccessProviderNotFound for the route's 404. +func TestHostNetworkAccessOperationsAnswerForThisHost(t *testing.T) { + f := newResidentFixture(t, ResidentOptions{}) + ctx := context.Background() + broker := f.broker + f.service.SetNetworkAccessStatusSink(broker) + + report, err := f.service.HostNetworkAccessStatus(ctx) + if err != nil { + t.Fatal(err) + } + if len(report.Providers) != 1 || report.Providers[0].State != netaccess.StateUnavailable { + t.Fatalf("status before start = %+v", report) + } + status, err := f.service.HostNetworkAccessProviderStatus(ctx, "stub") + if err != nil || status.Provider != "stub" || status.State != netaccess.StateUnavailable { + t.Fatalf("provider status before start = %+v, %v", status, err) + } + + f.service.StartResidents(ctx) + waitState(t, f.service, 5, "running", running) + + status, err = f.service.HostNetworkAccessConnect(ctx, "stub") + if err != nil || status.State != netaccess.StateConnected || status.Origin != "https://silo.stub.test" { + t.Fatalf("connect = %+v, %v", status, err) + } + if cached, ok := broker.Status.Get(5); !ok || !cached.Connected() { + t.Fatalf("connect did not refresh the status cache: %+v %v", cached, ok) + } + if report := broker.Status.NodeNetworkAccess(); report["stub"].Origin != "https://silo.stub.test" { + t.Fatalf("health report = %+v", report) + } + report, err = f.service.HostNetworkAccessStatus(ctx) + if err != nil || len(report.Providers) != 1 || report.Providers[0].State != netaccess.StateConnected { + t.Fatalf("status after connect = %+v, %v", report, err) + } + status, err = f.service.HostNetworkAccessProviderStatus(ctx, "stub") + if err != nil || status.Provider != "stub" || status.State != netaccess.StateConnected { + t.Fatalf("provider status after connect = %+v, %v", status, err) + } + if _, err := f.service.HostNetworkAccessProviderStatus(ctx, "netbird"); !errors.Is(err, ErrNetworkAccessProviderNotFound) { + t.Fatalf("unknown provider status err = %v", err) + } + status, err = f.service.HostNetworkAccessDisconnect(ctx, "stub") + if err != nil || status.State != netaccess.StateDisconnected { + t.Fatalf("disconnect = %+v, %v", status, err) + } + if _, err := f.service.HostNetworkAccessConnect(ctx, "netbird"); !errors.Is(err, ErrNetworkAccessProviderNotFound) || !errors.Is(err, netaccess.ErrProviderNotFound) { + t.Fatalf("unknown provider err = %v", err) + } +} + +// A closed resident gate keeps every resident stopped and names the reason +// in the admin status; opening it lets the next reconcile start them. +func TestResidentGateHoldsResidentsUntilItOpens(t *testing.T) { + f := newResidentFixture(t, ResidentOptions{}) + ctx := context.Background() + f.service.SetNetworkAccessStatusSink(f.broker) + var mu sync.Mutex + gateErr := errors.New("this proxy's stream_nodes row is not known yet") + f.service.SetResidentGate(func(context.Context) error { + mu.Lock() + defer mu.Unlock() + return gateErr + }) + + f.service.StartResidents(ctx) + if _, ok := f.service.Residents().State(5); ok { + t.Fatal("resident was tracked while the gate was closed") + } + if _, err := f.host.Client(5); !errors.Is(err, pluginhost.ErrClientNotFound) { + t.Fatalf("resident started behind a closed gate: %v", err) + } + report, err := f.service.NetworkAccessStatus(ctx, "stub") + if err != nil { + t.Fatal(err) + } + if got := report.Hosts[0].Status; got.State != netaccess.StateUnavailable || got.Error != gateErr.Error() { + t.Fatalf("status behind the gate = %+v", got) + } + + mu.Lock() + gateErr = nil + mu.Unlock() + f.service.OnLifecycleChange(ctx) + waitState(t, f.service, 5, "running after the gate opened", running) + if reason := f.service.Residents().GateError(); reason != "" { + t.Fatalf("gate error after opening = %q", reason) + } + + // Closing the gate again stops the resident. + mu.Lock() + gateErr = errors.New("row was deleted") + mu.Unlock() + f.service.OnLifecycleChange(ctx) + deadline := time.Now().Add(30 * time.Second) + for { + _, tracked := f.service.Residents().State(5) + _, clientErr := f.host.Client(5) + if !tracked && errors.Is(clientErr, pluginhost.ErrClientNotFound) { + break + } + if time.Now().After(deadline) { + t.Fatalf("resident still running behind a re-closed gate (tracked=%v, client err=%v)", tracked, clientErr) + } + time.Sleep(10 * time.Millisecond) + } +} + +// A following host reconciles on EventPluginsChanged: the publisher's +// lifecycle change starts the resident on the follower, a restart event +// replaces the follower's process, and a disable stops it. +func TestFollowLifecycleChangesReconcilesOnPublishedEvents(t *testing.T) { + bus := newFakeBus() + follower := newResidentFixture(t, ResidentOptions{}) + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + + // The follower's store starts with the installation disabled; the + // publisher (any service with the bus wired) announces changes. + disabled := false + follower.store.byID[5].Enabled = disabled + follower.service.StartResidents(ctx) + if _, ok := follower.service.Residents().State(5); ok { + t.Fatal("disabled installation was supervised") + } + if err := follower.service.FollowLifecycleChanges(ctx, bus, time.Hour); err != nil { + t.Fatal(err) + } + + publisher := &Service{} + publisher.PublishLifecycleChanges(bus) + + follower.store.byID[5].Enabled = true + publisher.OnLifecycleChange(ctx) + waitState(t, follower.service, 5, "running after the publisher's change", running) + first, err := follower.host.Client(5) + if err != nil { + t.Fatal(err) + } + + publisher.publishPluginsChanged(ctx, PluginsChangedEvent{InstallationID: 5, Restart: true}) + deadline := time.Now().Add(30 * time.Second) + for { + current, err := follower.host.Client(5) + state, _ := follower.service.Residents().State(5) + if err == nil && current != first && state.State == ResidentRunning { + break + } + if time.Now().After(deadline) { + t.Fatalf("follower did not replace its process on the restart event (err=%v, state=%+v)", err, state) + } + time.Sleep(10 * time.Millisecond) + } + + follower.store.byID[5].Enabled = false + publisher.OnLifecycleChange(ctx) + deadline = time.Now().Add(30 * time.Second) + for { + _, tracked := follower.service.Residents().State(5) + _, clientErr := follower.host.Client(5) + if !tracked && errors.Is(clientErr, pluginhost.ErrClientNotFound) { + break + } + if time.Now().After(deadline) { + t.Fatalf("follower kept the resident after the disable event (tracked=%v, err=%v)", tracked, clientErr) + } + time.Sleep(10 * time.Millisecond) + } + _ = cache.EventPluginsChanged +} + +// failingSubscribeBus is a bus whose subscription cannot be established, as +// when Redis is down while a proxy boots. +type failingSubscribeBus struct{ fakeBus } + +func (b *failingSubscribeBus) Subscribe(context.Context, string, cache.EventHandler) error { + return errors.New("redis unavailable") +} + +// A follower whose subscription fails still reconciles on the poll: the +// error is reported for logging, but the poll goroutine is started anyway so +// later changes reach the proxy without a restart. +func TestFollowLifecycleChangesPollsWhenTheSubscriptionFails(t *testing.T) { + bus := &failingSubscribeBus{fakeBus: *newFakeBus()} + follower := newResidentFixture(t, ResidentOptions{}) + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + + follower.store.byID[5].Enabled = false + follower.service.StartResidents(ctx) + if _, ok := follower.service.Residents().State(5); ok { + t.Fatal("disabled installation was supervised") + } + if err := follower.service.FollowLifecycleChanges(ctx, bus, 20*time.Millisecond); err == nil { + t.Fatal("subscription error was not reported") + } + + // Nothing publishes to this follower; only the poll can notice. + follower.store.byID[5].Enabled = true + waitState(t, follower.service, 5, "running after the poll noticed the change", running) +} diff --git a/internal/plugins/network_access_test.go b/internal/plugins/network_access_test.go new file mode 100644 index 0000000000..d812bcbdc3 --- /dev/null +++ b/internal/plugins/network_access_test.go @@ -0,0 +1,137 @@ +package plugins + +import ( + "context" + "errors" + "strings" + "testing" + + "github.com/Silo-Server/silo-server/internal/netaccess" + "github.com/Silo-Server/silo-server/internal/pluginhost" +) + +// TestNetworkAccessProvidersAndCommands drives the admin service layer +// against the real fixture plugin: providers are listed from the manifest +// without a launch, an installation whose process is not running answers +// unavailable, and once the supervisor has it running the status, connect +// and disconnect RPCs reach the plugin and refresh the status sink. +func TestNetworkAccessProvidersAndCommands(t *testing.T) { + f := newResidentFixture(t, ResidentOptions{}) + ctx := context.Background() + broker := f.broker + f.service.SetNetworkAccessStatusSink(broker) + f.service.SetNetworkAccessHostInfo(func(context.Context) (pluginhost.HostInfo, error) { + return pluginhost.HostInfo{Role: pluginhost.HostRoleAPI, Name: "Living Room"}, nil + }) + + providers, err := f.service.ListNetworkAccessProviders(ctx) + if err != nil { + t.Fatal(err) + } + if len(providers) != 1 || providers[0].InstallationID != 5 || providers[0].Provider != "stub" || providers[0].CapabilityID != "stub" || providers[0].DisplayName != "Resident Fixture" { + t.Fatalf("providers = %+v", providers) + } + if _, err := f.host.Client(5); !errors.Is(err, pluginhost.ErrClientNotFound) { + t.Fatalf("listing providers launched the plugin: %v", err) + } + + if _, err := f.service.NetworkAccessStatus(ctx, "netbird"); !errors.Is(err, ErrNetworkAccessProviderNotFound) { + t.Fatalf("unknown provider err = %v", err) + } + if _, err := f.service.ConnectNetworkAccess(ctx, "stub", []string{"node:3"}); !errors.Is(err, ErrNetworkAccessHostUnknown) { + t.Fatalf("unknown host err = %v", err) + } + + // Before the supervisor arms, the api host reports unavailable and the + // read does not start the process. + report, err := f.service.NetworkAccessStatus(ctx, "stub") + if err != nil { + t.Fatal(err) + } + if len(report.Hosts) != 1 || report.Hosts[0].Host.ID != HostScopeAPI || report.Hosts[0].Host.Role != "api" || report.Hosts[0].Host.Name != "Living Room" { + t.Fatalf("hosts = %+v", report.Hosts) + } + if got := report.Hosts[0].Status; got.State != netaccess.StateUnavailable || got.Error == "" || got.InstallationID != 5 || got.Provider != "stub" { + t.Fatalf("status before start = %+v", got) + } + if _, err := f.host.Client(5); !errors.Is(err, pluginhost.ErrClientNotFound) { + t.Fatalf("status read launched the plugin: %v", err) + } + + f.service.StartResidents(ctx) + waitState(t, f.service, 5, "running", running) + + report, err = f.service.NetworkAccessStatus(ctx, "stub") + if err != nil { + t.Fatal(err) + } + if got := report.Hosts[0].Status; got.State != netaccess.StateDisconnected || !strings.HasPrefix(got.ProviderVersion, "stub 0.1.0") || got.UpdatedAt.IsZero() { + t.Fatalf("status when running = %+v", got) + } + + report, err = f.service.ConnectNetworkAccess(ctx, "stub", nil) + if err != nil { + t.Fatal(err) + } + if got := report.Hosts[0].Status; got.State != netaccess.StateConnected || got.Origin != "https://silo.stub.test" || len(got.Listeners) != 1 || !got.DesiredConnected { + t.Fatalf("status after connect = %+v", got) + } + if cached, ok := broker.Status.Get(5); !ok || !cached.Connected() { + t.Fatalf("connect did not refresh the status sink: %+v %v", cached, ok) + } + if origins := broker.Status.ConnectedOrigins(); len(origins) != 1 || origins[0] != "https://silo.stub.test" { + t.Fatalf("connected origins = %v", origins) + } + + // Naming the api host explicitly applies the command to it too. + report, err = f.service.DisconnectNetworkAccess(ctx, "stub", []string{HostScopeAPI}) + if err != nil { + t.Fatal(err) + } + if got := report.Hosts[0].Status; got.State != netaccess.StateDisconnected || got.DesiredConnected { + t.Fatalf("status after disconnect = %+v", got) + } + if origins := broker.Status.ConnectedOrigins(); len(origins) != 0 { + t.Fatalf("connected origins after disconnect = %v", origins) + } +} + +// A provider process that is up but does not answer must not keep +// advertising its last connected origin: the failed read replaces the cached +// status with unavailable so the origin check and the node health report +// stop naming it. +func TestNetworkAccessFailedRPCReportsUnavailableToTheStatusSink(t *testing.T) { + f := newResidentFixture(t, ResidentOptions{}) + ctx := context.Background() + broker := f.broker + f.service.SetNetworkAccessStatusSink(broker) + f.service.SetNetworkAccessHostInfo(func(context.Context) (pluginhost.HostInfo, error) { + return pluginhost.HostInfo{Role: pluginhost.HostRoleAPI, Name: "Living Room"}, nil + }) + f.service.StartResidents(ctx) + waitState(t, f.service, 5, "running", running) + if _, err := f.service.ConnectNetworkAccess(ctx, "stub", nil); err != nil { + t.Fatal(err) + } + if origins := broker.Status.ConnectedOrigins(); len(origins) != 1 { + t.Fatalf("connected origins = %v", origins) + } + + // An already-expired context makes the RPC fail while the process and + // its cached connected status are both still in place. + expired, cancel := context.WithCancel(ctx) + cancel() + report, err := f.service.NetworkAccessStatus(expired, "stub") + if err != nil { + t.Fatal(err) + } + if got := report.Hosts[0].Status; got.State != netaccess.StateUnavailable || got.Error == "" { + t.Fatalf("status on failed RPC = %+v", got) + } + if origins := broker.Status.ConnectedOrigins(); len(origins) != 0 { + t.Fatalf("dead origin still advertised after a failed RPC: %v", origins) + } + if cached, ok := broker.Status.Get(5); !ok || cached.State != netaccess.StateUnavailable { + t.Fatalf("cache after failed RPC = %+v %v", cached, ok) + } +} diff --git a/internal/plugins/node_service.go b/internal/plugins/node_service.go new file mode 100644 index 0000000000..9dd9e1e0ab --- /dev/null +++ b/internal/plugins/node_service.go @@ -0,0 +1,29 @@ +package plugins + +import "context" + +// NewNodeService builds the plugin service a proxy node runs. A proxy hosts +// the same network access provider installations as the API server, each +// with its own overlay identity, and nothing else: no catalog, no installer, +// no admin routes, no metadata or other capability dispatch. Everything the +// returned service can do is what the resident supervisor and the node's +// network-access routes need: read enabled resident installations, rehydrate +// their archives from plugin_archives into cacheDir (this node's own plugin +// cache root; the install paths the API server recorded belong to its +// filesystem, not this one), launch them with their runtime configuration, +// and drive Connect/Disconnect/GetStatus. +// +// Lifecycle mutations still happen on the API server; the node learns about +// them through FollowLifecycleChanges. +func NewNodeService(installations *InstallationStore, configs *RuntimeConfigStore, host Host, cacheDir string) *Service { + svc := &Service{ + installations: installations, + configs: configs, + archiveCache: NewArchiveCacheAt(installations, cacheDir), + host: host, + } + svc.AddLifecycleHook(func(context.Context) { svc.invalidateInstallationCache() }) + svc.resident = newResidentSupervisor(svc, ResidentOptions{}) + svc.AddLifecycleHook(func(ctx context.Context) { svc.resident.Reconcile(ctx) }) + return svc +} diff --git a/internal/plugins/resident.go b/internal/plugins/resident.go new file mode 100644 index 0000000000..f63a9d6657 --- /dev/null +++ b/internal/plugins/resident.go @@ -0,0 +1,854 @@ +package plugins + +import ( + "context" + "errors" + "fmt" + "log/slog" + "math/rand/v2" + "sync" + "time" + + pluginv1 "github.com/Silo-Server/silo-plugin-sdk/pkg/pluginproto/silo/plugin/v1" + "github.com/Silo-Server/silo-plugin-sdk/pkg/pluginsdk/capability" + "github.com/Silo-Server/silo-server/internal/pluginhost" +) + +// ResidentState is one step of the resident supervisor's per-installation +// state machine: stopped → starting → running, with backoff after a crash +// or a failed start and failed once the consecutive-failure budget is spent. +type ResidentState string + +const ( + ResidentStopped ResidentState = "stopped" + ResidentStarting ResidentState = "starting" + ResidentRunning ResidentState = "running" + ResidentBackoff ResidentState = "backoff" + ResidentFailed ResidentState = "failed" +) + +// RuntimeState is the admin-visible view of one installation's process. +// Resident is true for installations the supervisor owns (started at boot, +// restarted on crash); the other fields describe the supervisor's machine. +// For a lazily started plugin State is running or stopped and the counters +// stay zero. +type RuntimeState struct { + Resident bool + State ResidentState + RestartCount int + LastError string + LastStartedAt *time.Time + NextRestartAt *time.Time +} + +const ( + DefaultResidentMinBackoff = time.Second + DefaultResidentMaxBackoff = time.Minute + DefaultResidentStableAfter = 5 * time.Minute + DefaultResidentMaxFailures = 10 +) + +// ErrNotResident reports a supervisor operation on an installation it does +// not own. +var ErrNotResident = errors.New("plugin installation is not resident") + +// residentCapabilityTypes lists the capability types whose installations +// must run as residents: started when the API listener is up, restarted +// after a crash, stopped before the HTTP drain. Network access providers are +// the first kind; later kinds are one more entry here. +var residentCapabilityTypes = []string{capability.NetworkAccessProvider} + +// IsResidentCapabilityType reports whether an installation declaring the +// capability type must run as a resident. +func IsResidentCapabilityType(capabilityType string) bool { + for _, resident := range residentCapabilityTypes { + if capabilityType == resident { + return true + } + } + return false +} + +func isResidentManifest(manifest *pluginv1.PluginManifest) bool { + for _, declared := range manifest.GetCapabilities() { + if IsResidentCapabilityType(declared.GetType()) { + return true + } + } + return false +} + +// ResidentOptions tunes the supervisor. Zero values take the defaults above. +type ResidentOptions struct { + MinBackoff time.Duration + MaxBackoff time.Duration + StableAfter time.Duration + MaxFailures int + Logger *slog.Logger + // now and jitter are test seams. + now func() time.Time + jitter func(time.Duration) time.Duration +} + +func (o ResidentOptions) withDefaults() ResidentOptions { + if o.MinBackoff <= 0 { + o.MinBackoff = DefaultResidentMinBackoff + } + if o.MaxBackoff <= 0 { + o.MaxBackoff = DefaultResidentMaxBackoff + } + if o.MaxBackoff < o.MinBackoff { + o.MaxBackoff = o.MinBackoff + } + if o.StableAfter <= 0 { + o.StableAfter = DefaultResidentStableAfter + } + if o.MaxFailures <= 0 { + o.MaxFailures = DefaultResidentMaxFailures + } + if o.Logger == nil { + o.Logger = slog.Default() + } + if o.now == nil { + o.now = time.Now + } + if o.jitter == nil { + o.jitter = defaultBackoffJitter + } + return o +} + +// defaultBackoffJitter spreads restarts by up to a quarter of the base delay +// so several residents that crashed together do not restart in lockstep. +func defaultBackoffJitter(base time.Duration) time.Duration { + if base <= 0 { + return 0 + } + return time.Duration(rand.Int64N(int64(base)/4 + 1)) +} + +// residentBackoff is the base delay before the failures-th consecutive +// restart: min doubling each time, capped at max. +func residentBackoff(failures int, minBackoff, maxBackoff time.Duration) time.Duration { + if failures < 1 { + failures = 1 + } + delay := minBackoff + for i := 1; i < failures; i++ { + delay *= 2 + if delay >= maxBackoff || delay <= 0 { + return maxBackoff + } + } + if delay > maxBackoff { + return maxBackoff + } + return delay +} + +type residentEntry struct { + id int + version string + installPath string + // hostIdentity is the host identity the running process was launched + // under (a proxy's instance-state scope). A change means the process + // holds another host's in-memory identity and is replaced. + hostIdentity string + runtimeGeneration int64 + + state ResidentState + failures int + restartCount int + lastError string + lastStartedAt time.Time + nextRestartAt time.Time + + timer *time.Timer + // gen invalidates an in-flight start or a pending timer when the entry + // is reset, halted, or superseded by a newer start. + gen uint64 +} + +func (e *residentEntry) snapshot() RuntimeState { + state := RuntimeState{Resident: true, State: e.state, RestartCount: e.restartCount, LastError: e.lastError} + if !e.lastStartedAt.IsZero() { + at := e.lastStartedAt + state.LastStartedAt = &at + } + if e.state == ResidentBackoff && !e.nextRestartAt.IsZero() { + at := e.nextRestartAt + state.NextRestartAt = &at + } + return state +} + +func (e *residentEntry) cancelTimer() { + if e.timer != nil { + e.timer.Stop() + e.timer = nil + } +} + +// ResidentSupervisor keeps resident installations running. It is armed by +// Enable once the API listener is bound, reconciles the desired set on every +// plugin lifecycle change, learns about crashes from the plugin host's exit +// handler, and is halted before the HTTP drain. +type ResidentSupervisor struct { + service *Service + opts ResidentOptions + + // reconcileMu serializes Reconcile end to end. The desired set and the + // gate result are computed outside mu (they read the database), so two + // overlapping reconciles could otherwise apply results in the wrong + // order: an older one resurrecting a resident a newer one just removed, + // or starting one after the gate closed. Lifecycle hooks, the proxy's + // event follower, and the poll all call Reconcile concurrently. + reconcileMu sync.Mutex + + mu sync.Mutex + armed bool + halted bool + ctx context.Context + entries map[int]*residentEntry + starts sync.WaitGroup + // gate, when set, decides whether this host may run residents at all. A + // proxy whose stream_nodes row is unknown has no instance-state scope to + // keep overlay node keys under, so it runs none until the row resolves. + gate func(ctx context.Context) error + gateErr string + // hostIdentity, when set, names the identity residents run under on this + // host; a proxy returns its node scope. Reconcile replaces every running + // resident when it changes, since a process keeps the identity it was + // started with (its overlay node key, its reported node id) in memory. + hostIdentity func() string +} + +// SetResidentGate installs a precondition every reconcile checks before it +// starts residents. While the gate returns an error no resident runs (running +// ones are stopped), the reason is logged once per distinct message, and the +// admin status reports it as the reason the host is unavailable. A nil gate +// clears it. +func (s *Service) SetResidentGate(gate func(ctx context.Context) error) { + if s == nil || s.resident == nil { + return + } + s.resident.mu.Lock() + s.resident.gate = gate + s.resident.mu.Unlock() +} + +// SetResidentHostIdentity installs the identity residents run under on this +// host. When the returned value changes between reconciles, every running +// resident is stopped and started again so no process keeps serving under +// a previous identity; a proxy passes its node scope so a deleted and +// re-registered row (new stream_nodes id) restarts its providers. +func (s *Service) SetResidentHostIdentity(identity func() string) { + if s == nil || s.resident == nil { + return + } + s.resident.mu.Lock() + s.resident.hostIdentity = identity + s.resident.mu.Unlock() +} + +// ResidentsArmed reports whether the supervisor has been enabled, after +// which its entries, not the manifests, say which installations are +// resident. +func (s *Service) ResidentsArmed() bool { + if s == nil || s.resident == nil { + return false + } + s.resident.mu.Lock() + defer s.resident.mu.Unlock() + return s.resident.armed +} + +// GateError returns why the host currently runs no residents, or "" when the +// gate (if any) is open. +func (r *ResidentSupervisor) GateError() string { + if r == nil { + return "" + } + r.mu.Lock() + defer r.mu.Unlock() + return r.gateErr +} + +func newResidentSupervisor(service *Service, opts ResidentOptions) *ResidentSupervisor { + return &ResidentSupervisor{ + service: service, + opts: opts.withDefaults(), + ctx: context.Background(), + entries: make(map[int]*residentEntry), + } +} + +// Enable arms the supervisor and runs the first reconcile. Reconcile is a +// no-op before this so the boot-time lifecycle hooks (preload, dispatcher +// backfill) do not launch residents before the listener they proxy to exists. +func (r *ResidentSupervisor) Enable(ctx context.Context) { + if r == nil { + return + } + r.mu.Lock() + r.armed = true + r.ctx = context.WithoutCancel(ctx) + r.mu.Unlock() + r.Reconcile(ctx) +} + +// Reconcile aligns the supervised set with the enabled installations that +// declare a resident capability: new residents start, residents that lost +// enablement or the capability stop, and a running resident whose process +// was stopped elsewhere (config save, replace, auto-update) starts again. A +// failed entry stays failed until Restart or Reset. +func (r *ResidentSupervisor) Reconcile(ctx context.Context) { + if r == nil { + return + } + r.reconcileMu.Lock() + defer r.reconcileMu.Unlock() + r.mu.Lock() + ready := r.armed && !r.halted + r.mu.Unlock() + if !ready { + return + } + + r.mu.Lock() + gate := r.gate + previousGateErr := r.gateErr + identity := r.hostIdentity + r.mu.Unlock() + hostIdentity := "" + if identity != nil { + hostIdentity = identity() + } + + var ( + desired map[int]*Installation + gateErr string + ) + if gate != nil { + if err := gate(ctx); err != nil { + gateErr = err.Error() + } + } + if gateErr == "" { + var err error + desired, err = r.desiredResidents(ctx) + if err != nil { + r.opts.Logger.WarnContext(ctx, "resident plugin reconcile skipped", "component", "plugins", "error", err) + return + } + } + + type relaunch struct { + id int + gen uint64 + } + var ( + stop []int + replace []relaunch + ) + r.mu.Lock() + if r.halted { + r.mu.Unlock() + return + } + r.gateErr = gateErr + if gateErr != previousGateErr { + if gateErr != "" { + r.opts.Logger.WarnContext(ctx, "resident plugins are not started on this host", "component", "plugins", "reason", gateErr) + } else if gate != nil { + r.opts.Logger.InfoContext(ctx, "resident plugins may start on this host", "component", "plugins") + } + } + for id, entry := range r.entries { + if _, keep := desired[id]; keep { + continue + } + entry.cancelTimer() + entry.gen++ + delete(r.entries, id) + stop = append(stop, id) + } + for id, installation := range desired { + entry, ok := r.entries[id] + if !ok { + entry = &residentEntry{id: id, version: installation.Version, installPath: installation.InstallPath, hostIdentity: hostIdentity, runtimeGeneration: installation.RuntimeGeneration, state: ResidentStopped} + r.entries[id] = entry + r.startLocked(entry) + continue + } + if entry.hostIdentity != hostIdentity { + r.opts.Logger.InfoContext(ctx, "host identity changed; replacing resident plugin process", "component", "plugins", + "installation_id", id, "previous_identity", entry.hostIdentity, "identity", hostIdentity) + } + if entry.version != installation.Version || entry.installPath != installation.InstallPath || entry.hostIdentity != hostIdentity || entry.runtimeGeneration != installation.RuntimeGeneration { + // A replaced or auto-updated binary gets a fresh failure budget, + // and the process built from the old binary is stopped before + // the new one starts. The host that made the change already + // stopped its own process; on any other host (a proxy node) it + // is still alive, and ensureClient would keep it because the + // row and the running manifest agree on nothing it checks. + entry.version, entry.installPath, entry.hostIdentity, entry.runtimeGeneration = installation.Version, installation.InstallPath, hostIdentity, installation.RuntimeGeneration + r.resetLocked(entry) + entry.state = ResidentStarting + entry.gen++ + // Registered under the lock, like Restart, so Halt always waits + // for this launch. + r.starts.Add(1) + replace = append(replace, relaunch{id: id, gen: entry.gen}) + continue + } + switch entry.state { + case ResidentStopped: + r.startLocked(entry) + case ResidentRunning: + if r.service.host == nil { + continue + } + if _, err := r.service.host.Client(id); errors.Is(err, pluginhost.ErrClientNotFound) { + r.opts.Logger.InfoContext(ctx, "resident plugin was stopped; starting again", "component", "plugins", "installation_id", id) + r.startLocked(entry) + } + } + } + r.mu.Unlock() + + for _, id := range stop { + r.stopProcess(ctx, id) + } + for _, launch := range replace { + r.opts.Logger.InfoContext(ctx, "resident plugin binary changed; replacing its process", "component", "plugins", "installation_id", launch.id) + go func(launch relaunch) { + defer r.starts.Done() + r.stopProcess(ctx, launch.id) + r.runStart(context.WithoutCancel(ctx), launch.id, launch.gen) + }(launch) + } +} + +func (r *ResidentSupervisor) desiredResidents(ctx context.Context) (map[int]*Installation, error) { + if r.service == nil || r.service.installations == nil { + return nil, nil + } + installations, err := r.service.installations.ListEnabledWithCapabilityTypes(ctx, residentCapabilityTypes) + if err != nil { + return nil, fmt.Errorf("list enabled resident installations: %w", err) + } + desired := make(map[int]*Installation, len(installations)) + // A provider slug is owned by the lowest enabled installation declaring + // it (ListNetworkAccessProviders). A later duplicate is not commanded, + // reported, or listed anywhere, so it must not run either: a resident + // nobody can disconnect would keep serving ingress unseen. Installations + // arrive in id order, so the first holder of a slug is the owner. + slugOwner := make(map[string]int, len(installations)) + for _, installation := range installations { + if installation == nil || installation.IsBuiltin() { + continue + } + manifest, err := r.service.networkAccessManifest(ctx, installation) + if err != nil { + r.opts.Logger.WarnContext(ctx, "resident plugin manifest unavailable; not starting it", "component", "plugins", + "installation_id", installation.ID, "plugin_id", installation.PluginID, "error", err) + continue + } + if _, slug := pluginhost.NetworkAccessProviderCapability(manifest); slug != "" { + if owner, dup := slugOwner[slug]; dup { + r.opts.Logger.WarnContext(ctx, "network access provider slug is already owned by another installation; not starting the duplicate", "component", "plugins", + "provider", slug, "installation_id", owner, "skipped_installation_id", installation.ID, "plugin_id", installation.PluginID) + continue + } + slugOwner[slug] = installation.ID + } + desired[installation.ID] = installation + } + return desired, nil +} + +// startLocked moves the entry to starting and launches the process on a +// goroutine so callers (admin requests, lifecycle hooks) do not wait on the +// plugin handshake. +func (r *ResidentSupervisor) startLocked(entry *residentEntry) { + entry.cancelTimer() + entry.state = ResidentStarting + entry.nextRestartAt = time.Time{} + entry.gen++ + gen := entry.gen + id := entry.id + ctx := r.ctx + r.starts.Add(1) + go func() { + defer r.starts.Done() + r.runStart(ctx, id, gen) + }() +} + +// runStart launches the plugin and records the outcome, unless the entry was +// reset, removed, or superseded while the launch was in flight. ensureClientForStart +// reuses a process a lazy RPC already started instead of replacing it. +func (r *ResidentSupervisor) runStart(ctx context.Context, id int, gen uint64) { + var err error + accepted := false + for attempt := 0; attempt < 3; attempt++ { + floor := r.service.host.NextStartSeq() + var client pluginClient + client, err = r.service.ensureClientForStart(ctx, id, true) + if err != nil { + break + } + // A client at or below the floor came from an older generation via the + // singleflight launch join; stop it and let the next attempt relaunch. + if c, ok := client.(interface{ StartSeq() uint64 }); ok && c.StartSeq() <= floor { + _ = r.service.host.Stop(id) + continue + } + accepted = true + break + } + if err == nil && !accepted { + err = fmt.Errorf("superseded launch adopted 3 times") + } + r.mu.Lock() + entry := r.entries[id] + if entry == nil || entry.gen != gen || r.halted { + // A superseded launch's process is stopped only when nothing will + // take it over: the entry is gone, halted, or parked (stopped or + // failed). While a newer generation is starting, that generation + // either adopts this process (start sequence above its floor) or + // stops and relaunches it, so stopping here would kill a process + // the newer generation may already have accepted. + orphaned := err == nil && (entry == nil || r.halted || entry.state == ResidentStopped || entry.state == ResidentFailed) + r.mu.Unlock() + if orphaned { + r.stopProcess(ctx, id) + } + return + } + if err == nil && r.service.host != nil { + // The process may have died between the handshake and this point; + // HandleExit ignored it because the entry was still starting. + if _, clientErr := r.service.host.Client(id); clientErr != nil { + err = clientErr + } + } + if err != nil { + r.failLocked(ctx, entry, fmt.Errorf("start: %w", err)) + r.mu.Unlock() + return + } + entry.state = ResidentRunning + entry.lastStartedAt = r.opts.now() + entry.lastError = "" + entry.nextRestartAt = time.Time{} + r.mu.Unlock() + r.opts.Logger.InfoContext(ctx, "resident plugin running", "component", "plugins", "installation_id", id) +} + +// failLocked counts one failure, resetting the count first when the plugin +// had been running stably, then either schedules a restart with exponential +// backoff plus jitter or parks the entry as failed. +func (r *ResidentSupervisor) failLocked(ctx context.Context, entry *residentEntry, cause error) { + now := r.opts.now() + if entry.state == ResidentRunning && !entry.lastStartedAt.IsZero() && now.Sub(entry.lastStartedAt) >= r.opts.StableAfter { + entry.failures = 0 + entry.restartCount = 0 + } + entry.failures++ + entry.lastError = cause.Error() + entry.cancelTimer() + if entry.failures >= r.opts.MaxFailures { + entry.state = ResidentFailed + entry.nextRestartAt = time.Time{} + entry.gen++ + r.opts.Logger.ErrorContext(ctx, "resident plugin failed; restart it from the admin API", "component", "plugins", + "installation_id", entry.id, "consecutive_failures", entry.failures, "error", cause) + return + } + delay := residentBackoff(entry.failures, r.opts.MinBackoff, r.opts.MaxBackoff) + delay += r.opts.jitter(delay) + entry.state = ResidentBackoff + entry.restartCount++ + entry.nextRestartAt = now.Add(delay) + entry.gen++ + gen := entry.gen + id := entry.id + entry.timer = time.AfterFunc(delay, func() { r.fireRestart(id, gen) }) + r.opts.Logger.WarnContext(ctx, "resident plugin stopped; restart scheduled", "component", "plugins", + "installation_id", id, "consecutive_failures", entry.failures, "delay", delay, "error", cause) +} + +func (r *ResidentSupervisor) fireRestart(id int, gen uint64) { + r.mu.Lock() + defer r.mu.Unlock() + entry := r.entries[id] + if entry == nil || entry.gen != gen || entry.state != ResidentBackoff || r.halted { + return + } + r.startLocked(entry) +} + +// HandleExit is the plugin host's exit handler: a resident whose process +// went away on its own is treated as a failure and rescheduled. Exits of +// non-resident plugins and of entries not in running state (a failed start +// reports through runStart) are ignored. +func (r *ResidentSupervisor) HandleExit(installationID int) { + if r == nil { + return + } + r.mu.Lock() + defer r.mu.Unlock() + entry := r.entries[installationID] + if entry == nil || entry.state != ResidentRunning || r.halted { + return + } + r.failLocked(r.ctx, entry, errors.New("plugin process exited")) +} + +// resetLocked clears the failure budget and parks the entry as stopped so the +// next reconcile starts it. Anything in flight for the entry is invalidated. +func (r *ResidentSupervisor) resetLocked(entry *residentEntry) { + entry.cancelTimer() + entry.gen++ + entry.failures = 0 + entry.restartCount = 0 + entry.lastError = "" + entry.nextRestartAt = time.Time{} + entry.state = ResidentStopped +} + +// Reset forgets an entry's failure history, for a reconfigure that should +// give a failed resident a fresh budget. The following Reconcile starts it. +func (r *ResidentSupervisor) Reset(installationID int) { + if r == nil { + return + } + r.mu.Lock() + defer r.mu.Unlock() + if entry := r.entries[installationID]; entry != nil { + r.resetLocked(entry) + } +} + +// Restart stops the resident's process, clears its failure budget and starts +// it again, waiting for the launch so the caller's next read sees running or +// the recorded failure. A launch failure is recorded in the entry (state +// backoff or failed, LastError set) rather than returned. ErrNotResident is +// returned for installations the supervisor does not own. A canceled caller +// stops waiting; the accepted restart still records its eventual result. +func (r *ResidentSupervisor) Restart(ctx context.Context, installationID int) error { + if r == nil { + return ErrNotResident + } + r.mu.Lock() + entry := r.entries[installationID] + if entry == nil || r.halted { + r.mu.Unlock() + return ErrNotResident + } + r.resetLocked(entry) + entry.state = ResidentStarting + entry.gen++ + gen := entry.gen + // Registered under the lock so Halt, which flips halted under the same + // lock before waiting, always sees this launch. + r.starts.Add(1) + r.mu.Unlock() + done := make(chan struct{}) + go func() { + defer r.starts.Done() + defer close(done) + launchCtx := context.WithoutCancel(ctx) + r.stopProcess(launchCtx, installationID) + r.runStart(launchCtx, installationID, gen) + }() + select { + case <-done: + return nil + case <-ctx.Done(): + return ctx.Err() + } +} + +// setRuntimeGeneration records the installation's persisted runtime +// generation on its entry, for a host that is about to restart the process +// itself and must not treat the bump it just wrote as someone else's. +func (r *ResidentSupervisor) setRuntimeGeneration(installationID int, generation int64) { + if r == nil { + return + } + r.mu.Lock() + defer r.mu.Unlock() + if entry := r.entries[installationID]; entry != nil { + entry.runtimeGeneration = generation + } +} + +// Halt stops every resident and refuses further starts. It runs before the +// HTTP servers drain so overlay ingress goes away first, and waits (bounded +// by ctx) for in-flight launches to settle. +func (r *ResidentSupervisor) Halt(ctx context.Context) error { + if r == nil { + return nil + } + r.mu.Lock() + r.halted = true + ids := make([]int, 0, len(r.entries)) + for id, entry := range r.entries { + entry.cancelTimer() + entry.gen++ + entry.state = ResidentStopped + entry.nextRestartAt = time.Time{} + ids = append(ids, id) + } + r.mu.Unlock() + + for _, id := range ids { + r.stopProcess(ctx, id) + } + done := make(chan struct{}) + go func() { + r.starts.Wait() + close(done) + }() + select { + case <-done: + return nil + case <-ctx.Done(): + return ctx.Err() + } +} + +// State reports the supervisor's view of one installation; ok is false when +// the installation is not resident. +func (r *ResidentSupervisor) State(installationID int) (RuntimeState, bool) { + if r == nil { + return RuntimeState{}, false + } + r.mu.Lock() + defer r.mu.Unlock() + entry := r.entries[installationID] + if entry == nil { + return RuntimeState{}, false + } + return entry.snapshot(), true +} + +func (r *ResidentSupervisor) stopProcess(ctx context.Context, installationID int) { + if r.service == nil || r.service.host == nil { + return + } + if err := r.service.host.Stop(installationID); err != nil && !errors.Is(err, pluginhost.ErrClientNotFound) { + r.opts.Logger.WarnContext(ctx, "stop resident plugin", "component", "plugins", "installation_id", installationID, "error", err) + } +} + +// Residents returns the resident supervisor. +func (s *Service) Residents() *ResidentSupervisor { + if s == nil { + return nil + } + return s.resident +} + +// StartResidents arms the supervisor once the API listener is bound. +func (s *Service) StartResidents(ctx context.Context) { + if s == nil { + return + } + s.resident.Enable(ctx) +} + +// StopResidents halts the supervisor and stops every resident; call it +// before the HTTP servers drain. +func (s *Service) StopResidents(ctx context.Context) error { + if s == nil { + return nil + } + return s.resident.Halt(ctx) +} + +// HandleResidentExit is wired as the plugin host's exit handler. +func (s *Service) HandleResidentExit(installationID int) { + if s == nil { + return + } + s.resident.HandleExit(installationID) +} + +// RuntimeState reports the process state of one installation for the admin +// API. A supervised resident answers from the supervisor; anything else is +// running or stopped as the host sees it, with Resident false. A caller that +// knows the installation's capabilities (a disabled resident has no +// supervisor entry) sets Resident from IsResidentCapabilityType itself. +func (s *Service) RuntimeState(installationID int) RuntimeState { + state := RuntimeState{State: ResidentStopped} + if s == nil { + return state + } + if tracked, ok := s.resident.State(installationID); ok { + return tracked + } + if s.host != nil { + if _, err := s.host.Client(installationID); err == nil { + state.State = ResidentRunning + } + } + return state +} + +// RestartInstallation stops the installation's process and, for a resident, +// starts it again immediately with a fresh failure budget. A non-resident +// plugin is only stopped; its next RPC launches it lazily. +func (s *Service) RestartInstallation(ctx context.Context, installationID int) error { + if s == nil { + return nil + } + if _, err := s.loadInstallation(ctx, installationID, true); err != nil { + return err + } + // The restart is durable: advancing runtime_generation lets a host whose + // lifecycle subscription missed the event (a proxy with Redis down) + // replace its process on the next poll, and clears a failed entry's + // budget there the same way the event would. + durable := false + if s.installations != nil { + if err := s.installations.Update(ctx, installationID, UpdateInstallationInput{Restart: true}); err != nil { + return fmt.Errorf("record plugin restart: %w", err) + } + s.invalidateInstallationCache() + if installation, err := s.loadInstallation(ctx, installationID, true); err == nil { + // This host restarts synchronously below; record the new + // generation on its entry so the next reconcile does not + // replace the fresh process a second time. + s.resident.setRuntimeGeneration(installationID, installation.RuntimeGeneration) + durable = true + } + } + // Every other host running the resident (proxy nodes) replaces its + // process too, so one admin action clears a failed instance everywhere. + // With the generation persisted a plain event is enough: the follower's + // reconcile sees the new generation and replaces its process once. A + // restart event on top of that would make it restart, then reconcile and + // replace the fresh process again for the generation it had not + // recorded. Without a durable generation the event carries the restart. + // Published whether or not this host runs the resident itself: the + // restart is recorded regardless of what this host does with it. + defer s.publishPluginsChanged(ctx, PluginsChangedEvent{InstallationID: installationID, Restart: !durable}) + err := s.resident.Restart(ctx, installationID) + if err == nil { + return nil + } + if !errors.Is(err, ErrNotResident) { + return err + } + if s.host == nil { + return nil + } + if err := s.host.Stop(installationID); err != nil && !errors.Is(err, pluginhost.ErrClientNotFound) { + return fmt.Errorf("stop plugin installation %d: %w", installationID, err) + } + return nil +} diff --git a/internal/plugins/resident_generation_test.go b/internal/plugins/resident_generation_test.go new file mode 100644 index 0000000000..f720b074c8 --- /dev/null +++ b/internal/plugins/resident_generation_test.go @@ -0,0 +1,158 @@ +package plugins + +import ( + "context" + "sync" + "sync/atomic" + "testing" + "time" + + pluginv1 "github.com/Silo-Server/silo-plugin-sdk/pkg/pluginproto/silo/plugin/v1" + "github.com/Silo-Server/silo-plugin-sdk/pkg/pluginsdk/capability" + + "github.com/Silo-Server/silo-server/internal/pluginhost" +) + +// seqFakeHost is a Host whose first Start assigns its start sequence and then +// parks until the test releases it, without registering the client. That is +// the window the real host has between spawning a process and completing the +// handshake, during which a newer supervisor generation can join the +// in-flight launch through Service.ensureClient's singleflight. +type seqFakeHost struct { + mu sync.Mutex + seq atomic.Uint64 + clients map[int]*seqFakeClient + starts int + stoppedSeqs []uint64 + gate chan struct{} + entered chan struct{} + release sync.Once + floorReads atomic.Int32 +} + +type seqFakeClient struct { + *pluginhost.Client + manifest *pluginv1.PluginManifest + seq uint64 +} + +func (c *seqFakeClient) Manifest() *pluginv1.PluginManifest { return c.manifest } +func (c *seqFakeClient) StartSeq() uint64 { return c.seq } + +func newSeqFakeHost() *seqFakeHost { + return &seqFakeHost{clients: map[int]*seqFakeClient{}, gate: make(chan struct{}), entered: make(chan struct{})} +} + +func (h *seqFakeHost) Start(_ context.Context, req pluginhost.StartRequest) (pluginClient, error) { + seq := h.seq.Add(1) + h.mu.Lock() + h.starts++ + first := h.starts == 1 + h.mu.Unlock() + if first { + close(h.entered) + <-h.gate + } + client := &seqFakeClient{manifest: req.Manifest, seq: seq} + h.mu.Lock() + h.clients[req.InstallationID] = client + h.mu.Unlock() + return client, nil +} + +func (h *seqFakeHost) Client(installationID int) (pluginClient, error) { + h.mu.Lock() + defer h.mu.Unlock() + if client, ok := h.clients[installationID]; ok { + return client, nil + } + return nil, pluginhost.ErrClientNotFound +} + +func (h *seqFakeHost) Stop(installationID int) error { + h.mu.Lock() + defer h.mu.Unlock() + client, ok := h.clients[installationID] + if !ok { + return pluginhost.ErrClientNotFound + } + h.stoppedSeqs = append(h.stoppedSeqs, client.seq) + delete(h.clients, installationID) + return nil +} + +func (h *seqFakeHost) Shutdown(context.Context) error { return nil } + +// NextStartSeq mirrors pluginhost.Host. The second floor read is the newer +// generation's runStart, one instruction before it joins the in-flight +// launch, so releasing the parked leader here makes the join as likely as it +// can be without a hook inside singleflight. The assertions hold either way: +// a generation that misses the join relaunches through manifest drift. +func (h *seqFakeHost) NextStartSeq() uint64 { + if h.floorReads.Add(1) == 2 { + h.release.Do(func() { close(h.gate) }) + } + return h.seq.Load() +} + +func (h *seqFakeHost) snapshot() (starts int, stopped []uint64) { + h.mu.Lock() + defer h.mu.Unlock() + return h.starts, append([]uint64(nil), h.stoppedSeqs...) +} + +func TestResidentSupervisorNewGenerationNeverAdoptsInFlightOlderLaunch(t *testing.T) { + old := buildResidentFixture(t) + store := newFakeServiceInstallationStore(&Installation{ID: 5, PluginID: "silo.test.resident", Version: "0.1.0", InstallPath: old, Enabled: true, Kind: KindPlugin}) + store.listCapabilities = []*Capability{{InstallationID: 5, Type: capability.NetworkAccessProvider, ID: "stub"}} + host := newSeqFakeHost() + service := &Service{installations: store, host: host} + service.resident = newResidentSupervisor(service, ResidentOptions{MinBackoff: 10 * time.Millisecond, MaxBackoff: 50 * time.Millisecond}) + service.AddLifecycleHook(func(context.Context) { service.invalidateInstallationCache() }) + service.AddLifecycleHook(func(ctx context.Context) { service.resident.Reconcile(ctx) }) + t.Cleanup(func() { + host.release.Do(func() { close(host.gate) }) + ctx, cancel := context.WithTimeout(context.Background(), 10*time.Second) + defer cancel() + _ = service.StopResidents(ctx) + }) + + ctx := context.Background() + service.StartResidents(ctx) + select { + case <-host.entered: + case <-time.After(10 * time.Second): + t.Fatal("first launch never reached the host") + } + + // The row changes while generation 1's launch is parked before it + // registered anything, exactly like an auto-update racing boot. + next := buildResidentFixtureVersion(t, "0.2.0") + version := "0.2.0" + if err := store.Update(ctx, 5, UpdateInstallationInput{Version: &version, InstallPath: &next}); err != nil { + t.Fatal(err) + } + service.OnLifecycleChange(ctx) + + state := waitState(t, service, 5, "generation 2 running", running) + if state.LastError != "" || state.RestartCount != 0 { + t.Fatalf("replacement recorded as a failure: %+v", state) + } + current, err := host.Client(5) + if err != nil { + t.Fatal(err) + } + if got := current.Manifest().GetVersion(); got != "0.2.0" { + t.Fatalf("running version = %q, want the generation 2 release", got) + } + if seq := current.(*seqFakeClient).seq; seq != 2 { + t.Fatalf("running client start seq = %d, want 2 (the launch generation 2 issued)", seq) + } + starts, stopped := host.snapshot() + if starts != 2 { + t.Fatalf("host starts = %d, want 2", starts) + } + if len(stopped) != 1 || stopped[0] != 1 { + t.Fatalf("stopped seqs = %v, want the adopted generation 1 process (seq 1) stopped exactly once", stopped) + } +} diff --git a/internal/plugins/resident_ingress_test.go b/internal/plugins/resident_ingress_test.go new file mode 100644 index 0000000000..db0fff0c77 --- /dev/null +++ b/internal/plugins/resident_ingress_test.go @@ -0,0 +1,68 @@ +package plugins + +import ( + "context" + "testing" + "time" + + "github.com/hashicorp/go-hclog" + "google.golang.org/protobuf/encoding/protojson" + + pluginv1 "github.com/Silo-Server/silo-plugin-sdk/pkg/pluginproto/silo/plugin/v1" + + "github.com/Silo-Server/silo-server/internal/netaccess" + "github.com/Silo-Server/silo-server/internal/pluginhost" +) + +// Starting a network access provider issues its ingress token before the +// plugin can ask for it; stopping the process revokes it and forgets its +// status, so a request still carrying the old token is refused. +func TestHostStartIssuesAndStopRevokesIngressToken(t *testing.T) { + bin := buildResidentFixture(t) + manifest, err := LoadManifestFile(InstalledManifestPath(bin)) + if err != nil { + t.Fatal(err) + } + raw, _ := protojson.Marshal(manifest) + live := &pluginv1.PluginManifest{} + if err := protojson.Unmarshal(raw, live); err != nil { + t.Fatal(err) + } + + broker := netaccess.NewBroker() + host := pluginhost.NewHost(pluginhost.Config{Logger: hclog.NewNullLogger(), NetworkAccess: broker}) + ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second) + defer cancel() + if _, err := host.Start(ctx, pluginhost.StartRequest{InstallationID: 5, BinaryPath: bin, Manifest: live}); err != nil { + t.Fatalf("host.Start: %v", err) + } + t.Cleanup(func() { _ = host.Stop(5) }) + + token, ok := broker.IngressToken(5) + if !ok || token == "" { + t.Fatal("no ingress token issued for the running provider") + } + if ingress, ok := broker.Registry.Lookup(token); !ok || ingress.Provider != "stub" || ingress.InstallationID != 5 { + t.Fatalf("token resolves to %+v, %v", ingress, ok) + } + broker.ReportFor(5, token, netaccess.Status{InstallationID: 5, Provider: "stub", State: netaccess.StateConnected, Origin: "https://stub.example"}) + + if err := host.Stop(5); err != nil { + t.Fatalf("host.Stop: %v", err) + } + if _, ok := broker.Registry.Lookup(token); ok { + t.Fatal("token still valid after the process stopped") + } + if _, ok := broker.Status.Get(5); ok { + t.Fatal("status survived the process stop") + } + + // A restart rotates the token. + if _, err := host.Start(ctx, pluginhost.StartRequest{InstallationID: 5, BinaryPath: bin, Manifest: live}); err != nil { + t.Fatalf("host.Start again: %v", err) + } + rotated, ok := broker.IngressToken(5) + if !ok || rotated == token { + t.Fatalf("token after restart = %q (ok=%v), want a fresh one", rotated, ok) + } +} diff --git a/internal/plugins/resident_test.go b/internal/plugins/resident_test.go new file mode 100644 index 0000000000..a20cf73ff8 --- /dev/null +++ b/internal/plugins/resident_test.go @@ -0,0 +1,682 @@ +package plugins + +import ( + "context" + "errors" + "os" + "os/exec" + "path/filepath" + "sync" + "sync/atomic" + "testing" + "time" + + "github.com/hashicorp/go-hclog" + "google.golang.org/protobuf/encoding/protojson" + + pluginv1 "github.com/Silo-Server/silo-plugin-sdk/pkg/pluginproto/silo/plugin/v1" + "github.com/Silo-Server/silo-plugin-sdk/pkg/pluginsdk/capability" + + "github.com/Silo-Server/silo-server/internal/netaccess" + "github.com/Silo-Server/silo-server/internal/pluginhost" +) + +// buildResidentFixture compiles testdata/residentplugin into a temp install +// dir, writes the manifest the binary reports (checksum included) next to +// it as manifest.json, and returns the binary path. +func buildResidentFixture(t *testing.T) string { + t.Helper() + dir := t.TempDir() + bin := filepath.Join(dir, "residentplugin") + build := exec.Command("go", "build", "-o", bin, "./testdata/residentplugin") + if out, err := build.CombinedOutput(); err != nil { + t.Fatalf("build residentplugin: %v\n%s", err, out) + } + raw, err := exec.Command(bin, "manifest").Output() + if err != nil { + t.Fatalf("read fixture manifest: %v", err) + } + manifest := &pluginv1.PluginManifest{} + if err := protojson.Unmarshal(raw, manifest); err != nil { + t.Fatalf("decode fixture manifest: %v", err) + } + if err := os.WriteFile(InstalledManifestPath(bin), raw, 0o644); err != nil { + t.Fatal(err) + } + return bin +} + +type residentFixture struct { + service *Service + store *fakeServiceInstallationStore + host *pluginhost.Host + exit string + broker *netaccess.Broker +} + +func newResidentFixture(t *testing.T, opts ResidentOptions) *residentFixture { + t.Helper() + bin := buildResidentFixture(t) + exitFile := filepath.Join(t.TempDir(), "exit-now") + t.Setenv("SILO_TEST_PLUGIN_EXIT_FILE", exitFile) + + store := newFakeServiceInstallationStore(&Installation{ + ID: 5, + PluginID: "silo.test.resident", + Version: "0.1.0", + InstallPath: bin, + Enabled: true, + Kind: KindPlugin, + }) + store.listCapabilities = []*Capability{{InstallationID: 5, Type: capability.NetworkAccessProvider, ID: "stub"}} + + broker := netaccess.NewBroker() + host := pluginhost.NewHost(pluginhost.Config{ + NetworkAccess: broker, + Logger: hclog.NewNullLogger(), + ExitCheckInterval: 20 * time.Millisecond, + // Bind the RuntimeHost broker so fixtures can call back into the host. + HostInfo: func(context.Context) (pluginhost.HostInfo, error) { + return pluginhost.HostInfo{Role: pluginhost.HostRoleAPI, Name: "fixture"}, nil + }, + }) + service := &Service{installations: store, host: NewHostAdapter(host)} + if opts.MinBackoff == 0 { + opts.MinBackoff = 10 * time.Millisecond + } + if opts.MaxBackoff == 0 { + opts.MaxBackoff = 50 * time.Millisecond + } + service.resident = newResidentSupervisor(service, opts) + // Same hook order as NewService and NewNodeService: the row cache is + // dropped before the supervisor reads the rows. + service.AddLifecycleHook(func(context.Context) { service.invalidateInstallationCache() }) + service.AddLifecycleHook(func(ctx context.Context) { service.resident.Reconcile(ctx) }) + host.SetExitHandler(service.HandleResidentExit) + t.Cleanup(func() { + ctx, cancel := context.WithTimeout(context.Background(), 10*time.Second) + defer cancel() + _ = service.StopResidents(ctx) + _ = host.Shutdown(ctx) + }) + return &residentFixture{service: service, store: store, host: host, exit: exitFile, broker: broker} +} + +func (f *residentFixture) crash(t *testing.T) { + t.Helper() + if err := os.WriteFile(f.exit, []byte("x"), 0o644); err != nil { + t.Fatal(err) + } +} + +func (f *residentFixture) heal(t *testing.T) { + t.Helper() + if err := os.Remove(f.exit); err != nil && !errors.Is(err, os.ErrNotExist) { + t.Fatal(err) + } +} + +// waitState polls the supervisor until pred accepts the installation's +// state or the deadline passes. +func waitState(t *testing.T, service *Service, id int, what string, pred func(RuntimeState, bool) bool) RuntimeState { + t.Helper() + deadline := time.Now().Add(30 * time.Second) + for { + state, ok := service.Residents().State(id) + if pred(state, ok) { + return state + } + if time.Now().After(deadline) { + t.Fatalf("timed out waiting for %s; last state %+v (tracked=%v)", what, state, ok) + } + time.Sleep(10 * time.Millisecond) + } +} + +func running(state RuntimeState, ok bool) bool { return ok && state.State == ResidentRunning } + +func TestResidentSupervisorStartsAtReconcileAndStopsOnDisable(t *testing.T) { + f := newResidentFixture(t, ResidentOptions{}) + ctx := context.Background() + + // Lifecycle hooks before the API listener is bound must not launch. + f.service.OnLifecycleChange(ctx) + if _, ok := f.service.Residents().State(5); ok { + t.Fatal("supervisor tracked the installation before StartResidents") + } + if got := f.service.RuntimeState(5); got.State != ResidentStopped { + t.Fatalf("RuntimeState before start = %+v, want stopped", got) + } + + f.service.StartResidents(ctx) + state := waitState(t, f.service, 5, "running after StartResidents", running) + if !state.Resident || state.RestartCount != 0 || state.LastStartedAt == nil || state.LastError != "" { + t.Fatalf("running state = %+v", state) + } + if _, err := f.host.Client(5); err != nil { + t.Fatalf("host.Client after start: %v", err) + } + if got := f.service.RuntimeState(5); got.State != ResidentRunning || !got.Resident { + t.Fatalf("RuntimeState = %+v", got) + } + + // Disable: the next reconcile stops the process and forgets the entry. + disabled := false + if err := f.store.Update(ctx, 5, UpdateInstallationInput{Enabled: &disabled}); err != nil { + t.Fatal(err) + } + f.service.OnLifecycleChange(ctx) + if _, ok := f.service.Residents().State(5); ok { + t.Fatal("supervisor still tracks a disabled installation") + } + if _, err := f.host.Client(5); !errors.Is(err, pluginhost.ErrClientNotFound) { + t.Fatalf("host.Client after disable = %v, want ErrClientNotFound", err) + } + if got := f.service.RuntimeState(5); got.State != ResidentStopped { + t.Fatalf("RuntimeState after disable = %+v", got) + } + + // Re-enable: reconcile starts it again from a clean entry. + enabled := true + if err := f.store.Update(ctx, 5, UpdateInstallationInput{Enabled: &enabled}); err != nil { + t.Fatal(err) + } + f.service.OnLifecycleChange(ctx) + waitState(t, f.service, 5, "running after re-enable", running) +} + +func TestResidentSupervisorRestartsAfterCrashWithBackoff(t *testing.T) { + f := newResidentFixture(t, ResidentOptions{MaxFailures: 10}) + ctx := context.Background() + f.service.StartResidents(ctx) + first := waitState(t, f.service, 5, "initial running", running) + + f.crash(t) + backoff := waitState(t, f.service, 5, "backoff after crash", func(s RuntimeState, ok bool) bool { + return ok && s.State != ResidentRunning + }) + f.heal(t) + if backoff.RestartCount < 1 || backoff.LastError == "" { + t.Fatalf("state after crash = %+v", backoff) + } + if backoff.State == ResidentBackoff && backoff.NextRestartAt == nil { + t.Fatalf("backoff without NextRestartAt: %+v", backoff) + } + + again := waitState(t, f.service, 5, "running after restart", func(s RuntimeState, ok bool) bool { + return running(s, ok) && s.LastStartedAt != nil && s.LastStartedAt.After(*first.LastStartedAt) + }) + if again.RestartCount < 1 || again.LastError != "" || again.NextRestartAt != nil { + t.Fatalf("state after restart = %+v", again) + } + if _, err := f.host.Client(5); err != nil { + t.Fatalf("host.Client after restart: %v", err) + } +} + +func TestResidentSupervisorFailsAfterConsecutiveFailuresAndAdminRestartRecovers(t *testing.T) { + f := newResidentFixture(t, ResidentOptions{MaxFailures: 3}) + ctx := context.Background() + f.service.StartResidents(ctx) + waitState(t, f.service, 5, "initial running", running) + + // The exit trigger stays in place, so every relaunch dies again. + f.crash(t) + failed := waitState(t, f.service, 5, "failed state", func(s RuntimeState, ok bool) bool { + return ok && s.State == ResidentFailed + }) + if failed.LastError == "" || failed.NextRestartAt != nil { + t.Fatalf("failed state = %+v", failed) + } + if failed.RestartCount != 2 { + t.Fatalf("restart_count in failed state = %d, want 2 (MaxFailures-1)", failed.RestartCount) + } + // Reconcile does not resurrect a failed resident. + f.service.OnLifecycleChange(ctx) + if s, _ := f.service.Residents().State(5); s.State != ResidentFailed { + t.Fatalf("state after reconcile = %+v, want failed", s) + } + + f.heal(t) + if err := f.service.RestartInstallation(ctx, 5); err != nil { + t.Fatalf("RestartInstallation: %v", err) + } + state, ok := f.service.Residents().State(5) + if !ok || state.State != ResidentRunning || state.RestartCount != 0 || state.LastError != "" { + t.Fatalf("state after admin restart = %+v (tracked=%v)", state, ok) + } + if _, err := f.host.Client(5); err != nil { + t.Fatalf("host.Client after admin restart: %v", err) + } +} + +func TestResidentLazyRPCDoesNotBypassFailureBudget(t *testing.T) { + for _, state := range []ResidentState{ResidentFailed, ResidentBackoff} { + t.Run(string(state), func(t *testing.T) { + maxFailures := 1 + if state == ResidentBackoff { + maxFailures = 2 + } + f := newResidentFixture(t, ResidentOptions{ + MaxFailures: maxFailures, MinBackoff: time.Hour, MaxBackoff: time.Hour, + }) + f.service.StartResidents(t.Context()) + waitState(t, f.service, 5, "initial running", running) + f.crash(t) + waitState(t, f.service, 5, "parked after crash", func(s RuntimeState, ok bool) bool { + return ok && s.State == state + }) + f.heal(t) + // Every lazy capability RPC obtains its process through this path, + // before checking whether the manifest declares that capability. + if _, err := f.service.MetadataProviderClient(t.Context(), 5, "metadata"); !errors.Is(err, pluginhost.ErrPluginUnhealthy) { + t.Errorf("lazy RPC error = %v, want ErrPluginUnhealthy", err) + } + if _, err := f.host.Client(5); !errors.Is(err, pluginhost.ErrClientNotFound) { + t.Fatalf("lazy RPC revived the parked resident: %v", err) + } + if err := f.service.RestartInstallation(t.Context(), 5); err != nil { + t.Fatal(err) + } + waitState(t, f.service, 5, "running after admin restart", running) + }) + } +} + +func TestResidentSupervisorHaltStopsResidents(t *testing.T) { + f := newResidentFixture(t, ResidentOptions{}) + ctx := context.Background() + f.service.StartResidents(ctx) + waitState(t, f.service, 5, "initial running", running) + + if err := f.service.StopResidents(ctx); err != nil { + t.Fatalf("StopResidents: %v", err) + } + if _, err := f.host.Client(5); !errors.Is(err, pluginhost.ErrClientNotFound) { + t.Fatalf("host.Client after halt = %v, want ErrClientNotFound", err) + } + // Halted: neither a reconcile nor an admin restart may relaunch. + f.service.OnLifecycleChange(ctx) + if err := f.service.Residents().Restart(ctx, 5); !errors.Is(err, ErrNotResident) { + t.Fatalf("Restart after halt = %v, want ErrNotResident", err) + } + if _, err := f.host.Client(5); !errors.Is(err, pluginhost.ErrClientNotFound) { + t.Fatalf("host.Client after halted reconcile = %v, want ErrClientNotFound", err) + } +} + +func TestResidentBackoffSchedule(t *testing.T) { + minBackoff, maxBackoff := time.Second, time.Minute + for _, tc := range []struct { + failures int + want time.Duration + }{{0, time.Second}, {1, time.Second}, {2, 2 * time.Second}, {3, 4 * time.Second}, {6, 32 * time.Second}, {7, time.Minute}, {10, time.Minute}, {64, time.Minute}} { + if got := residentBackoff(tc.failures, minBackoff, maxBackoff); got != tc.want { + t.Errorf("residentBackoff(%d) = %v, want %v", tc.failures, got, tc.want) + } + } + for i := 0; i < 100; i++ { + if j := defaultBackoffJitter(time.Minute); j < 0 || j > 15*time.Second { + t.Fatalf("jitter %v outside [0, 15s]", j) + } + } +} + +func TestResidentPredicateNeedsCapability(t *testing.T) { + if !IsResidentCapabilityType(capability.NetworkAccessProvider) { + t.Fatal("network_access_provider.v1 must be resident") + } + for _, typ := range []string{capability.MetadataProvider, capability.ScheduledTask, capability.WatchSyncProvider, ""} { + if IsResidentCapabilityType(typ) { + t.Fatalf("%q must not be resident", typ) + } + } +} + +func TestResidentRestartOfNonResidentOnlyStops(t *testing.T) { + host := &fakeServiceHost{} + store := newFakeServiceInstallationStore(&Installation{ID: 2, PluginID: "silo.metadb", Version: "1", InstallPath: "/x/plugin", Enabled: true, Kind: KindPlugin}) + service := &Service{installations: store, host: host} + service.resident = newResidentSupervisor(service, ResidentOptions{}) + if err := service.RestartInstallation(context.Background(), 2); err != nil { + t.Fatalf("RestartInstallation: %v", err) + } + if len(host.stopped) != 1 || host.stopped[0] != 2 || len(host.started) != 0 { + t.Fatalf("stopped=%v started=%d, want one stop and no start", host.stopped, len(host.started)) + } + disabled := false + _ = store.Update(context.Background(), 2, UpdateInstallationInput{Enabled: &disabled}) + service.invalidateInstallationCache() + if err := service.RestartInstallation(context.Background(), 2); !errors.Is(err, ErrInstallationDisabled) { + t.Fatalf("RestartInstallation(disabled) = %v, want ErrInstallationDisabled", err) + } +} + +// buildResidentFixtureVersion is buildResidentFixture for a second release +// of the same plugin: the binary reports the given version in its manifest. +func buildResidentFixtureVersion(t *testing.T, version string) string { + t.Helper() + dir := t.TempDir() + bin := filepath.Join(dir, "residentplugin") + build := exec.Command("go", "build", "-ldflags", "-X main.version="+version, "-o", bin, "./testdata/residentplugin") + if out, err := build.CombinedOutput(); err != nil { + t.Fatalf("build residentplugin %s: %v\n%s", version, err, out) + } + raw, err := exec.Command(bin, "manifest").Output() + if err != nil { + t.Fatalf("read fixture manifest: %v", err) + } + if err := os.WriteFile(InstalledManifestPath(bin), raw, 0o644); err != nil { + t.Fatal(err) + } + return bin +} + +// A replaced or auto-updated binary changes the row's version and install +// path. On the host that made the change the process is already gone; on +// any other host running the installation (a proxy node) it is still alive +// and must be replaced by the next reconcile, not kept because it answers +// health checks. +func TestResidentSupervisorReplacesProcessWhenBinaryChanges(t *testing.T) { + f := newResidentFixture(t, ResidentOptions{}) + ctx := context.Background() + f.service.StartResidents(ctx) + waitState(t, f.service, 5, "initial running", running) + first, err := f.host.Client(5) + if err != nil { + t.Fatal(err) + } + if got := first.Manifest().GetVersion(); got != "0.1.0" { + t.Fatalf("initial version = %q", got) + } + + next := buildResidentFixtureVersion(t, "0.2.0") + version := "0.2.0" + if err := f.store.Update(ctx, 5, UpdateInstallationInput{Version: &version, InstallPath: &next}); err != nil { + t.Fatal(err) + } + f.service.OnLifecycleChange(ctx) + + deadline := time.Now().Add(30 * time.Second) + for { + current, err := f.host.Client(5) + state, _ := f.service.Residents().State(5) + if err == nil && current != first && current.Manifest().GetVersion() == "0.2.0" && state.State == ResidentRunning { + if state.RestartCount != 0 || state.LastError != "" { + t.Fatalf("replacement counted as a failure: %+v", state) + } + break + } + if time.Now().After(deadline) { + t.Fatalf("resident kept the old process after the binary changed (err=%v, state=%+v)", err, state) + } + time.Sleep(10 * time.Millisecond) + } + + // The same reconcile again is a no-op: nothing changed. + replaced, _ := f.host.Client(5) + f.service.OnLifecycleChange(ctx) + if again, err := f.host.Client(5); err != nil || again != replaced { + t.Fatalf("an unchanged row replaced the process (err=%v)", err) + } +} + +// fixtureArchive packs a built fixture binary the way the installer stores +// it in plugin_archives, so a service with an archive cache can rehydrate it. +func fixtureArchive(t *testing.T, bin string, installationID int) *InstallationArchive { + t.Helper() + binaryData, err := os.ReadFile(bin) + if err != nil { + t.Fatal(err) + } + manifestBytes, err := os.ReadFile(InstalledManifestPath(bin)) + if err != nil { + t.Fatal(err) + } + manifest, err := LoadManifestBytes(manifestBytes) + if err != nil { + t.Fatal(err) + } + archiveBytes, err := buildBinaryPluginArchive(manifestBytes, binaryData) + if err != nil { + t.Fatal(err) + } + return &InstallationArchive{InstallationID: installationID, ManifestJSON: manifestBytes, Checksum: manifest.GetChecksum(), Bytes: archiveBytes} +} + +// A proxy node never has the API server's install path: it rehydrates each +// release from plugin_archives into its own cache root. When the row moves +// to a new release the old process, still healthy, is replaced by one built +// from the new archive, and the old release leaves the cache. +func TestResidentSupervisorReplacesRehydratedProcessOnNewRelease(t *testing.T) { + f := newResidentFixture(t, ResidentOptions{}) + ctx := context.Background() + root := filepath.Join(t.TempDir(), "proxy-cache") + apiMachine := filepath.Join(string(filepath.Separator), "api-machine-does-not-exist", "plugins", "silo.test.resident") + + first := f.store.byID[5].InstallPath + f.store.archives = map[int]*InstallationArchive{5: fixtureArchive(t, first, 5)} + f.store.byID[5].InstallPath = filepath.Join(apiMachine, "0.1.0", "install-aaaa", "plugin") + f.service.archiveCache = NewArchiveCacheAt(f.store, root) + + f.service.StartResidents(ctx) + waitState(t, f.service, 5, "running from the rehydrated 0.1.0", running) + initial, err := f.host.Client(5) + if err != nil { + t.Fatal(err) + } + firstLocal := filepath.Join(root, "silo.test.resident", "0.1.0", "install-aaaa", "plugin") + if _, err := os.Stat(firstLocal); err != nil { + t.Fatalf("0.1.0 was not rehydrated under the cache root: %v", err) + } + + next := buildResidentFixtureVersion(t, "0.2.0") + f.store.archives[5] = fixtureArchive(t, next, 5) + version := "0.2.0" + nextRecorded := filepath.Join(apiMachine, "0.2.0", "install-bbbb", "plugin") + if err := f.store.Update(ctx, 5, UpdateInstallationInput{Version: &version, InstallPath: &nextRecorded}); err != nil { + t.Fatal(err) + } + f.service.OnLifecycleChange(ctx) + + deadline := time.Now().Add(30 * time.Second) + for { + current, err := f.host.Client(5) + state, _ := f.service.Residents().State(5) + if err == nil && current != initial && current.Manifest().GetVersion() == "0.2.0" && state.State == ResidentRunning { + break + } + if time.Now().After(deadline) { + t.Fatalf("proxy kept the 0.1.0 process after the release changed (err=%v, state=%+v)", err, state) + } + time.Sleep(10 * time.Millisecond) + } + if _, err := os.Stat(filepath.Join(root, "silo.test.resident", "0.2.0", "install-bbbb", "plugin")); err != nil { + t.Fatalf("0.2.0 was not rehydrated under the cache root: %v", err) + } + if _, err := os.Stat(filepath.Dir(firstLocal)); !errors.Is(err, os.ErrNotExist) { + t.Fatalf("0.1.0 release still in the cache after the replacement: %v", err) + } + if _, err := os.Stat(apiMachine); !errors.Is(err, os.ErrNotExist) { + t.Fatalf("proxy touched the API server's install path: %v", err) + } +} + +// Two enabled installations declaring one provider slug: only the lowest id +// is resident. The duplicate is never commanded or reported, so it must not +// run either. +func TestResidentSupervisorDoesNotStartADuplicateProviderSlug(t *testing.T) { + f := newResidentFixture(t, ResidentOptions{}) + ctx := context.Background() + second := buildResidentFixture(t) + dup := &Installation{ID: 6, PluginID: "silo.test.resident-copy", Version: "0.1.0", InstallPath: second, Enabled: true, Kind: KindPlugin} + f.store.byID[dup.ID] = dup + f.store.byPluginID[dup.PluginID] = append(f.store.byPluginID[dup.PluginID], dup) + f.store.listCapabilities = append(f.store.listCapabilities, &Capability{InstallationID: 6, Type: capability.NetworkAccessProvider, ID: "stub"}) + + f.service.StartResidents(ctx) + waitState(t, f.service, 5, "owner running", running) + if _, tracked := f.service.Residents().State(6); tracked { + t.Fatal("duplicate slug installation became resident") + } + if _, err := f.host.Client(6); !errors.Is(err, pluginhost.ErrClientNotFound) { + t.Fatalf("duplicate slug installation was launched: %v", err) + } + if _, err := f.service.MetadataProviderClient(ctx, 6, "metadata"); !errors.Is(err, pluginhost.ErrPluginUnhealthy) { + t.Errorf("duplicate lazy RPC error = %v, want ErrPluginUnhealthy", err) + } + if _, err := f.host.Client(6); !errors.Is(err, pluginhost.ErrClientNotFound) { + t.Fatal("lazy RPC started the excluded duplicate provider") + } +} + +func TestResidentLazyRPCDoesNotLaunchBeforeTheGateOpens(t *testing.T) { + f := newResidentFixture(t, ResidentOptions{}) + for _, armed := range []bool{false, true} { + if armed { + f.service.SetResidentGate(func(context.Context) error { return errors.New("node disabled") }) + f.service.StartResidents(t.Context()) + } + if _, err := f.service.MetadataProviderClient(t.Context(), 5, "metadata"); !errors.Is(err, pluginhost.ErrPluginUnhealthy) { + t.Errorf("lazy RPC with armed=%v error = %v, want ErrPluginUnhealthy", armed, err) + } + if _, err := f.host.Client(5); !errors.Is(err, pluginhost.ErrClientNotFound) { + t.Fatalf("lazy RPC started a provider before the gate opened (armed=%v)", armed) + } + } +} + +// A change of host identity (a proxy whose stream_nodes row was deleted and +// re-registered under a new id) replaces every running resident: the process +// keeps the identity it started with in memory, so it must not go on serving +// under the new one. +func TestResidentSupervisorReplacesProcessWhenHostIdentityChanges(t *testing.T) { + f := newResidentFixture(t, ResidentOptions{}) + ctx := context.Background() + var identity atomic.Value + identity.Store("node:1") + f.service.SetResidentHostIdentity(func() string { return identity.Load().(string) }) + f.service.StartResidents(ctx) + waitState(t, f.service, 5, "running under node:1", running) + first, err := f.host.Client(5) + if err != nil { + t.Fatal(err) + } + + f.service.OnLifecycleChange(ctx) + if same, err := f.host.Client(5); err != nil || same != first { + t.Fatalf("an unchanged identity replaced the process (err=%v)", err) + } + + identity.Store("node:2") + f.service.OnLifecycleChange(ctx) + deadline := time.Now().Add(30 * time.Second) + for { + current, err := f.host.Client(5) + state, _ := f.service.Residents().State(5) + if err == nil && current != first && state.State == ResidentRunning { + if state.RestartCount != 0 || state.LastError != "" { + t.Fatalf("identity change counted as a failure: %+v", state) + } + break + } + if time.Now().After(deadline) { + t.Fatalf("resident kept the old process after the host identity changed (err=%v, state=%+v)", err, state) + } + time.Sleep(10 * time.Millisecond) + } +} + +// Overlapping reconciles must not let an older one, whose desired set was +// computed before a disable, resurrect the resident after a newer one +// removed it. Reconcile is serialized end to end, so the second call sees +// the disabled row. +func TestResidentSupervisorConcurrentReconcilesDoNotResurrectADisabledResident(t *testing.T) { + f := newResidentFixture(t, ResidentOptions{}) + ctx := context.Background() + f.service.StartResidents(ctx) + waitState(t, f.service, 5, "running", running) + + // Race many reconciles against a disable. Whatever the interleaving, the + // final state is "not resident" and no process is left behind. + var wg sync.WaitGroup + for i := 0; i < 8; i++ { + wg.Add(1) + go func() { + defer wg.Done() + f.service.resident.Reconcile(ctx) + }() + } + disabled := false + if err := f.store.Update(ctx, 5, UpdateInstallationInput{Enabled: &disabled}); err != nil { + t.Fatal(err) + } + f.service.OnLifecycleChange(ctx) + wg.Wait() + // One more reconcile after everything settled must be a no-op. + f.service.resident.Reconcile(ctx) + if _, tracked := f.service.Residents().State(5); tracked { + t.Fatal("disabled resident is still tracked after overlapping reconciles") + } + if _, err := f.host.Client(5); !errors.Is(err, pluginhost.ErrClientNotFound) { + t.Fatalf("disabled resident's process survived: %v", err) + } +} + +func TestResidentSupervisorKeepsRunningProviderWhenManifestUnavailable(t *testing.T) { + f := newResidentFixture(t, ResidentOptions{}) + ctx := t.Context() + f.service.StartResidents(ctx) + waitState(t, f.service, 5, "running", running) + first, err := f.host.Client(5) + if err != nil { + t.Fatal(err) + } + if err := os.Remove(InstalledManifestPath(f.store.byID[5].InstallPath)); err != nil { + t.Fatal(err) + } + f.service.OnLifecycleChange(ctx) + if current, err := f.host.Client(5); err != nil || current != first { + t.Fatalf("healthy process replaced after manifest read failure: %v", err) + } + if _, err := f.service.HostNetworkAccessDisconnect(ctx, "stub"); err != nil { + t.Fatalf("provider cannot be disconnected without its disk manifest: %v", err) + } + disabled := false + if err := f.store.Update(ctx, 5, UpdateInstallationInput{Enabled: &disabled}); err != nil { + t.Fatal(err) + } + f.service.OnLifecycleChange(ctx) + if _, err := f.host.Client(5); !errors.Is(err, pluginhost.ErrClientNotFound) { + t.Fatalf("disabled provider still running: %v", err) + } +} + +func TestResidentRestartReturnsWhenCallerCancels(t *testing.T) { + manifest := testPluginManifest(t, "silo.metadb", "0.0.36") + path := writeInstalledPluginManifest(t, manifest) + host := &ctxCaptureHost{entered: make(chan struct{}), proceed: make(chan struct{}), startResult: &fakePluginClient{manifest: manifest}} + store := newFakeServiceInstallationStore(&Installation{ID: 5, PluginID: manifest.PluginId, Version: manifest.Version, InstallPath: path, Enabled: true, Kind: KindPlugin}) + service := &Service{host: host, installations: store} + service.resident = newResidentSupervisor(service, ResidentOptions{}) + service.resident.entries[5] = &residentEntry{id: 5, state: ResidentRunning} + ctx, cancel := context.WithCancel(t.Context()) + defer cancel() + done := make(chan error, 1) + go func() { done <- service.resident.Restart(ctx, 5) }() + t.Cleanup(func() { close(host.proceed); _ = service.StopResidents(context.Background()) }) + select { + case <-host.entered: + case <-time.After(5 * time.Second): + t.Fatal("restart did not reach the host") + } + cancel() + select { + case err := <-done: + if !errors.Is(err, context.Canceled) { + t.Fatalf("restart cancellation=%v", err) + } + case <-time.After(time.Second): + t.Fatal("restart ignored caller cancellation") + } +} diff --git a/internal/plugins/runtime_config.go b/internal/plugins/runtime_config.go index 2e0ba5eb4b..3db23be03e 100644 --- a/internal/plugins/runtime_config.go +++ b/internal/plugins/runtime_config.go @@ -100,7 +100,9 @@ func (s *RuntimeConfigStore) PutGlobalConfig( // CompareAndSwapGlobalConfig persists value only when the row still matches // the version the caller merged. A nil expectedUpdatedAt creates the row only -// when it does not already exist. +// when it does not already exist. An admin save also advances the durable +// runtime generation in the same transaction, so proxy polls recover a +// missed lifecycle event. Plugin-originated writes use PutGlobalConfig. func (s *RuntimeConfigStore) CompareAndSwapGlobalConfig( ctx context.Context, installationID int, @@ -116,15 +118,31 @@ func (s *RuntimeConfigStore) CompareAndSwapGlobalConfig( return false, fmt.Errorf("marshaling plugin runtime config: %w", err) } + tx, err := s.pool.Begin(ctx) + if err != nil { + return false, fmt.Errorf("begin plugin config update: %w", err) + } + defer func() { _ = tx.Rollback(ctx) }() + // Lock the parent installation before touching plugin_runtime_configs. + // InstallationStore.Update takes this lock first; acquiring it here keeps + // the FK key-share lock and generation update in one order and prevents a + // concurrent lifecycle update from deadlocking with a config save. + var parentID int + if err := tx.QueryRow(ctx, `SELECT id FROM plugin_installations WHERE id = $1 FOR NO KEY UPDATE`, installationID).Scan(&parentID); err != nil { + if errors.Is(err, pgx.ErrNoRows) { + return false, nil + } + return false, fmt.Errorf("lock plugin installation for config update: %w", err) + } var tag pgconn.CommandTag if expectedUpdatedAt == nil { - tag, err = s.pool.Exec(ctx, ` + tag, err = tx.Exec(ctx, ` INSERT INTO plugin_runtime_configs (plugin_installation_id, config_key, config_value) VALUES ($1, $2, $3) ON CONFLICT (plugin_installation_id, config_key) DO NOTHING `, installationID, key, valueJSON) } else { - tag, err = s.pool.Exec(ctx, ` + tag, err = tx.Exec(ctx, ` UPDATE plugin_runtime_configs SET config_value = $3, updated_at = NOW() WHERE plugin_installation_id = $1 @@ -135,7 +153,18 @@ func (s *RuntimeConfigStore) CompareAndSwapGlobalConfig( if err != nil { return false, fmt.Errorf("compare-and-swap plugin runtime config: %w", err) } - return tag.RowsAffected() > 0, nil + if tag.RowsAffected() == 0 { + return false, nil + } + if _, err := tx.Exec(ctx, `UPDATE plugin_installations + SET runtime_generation = runtime_generation + 1, updated_at = NOW() + WHERE id = $1`, installationID); err != nil { + return false, fmt.Errorf("advance plugin runtime generation: %w", err) + } + if err := tx.Commit(ctx); err != nil { + return false, fmt.Errorf("commit plugin config update: %w", err) + } + return true, nil } func (s *RuntimeConfigStore) ListGlobalConfigs(ctx context.Context, installationID int) ([]*RuntimeConfig, error) { diff --git a/internal/plugins/service.go b/internal/plugins/service.go index d42bb94042..2324c8f0b0 100644 --- a/internal/plugins/service.go +++ b/internal/plugins/service.go @@ -22,6 +22,7 @@ import ( "google.golang.org/protobuf/types/known/structpb" pluginv1 "github.com/Silo-Server/silo-plugin-sdk/pkg/pluginproto/silo/plugin/v1" + "github.com/Silo-Server/silo-server/internal/cache" "github.com/Silo-Server/silo-server/internal/pluginhost" ) @@ -38,6 +39,7 @@ type pluginClient interface { AuthProvider(capabilityID string) (*pluginhost.AuthProviderClient, error) HTTPRoutes(capabilityID string) (*pluginhost.HTTPRoutesClient, error) WatchSyncProvider(capabilityID string) (*pluginhost.WatchSyncProviderClient, error) + NetworkAccessProvider(capabilityID string) (*pluginhost.NetworkAccessProviderClient, error) } type Host interface { @@ -45,6 +47,10 @@ type Host interface { Client(installationID int) (pluginClient, error) Stop(installationID int) error Shutdown(ctx context.Context) error + // NextStartSeq returns the host's monotonic start counter. A client whose + // StartSeq is at or below the value read before a launch was issued by + // an earlier launch that the singleflight joined. + NextStartSeq() uint64 } type serviceInstallationStore interface { @@ -52,6 +58,7 @@ type serviceInstallationStore interface { GetByID(ctx context.Context, id int) (*Installation, error) List(ctx context.Context) ([]*Installation, error) ListEnabled(ctx context.Context) ([]*Installation, error) + ListEnabledWithCapabilityTypes(ctx context.Context, capabilityTypes []string) ([]*Installation, error) ListByPluginID(ctx context.Context, pluginID string) ([]*Installation, error) Update(ctx context.Context, id int, input UpdateInstallationInput) error ListCapabilities(ctx context.Context, installationID int) ([]*Capability, error) @@ -82,6 +89,16 @@ type Service struct { lifecycleMu sync.RWMutex lifecycleHooks []func(context.Context) launchGroup singleflight.Group + resident *ResidentSupervisor + // lifecycleBus, when set by PublishLifecycleChanges, carries every + // lifecycle change to the proxy nodes running the same installations. + lifecycleBus cache.EventBus + + // networkAccessHostInfo and networkAccessStatus back the network access + // admin reads; see network_access.go. + networkAccessHostInfo pluginhost.HostInfoFunc + networkAccessStatus NetworkAccessStatusSink + networkAccessNodes NetworkAccessNodes // installationCache memoizes plugin_installations rows keyed by ID so the // hot plugin-RPC path (ensureClient -> loadInstallation) and the metadata @@ -163,6 +180,11 @@ func NewService( // enable / disable / update / uninstall) wipes the cache, keeping the // memoized rows correct without any external wiring. svc.AddLifecycleHook(func(context.Context) { svc.invalidateInstallationCache() }) + // Resident plugins (network access providers) are reconciled after every + // lifecycle change; the supervisor stays inert until StartResidents arms + // it once the API listener is bound. + svc.resident = newResidentSupervisor(svc, ResidentOptions{}) + svc.AddLifecycleHook(func(ctx context.Context) { svc.resident.Reconcile(ctx) }) return svc } @@ -243,10 +265,9 @@ func (s *Service) InstallLocal(ctx context.Context, req InstallArchiveRequest) ( } var result *InstallResult if existing != nil { - if err := s.stopInstallationIfRunning(existing); err != nil { - return nil, err - } - result, err = s.installer.ReplaceLocal(ctx, existing, req) + result, err = s.replaceStopped(ctx, existing, func() (*InstallResult, error) { + return s.installer.ReplaceLocal(ctx, existing, req) + }) } else { result, err = s.installer.InstallLocal(ctx, req) } @@ -286,10 +307,9 @@ func (s *Service) InstallCatalog(ctx context.Context, req InstallCatalogRequest) if existing == nil { result, err = s.installer.InstallRemote(ctx, archiveReq) } else { - if err = s.stopInstallationIfRunning(existing); err != nil { - return nil, err - } - result, err = s.installer.ReplaceRemote(ctx, existing, archiveReq) + result, err = s.replaceStopped(ctx, existing, func() (*InstallResult, error) { + return s.installer.ReplaceRemote(ctx, existing, archiveReq) + }) } } else { binaryReq := InstallBinaryRequest{ @@ -300,10 +320,9 @@ func (s *Service) InstallCatalog(ctx context.Context, req InstallCatalogRequest) if existing == nil { result, err = s.installer.InstallBinary(ctx, binaryReq) } else { - if err = s.stopInstallationIfRunning(existing); err != nil { - return nil, err - } - result, err = s.installer.ReplaceBinary(ctx, existing, binaryReq) + result, err = s.replaceStopped(ctx, existing, func() (*InstallResult, error) { + return s.installer.ReplaceBinary(ctx, existing, binaryReq) + }) } } if err != nil { @@ -361,10 +380,9 @@ func (s *Service) InstallBinary(ctx context.Context, req InstallBinaryRequest) ( return nil, existErr } if existing != nil { - if err = s.stopInstallationIfRunning(existing); err != nil { - return nil, err - } - result, err = s.installer.ReplaceBinary(ctx, existing, req) + result, err = s.replaceStopped(ctx, existing, func() (*InstallResult, error) { + return s.installer.ReplaceBinary(ctx, existing, req) + }) if err != nil { return nil, err } @@ -419,10 +437,9 @@ func (s *Service) InstallBinaryUpload(ctx context.Context, binaryData []byte) (* } oldInstallation := existing[0] - if err := s.stopInstallationIfRunning(oldInstallation); err != nil { - return nil, err - } - result, err = s.installer.replaceBinary(ctx, oldInstallation, binaryData, actualChecksum, manifest) + result, err = s.replaceStopped(ctx, oldInstallation, func() (*InstallResult, error) { + return s.installer.replaceBinary(ctx, oldInstallation, binaryData, actualChecksum, manifest) + }) if err != nil { return nil, err } @@ -458,6 +475,23 @@ func (s *Service) stopInstallationIfRunning(existing *Installation) error { return nil } +// replaceStopped stops the existing installation's process and runs replace. +// When replace fails the row still names the old release but its process is +// gone: a lazily started plugin comes back on its next RPC, while a resident +// only restarts on a lifecycle reconcile, so the hooks run before the error +// is returned. On success the caller runs them after the row has changed. +func (s *Service) replaceStopped(ctx context.Context, existing *Installation, replace func() (*InstallResult, error)) (*InstallResult, error) { + if err := s.stopInstallationIfRunning(existing); err != nil { + return nil, err + } + result, err := replace() + if err != nil { + s.OnLifecycleChange(ctx) + return nil, err + } + return result, nil +} + func (s *Service) PreloadEnabled(ctx context.Context) error { if s.installations == nil { return nil @@ -495,22 +529,42 @@ func (s *Service) PreloadEnabled(ctx context.Context) error { } func (s *Service) Start(ctx context.Context, installationID int) (pluginClient, error) { + return s.start(ctx, installationID, true) +} + +func (s *Service) start(ctx context.Context, installationID int, allowResident bool) (pluginClient, error) { installation, manifest, err := s.ensureInstallationCache(ctx, installationID, true) if err != nil { return nil, err } + if !allowResident && isResidentManifest(manifest) { + return nil, fmt.Errorf("%w: resident plugin installation %d requires supervision", pluginhost.ErrPluginUnhealthy, installationID) + } configEntries, err := s.globalConfigEntries(ctx, installation.ID) if err != nil { return nil, err } return s.host.Start(ctx, pluginhost.StartRequest{ InstallationID: installation.ID, - BinaryPath: installation.InstallPath, + BinaryPath: s.localInstallPath(installation), Manifest: manifest, Config: configEntries, }) } +// localInstallPath is where this host keeps the installation's binary: the +// recorded install path on the API server, the archive cache's own copy on a +// host with its own cache root (a proxy node). +func (s *Service) localInstallPath(installation *Installation) string { + if installation == nil { + return "" + } + if s == nil || s.archiveCache == nil { + return installation.InstallPath + } + return s.archiveCache.LocalInstallPath(installation) +} + func (s *Service) Stop(installationID int) error { if s.host == nil { return nil @@ -712,7 +766,7 @@ func (s *Service) ResolveAssetPath(ctx context.Context, installationID int, asse } for _, asset := range manifest.GetAssets() { if asset.GetPath() == assetPath { - resolved := filepath.Join(filepath.Dir(installation.InstallPath), assetPath) + resolved := filepath.Join(filepath.Dir(s.localInstallPath(installation)), assetPath) if _, err := os.Stat(resolved); err != nil { return "", fmt.Errorf("plugin asset %q: %w", assetPath, err) } @@ -737,17 +791,32 @@ func (s *Service) ManifestForInstallation( return s.manifestForInstallation(ctx, installationID, false) } -// ensureClient returns a running client for the installation, collapsing -// concurrent first-use of a cold installation into a single launch so a burst of -// callers does not spawn redundant plugin processes (Host.Start releases its lock -// during the slow launch and cannot dedupe). After the flight completes the key -// is freed, so subsequent callers re-run and hit the now-warm cache. +// ensureClient returns a running client for an RPC. Tracked residents may +// only use the process the supervisor owns; RPCs cannot launch or replace it. func (s *Service) ensureClient(ctx context.Context, installationID int) (pluginClient, error) { - v, err, _ := s.launchGroup.Do(strconv.Itoa(installationID), func() (any, error) { + if state, tracked := s.resident.State(installationID); tracked { + if _, err := s.loadInstallation(ctx, installationID, true); err != nil { + return nil, err + } + if state.State != ResidentRunning { + return nil, fmt.Errorf("%w: resident plugin installation %d is %s", pluginhost.ErrPluginUnhealthy, installationID, state.State) + } + return s.host.Client(installationID) + } + return s.ensureClientForStart(ctx, installationID, false) +} + +// ensureClientForStart collapses concurrent launches for a cold installation. +// Only accepted supervisor starts may launch resident-capability plugins. +// Separate flights keep a lazy RPC from joining an accepted resident launch, +// or making that launch fail because the RPC is forbidden from starting it. +func (s *Service) ensureClientForStart(ctx context.Context, installationID int, allowResident bool) (pluginClient, error) { + key := strconv.Itoa(installationID) + ":" + strconv.FormatBool(allowResident) + v, err, _ := s.launchGroup.Do(key, func() (any, error) { // Isolate the shared launch from the leader caller's cancellation: other // waiters depend on this in-flight launch, so a single caller's canceled // request must not tear it down. Values (tracing, auth) are preserved. - return s.doEnsureClient(context.WithoutCancel(ctx), installationID) + return s.doEnsureClient(context.WithoutCancel(ctx), installationID, allowResident) }) if err != nil { return nil, err @@ -755,15 +824,34 @@ func (s *Service) ensureClient(ctx context.Context, installationID int) (pluginC return v.(pluginClient), nil } -func (s *Service) doEnsureClient(ctx context.Context, installationID int) (pluginClient, error) { +func (s *Service) doEnsureClient(ctx context.Context, installationID int, allowResident bool) (pluginClient, error) { installation, err := s.loadInstallation(ctx, installationID, true) if err != nil { return nil, err } client, err := s.host.Client(installationID) if err == nil { - installedManifest, manifestErr := LoadManifestFile(InstalledManifestPath(installation.InstallPath)) + cachedManifest := client.Manifest() + if !allowResident && isResidentManifest(cachedManifest) { + return nil, fmt.Errorf("%w: resident plugin installation %d requires supervision", pluginhost.ErrPluginUnhealthy, installationID) + } + installedManifest, manifestErr := LoadManifestFile(InstalledManifestPath(s.localInstallPath(installation))) if manifestErr != nil { + // The files are not here (yet): on a proxy node the row may name + // a release this host has not rehydrated. The running process is + // still trustworthy only if it is the row's version. + if installation.Version != "" && manifestVersion(cachedManifest) != installation.Version { + slog.WarnContext(ctx, "plugin client runs a different version than the installation; restarting", "component", "plugins", + "installation_id", installation.ID, + "plugin_id", installation.PluginID, + "cached_version", manifestVersion(cachedManifest), + "installed_version", installation.Version, + ) + if stopErr := s.host.Stop(installationID); stopErr != nil && !errors.Is(stopErr, pluginhost.ErrClientNotFound) { + return nil, fmt.Errorf("stop stale plugin installation %d: %w", installationID, stopErr) + } + return s.start(ctx, installationID, allowResident) + } slog.WarnContext(ctx, "plugin installed manifest unavailable; reusing healthy client", "component", "plugins", "installation_id", installation.ID, "plugin_id", installation.PluginID, @@ -772,7 +860,6 @@ func (s *Service) doEnsureClient(ctx context.Context, installationID int) (plugi ) return client, nil } - cachedManifest := client.Manifest() if cachedManifest != nil && proto.Equal(cachedManifest, installedManifest) { return client, nil } @@ -787,16 +874,16 @@ func (s *Service) doEnsureClient(ctx context.Context, installationID int) (plugi if stopErr := s.host.Stop(installationID); stopErr != nil && !errors.Is(stopErr, pluginhost.ErrClientNotFound) { return nil, fmt.Errorf("stop stale plugin installation %d: %w", installationID, stopErr) } - return s.Start(ctx, installationID) + return s.start(ctx, installationID, allowResident) } if errors.Is(err, pluginhost.ErrPluginUnhealthy) { if stopErr := s.host.Stop(installationID); stopErr != nil && !errors.Is(stopErr, pluginhost.ErrClientNotFound) { return nil, fmt.Errorf("stop unhealthy plugin installation %d: %w", installationID, stopErr) } - return s.Start(ctx, installationID) + return s.start(ctx, installationID, allowResident) } if errors.Is(err, pluginhost.ErrClientNotFound) { - return s.Start(ctx, installationID) + return s.start(ctx, installationID, allowResident) } return nil, err } diff --git a/internal/plugins/service_admin_config.go b/internal/plugins/service_admin_config.go index 4db7f2ee59..cc09b2aa54 100644 --- a/internal/plugins/service_admin_config.go +++ b/internal/plugins/service_admin_config.go @@ -115,11 +115,16 @@ func (s *Service) SetGlobalConfigWithClears( stopErr = fmt.Errorf("reload plugin after config update: %w", err) } } + // A reconfigured resident gets a fresh failure budget; the lifecycle + // reconcile below starts it again. + s.resident.Reset(installationID) // Configuration is part of metadata match input identity. Notify hooks // after the old process has been stopped so resolver reloads observe the new // runtime. The durable config changed even when stopping failed, so hooks // still run before returning that error and parked rows are not left asleep. s.OnLifecycleChange(ctx) + // The config save advanced runtime_generation in the same transaction; + // both the lifecycle event above and a proxy's poll replace stale processes. return stopErr } diff --git a/internal/plugins/service_connection.go b/internal/plugins/service_connection.go index 600996e3fa..36fe438320 100644 --- a/internal/plugins/service_connection.go +++ b/internal/plugins/service_connection.go @@ -155,7 +155,7 @@ func (s *Service) TestGlobalConfigWithClears( testInstallationID := -int(s.testConfigSeq.Add(1)) client, err := s.host.Start(ctx, pluginhost.StartRequest{ InstallationID: testInstallationID, - BinaryPath: installation.InstallPath, + BinaryPath: s.localInstallPath(installation), Manifest: manifest, Config: configEntries, }) diff --git a/internal/plugins/service_hot_reload_test.go b/internal/plugins/service_hot_reload_test.go index 3dc9223775..8ffe056e06 100644 --- a/internal/plugins/service_hot_reload_test.go +++ b/internal/plugins/service_hot_reload_test.go @@ -2,11 +2,14 @@ package plugins import ( "archive/zip" + "cmp" "context" "crypto/sha256" "encoding/hex" "os" "path/filepath" + "slices" + "sync" "testing" "google.golang.org/protobuf/encoding/protojson" @@ -183,6 +186,8 @@ func (f *fakeServiceHost) Stop(installationID int) error { return nil } +func (f *fakeServiceHost) NextStartSeq() uint64 { return 0 } + func (f *fakeServiceHost) Shutdown(context.Context) error { return nil } @@ -241,8 +246,15 @@ func (f *fakePluginClient) WatchSyncProvider(string) (*pluginhost.WatchSyncProvi return nil, nil } +func (f *fakePluginClient) NetworkAccessProvider(string) (*pluginhost.NetworkAccessProviderClient, error) { + return nil, nil +} + type fakeServiceInstallationStore struct { - byID map[int]*Installation + byID map[int]*Installation + // mu guards every field: the resident supervisor reads the store from + // several goroutines while a test mutates it. + mu sync.Mutex byPluginID map[string][]*Installation createInputs []CreateInstallationInput updateIDs []int @@ -252,6 +264,9 @@ type fakeServiceInstallationStore struct { saveArchiveErr error listCapabilities []*Capability events *[]string + // archives, when set, backs GetArchive for the archive cache; unset ids + // report ErrArchiveNotFound. + archives map[int]*InstallationArchive } func newFakeServiceInstallationStore(installations ...*Installation) *fakeServiceInstallationStore { @@ -271,6 +286,8 @@ func newFakeServiceInstallationStore(installations ...*Installation) *fakeServic } func (s *fakeServiceInstallationStore) Create(_ context.Context, input CreateInstallationInput) (*Installation, error) { + s.mu.Lock() + defer s.mu.Unlock() recordTestEvent(s.events, "create") s.createInputs = append(s.createInputs, input) id := len(s.byID) + 1 @@ -287,12 +304,16 @@ func (s *fakeServiceInstallationStore) Create(_ context.Context, input CreateIns } func (s *fakeServiceInstallationStore) SaveArchive(_ context.Context, installationID int, _ []byte, _ string, _ []byte) error { + s.mu.Lock() + defer s.mu.Unlock() recordTestEvent(s.events, "save_archive") s.saveArchiveIDs = append(s.saveArchiveIDs, installationID) return s.saveArchiveErr } func (s *fakeServiceInstallationStore) Update(_ context.Context, id int, input UpdateInstallationInput) error { + s.mu.Lock() + defer s.mu.Unlock() recordTestEvent(s.events, "update") s.updateIDs = append(s.updateIDs, id) s.updateInputs = append(s.updateInputs, input) @@ -310,10 +331,15 @@ func (s *fakeServiceInstallationStore) Update(_ context.Context, id int, input U if input.Enabled != nil { installation.Enabled = *input.Enabled } + if input.Restart { + installation.RuntimeGeneration++ + } return nil } func (s *fakeServiceInstallationStore) Delete(_ context.Context, id int) error { + s.mu.Lock() + defer s.mu.Unlock() recordTestEvent(s.events, "delete") s.deleteIDs = append(s.deleteIDs, id) delete(s.byID, id) @@ -321,6 +347,8 @@ func (s *fakeServiceInstallationStore) Delete(_ context.Context, id int) error { } func (s *fakeServiceInstallationStore) GetByID(_ context.Context, id int) (*Installation, error) { + s.mu.Lock() + defer s.mu.Unlock() installation, ok := s.byID[id] if !ok { return nil, ErrInstallationNotFound @@ -330,6 +358,8 @@ func (s *fakeServiceInstallationStore) GetByID(_ context.Context, id int) (*Inst } func (s *fakeServiceInstallationStore) List(context.Context) ([]*Installation, error) { + s.mu.Lock() + defer s.mu.Unlock() result := make([]*Installation, 0, len(s.byID)) for _, installation := range s.byID { cloned := *installation @@ -339,6 +369,8 @@ func (s *fakeServiceInstallationStore) List(context.Context) ([]*Installation, e } func (s *fakeServiceInstallationStore) ListEnabled(_ context.Context) ([]*Installation, error) { + s.mu.Lock() + defer s.mu.Unlock() var result []*Installation for _, installation := range s.byID { if !installation.Enabled { @@ -347,10 +379,13 @@ func (s *fakeServiceInstallationStore) ListEnabled(_ context.Context) ([]*Instal cloned := *installation result = append(result, &cloned) } + slices.SortFunc(result, func(a, b *Installation) int { return cmp.Compare(a.ID, b.ID) }) return result, nil } func (s *fakeServiceInstallationStore) ListByPluginID(_ context.Context, pluginID string) ([]*Installation, error) { + s.mu.Lock() + defer s.mu.Unlock() list := s.byPluginID[pluginID] result := make([]*Installation, 0, len(list)) for _, installation := range list { @@ -361,10 +396,37 @@ func (s *fakeServiceInstallationStore) ListByPluginID(_ context.Context, pluginI } func (s *fakeServiceInstallationStore) ListCapabilities(context.Context, int) ([]*Capability, error) { + s.mu.Lock() + defer s.mu.Unlock() return s.listCapabilities, nil } -func (s *fakeServiceInstallationStore) GetArchive(context.Context, int) (*InstallationArchive, error) { +// ListEnabledWithCapabilityTypes mirrors ListCapabilities, whose fixed +// capability list applies to every installation in the fake. +func (s *fakeServiceInstallationStore) ListEnabledWithCapabilityTypes(ctx context.Context, capabilityTypes []string) ([]*Installation, error) { + matches := false + for _, record := range s.listCapabilities { + if record == nil { + continue + } + for _, capabilityType := range capabilityTypes { + if record.Type == capabilityType { + matches = true + } + } + } + if !matches { + return nil, nil + } + return s.ListEnabled(ctx) +} + +func (s *fakeServiceInstallationStore) GetArchive(_ context.Context, installationID int) (*InstallationArchive, error) { + s.mu.Lock() + defer s.mu.Unlock() + if archive, ok := s.archives[installationID]; ok && archive != nil { + return archive, nil + } return nil, ErrArchiveNotFound } diff --git a/internal/plugins/service_singleflight_test.go b/internal/plugins/service_singleflight_test.go index 0f3e4a19ac..19998c4674 100644 --- a/internal/plugins/service_singleflight_test.go +++ b/internal/plugins/service_singleflight_test.go @@ -55,6 +55,7 @@ func (h *countingHerdHost) Stop(id int) error { } func (h *countingHerdHost) Shutdown(context.Context) error { return nil } +func (h *countingHerdHost) NextStartSeq() uint64 { return 0 } func (h *countingHerdHost) startCount() int { h.mu.Lock() @@ -183,6 +184,7 @@ func (h *ctxCaptureHost) Stop(id int) error { } func (h *ctxCaptureHost) Shutdown(context.Context) error { return nil } +func (h *ctxCaptureHost) NextStartSeq() uint64 { return 0 } func (h *ctxCaptureHost) canceledDuringLaunch() bool { h.mu.Lock() diff --git a/internal/plugins/testdata/residentplugin/main.go b/internal/plugins/testdata/residentplugin/main.go new file mode 100644 index 0000000000..9e4302518f --- /dev/null +++ b/internal/plugins/testdata/residentplugin/main.go @@ -0,0 +1,97 @@ +// Command residentplugin is a test fixture for the resident plugin +// supervisor and the network access admin API: it declares +// network_access_provider.v1 (so the host treats it as resident), answers +// Connect/Disconnect/GetStatus from an in-memory state, and exits with status +// 3 as soon as the file named by SILO_TEST_PLUGIN_EXIT_FILE exists, so a test +// can make it crash on demand. +package main + +import ( + "context" + _ "embed" + "os" + "sync" + "time" + + pluginv1 "github.com/Silo-Server/silo-plugin-sdk/pkg/pluginproto/silo/plugin/v1" + sdkruntime "github.com/Silo-Server/silo-plugin-sdk/pkg/pluginsdk/runtime" +) + +//go:embed manifest.json +var manifestJSON []byte + +// version is the manifest version the fixture reports. Tests that need a +// second release of the same plugin build it with +// -ldflags "-X main.version=0.2.0". +var version = "0.1.0" + +// provider is a stub overlay: connecting "succeeds" at once with a fixed +// origin, so tests can observe the state each RPC reports. +type provider struct { + pluginv1.UnimplementedNetworkAccessProviderServer + + mu sync.Mutex + connected bool +} + +func (p *provider) status() *pluginv1.NetworkAccessStatus { + p.mu.Lock() + defer p.mu.Unlock() + status := &pluginv1.NetworkAccessStatus{State: "disconnected", ProviderVersion: "stub 0.1.0"} + if p.connected { + status.State = "connected" + status.Hostname = "silo.stub.test" + status.Origin = "https://silo.stub.test" + status.Addresses = []string{"100.64.0.7"} + status.Listeners = []*pluginv1.NetworkAccessListener{{Name: "api", Origin: "https://silo.stub.test"}} + status.DesiredConnected = true + } + return status +} + +func (p *provider) Connect(context.Context, *pluginv1.NetworkAccessConnectRequest) (*pluginv1.NetworkAccessStatus, error) { + p.mu.Lock() + p.connected = true + p.mu.Unlock() + return p.status(), nil +} + +func (p *provider) Disconnect(context.Context, *pluginv1.NetworkAccessDisconnectRequest) (*pluginv1.NetworkAccessStatus, error) { + p.mu.Lock() + p.connected = false + p.mu.Unlock() + return p.status(), nil +} + +func (p *provider) GetStatus(ctx context.Context, _ *pluginv1.NetworkAccessGetStatusRequest) (*pluginv1.NetworkAccessStatus, error) { + status := p.status() + // Exercise the host broker on demand: the status carries whether the + // callback path works after the broker connection is established. + if host := sdkruntime.Host(); host != nil { + if _, err := host.GetHostInfo(ctx); err != nil { + status.Error = "host callback: " + err.Error() + } else { + status.Error = "" + status.ProviderVersion += " host-ok" + } + } else { + status.Error = "host callback: broker unavailable" + } + return status, nil +} + +func main() { + if exitFile := os.Getenv("SILO_TEST_PLUGIN_EXIT_FILE"); exitFile != "" { + go func() { + for { + if _, err := os.Stat(exitFile); err == nil { + os.Exit(3) + } + time.Sleep(25 * time.Millisecond) + } + }() + } + sdkruntime.ServeManifest(manifestJSON, version, sdkruntime.CapabilityServers{ + NetworkAccessProvider: &provider{}, + }) +} diff --git a/internal/plugins/testdata/residentplugin/manifest.json b/internal/plugins/testdata/residentplugin/manifest.json new file mode 100644 index 0000000000..8e088d1667 --- /dev/null +++ b/internal/plugins/testdata/residentplugin/manifest.json @@ -0,0 +1,24 @@ +{ + "plugin_id": "silo.test.resident", + "version": "0.1.0", + "checksum": "__CHECKSUM__", + "silo_api_version": "v1", + "supported_platforms": [ + {"os": "linux", "arch": "amd64"}, + {"os": "linux", "arch": "arm64"}, + {"os": "darwin", "arch": "amd64"}, + {"os": "darwin", "arch": "arm64"} + ], + "capabilities": [ + { + "type": "network_access_provider.v1", + "id": "stub", + "display_name": "Resident Fixture", + "description": "Test fixture declaring a resident capability.", + "network_access_provider": { + "provider": "stub", + "display_name": "Resident Fixture" + } + } + ] +} diff --git a/internal/proxy/network_access.go b/internal/proxy/network_access.go new file mode 100644 index 0000000000..b1108295ab --- /dev/null +++ b/internal/proxy/network_access.go @@ -0,0 +1,108 @@ +package proxy + +import ( + "context" + "encoding/json" + "errors" + "log/slog" + "net/http" + + "github.com/go-chi/chi/v5" + + "github.com/Silo-Server/silo-server/internal/netaccess" +) + +// NetworkAccessProviderHost drives the network access provider plugin +// instances running beside this proxy. *plugins.Service implements it; nil +// (a proxy without a plugin host) answers 503 on the routes below. +type NetworkAccessProviderHost interface { + HostNetworkAccessStatus(ctx context.Context) (netaccess.HostStatusReport, error) + HostNetworkAccessProviderStatus(ctx context.Context, provider string) (netaccess.Status, error) + HostNetworkAccessConnect(ctx context.Context, provider string) (netaccess.Status, error) + HostNetworkAccessDisconnect(ctx context.Context, provider string) (netaccess.Status, error) +} + +// SetNetworkAccessProviderHost wires the plugin service the bearer +// network-access routes act on. Call it during construction. +func (s *Server) SetNetworkAccessProviderHost(host NetworkAccessProviderHost) { + s.networkAccessHost = host +} + +// handleNetworkAccessStatus answers the API's fan-out with the live status of +// every provider instance on this node. Unlike /health it takes the node +// bearer, so the report carries the auth URL and error text. +func (s *Server) handleNetworkAccessStatus(w http.ResponseWriter, r *http.Request) { + if s.networkAccessHost == nil { + http.Error(w, "network access providers are not hosted on this node", http.StatusServiceUnavailable) + return + } + report, err := s.networkAccessHost.HostNetworkAccessStatus(r.Context()) + if err != nil { + http.Error(w, "network access status: "+err.Error(), http.StatusInternalServerError) + return + } + writeNetworkAccessJSON(w, report) +} + +// handleNetworkAccessProviderStatus reads the named provider independently, +// so a slow provider cannot make another provider appear unreachable. +func (s *Server) handleNetworkAccessProviderStatus(w http.ResponseWriter, r *http.Request) { + if s.networkAccessHost == nil { + http.Error(w, "network access providers are not hosted on this node", http.StatusServiceUnavailable) + return + } + status, err := s.networkAccessHost.HostNetworkAccessProviderStatus(r.Context(), chi.URLParam(r, "provider")) + if err != nil { + if errors.Is(err, netaccess.ErrProviderNotFound) { + http.Error(w, "network access provider not found", http.StatusNotFound) + return + } + http.Error(w, "network access status: "+err.Error(), http.StatusInternalServerError) + return + } + writeNetworkAccessJSON(w, status) +} + +// handleNetworkAccessConnect brings the named provider up on this node. +func (s *Server) handleNetworkAccessConnect(w http.ResponseWriter, r *http.Request) { + s.applyNetworkAccessCommand(w, r, "connect", func(ctx context.Context, provider string) (netaccess.Status, error) { + return s.networkAccessHost.HostNetworkAccessConnect(ctx, provider) + }) +} + +// handleNetworkAccessDisconnect tears the named provider down on this node. +func (s *Server) handleNetworkAccessDisconnect(w http.ResponseWriter, r *http.Request) { + s.applyNetworkAccessCommand(w, r, "disconnect", func(ctx context.Context, provider string) (netaccess.Status, error) { + return s.networkAccessHost.HostNetworkAccessDisconnect(ctx, provider) + }) +} + +func (s *Server) applyNetworkAccessCommand(w http.ResponseWriter, r *http.Request, verb string, apply func(context.Context, string) (netaccess.Status, error)) { + if s.networkAccessHost == nil { + http.Error(w, "network access providers are not hosted on this node", http.StatusServiceUnavailable) + return + } + provider := chi.URLParam(r, "provider") + status, err := apply(r.Context(), provider) + if err != nil { + if errors.Is(err, netaccess.ErrProviderNotFound) { + http.Error(w, "network access provider not found", http.StatusNotFound) + return + } + http.Error(w, "network access "+verb+": "+err.Error(), http.StatusInternalServerError) + return + } + // State transitions only; the status may carry an auth URL, which is + // never logged. + slog.InfoContext(r.Context(), "network access command applied", slog.String("component", "proxy"), + slog.String("command", verb), slog.String("provider", provider), slog.String("state", status.State)) + writeNetworkAccessJSON(w, status) +} + +func writeNetworkAccessJSON(w http.ResponseWriter, body any) { + w.Header().Set("Content-Type", "application/json") + w.Header().Set("Cache-Control", "no-store") + if err := json.NewEncoder(w).Encode(body); err != nil { + slog.Warn("encode network access response", slog.String("component", "proxy"), slog.Any("error", err)) + } +} diff --git a/internal/proxy/network_access_health_test.go b/internal/proxy/network_access_health_test.go new file mode 100644 index 0000000000..fcf48a9f11 --- /dev/null +++ b/internal/proxy/network_access_health_test.go @@ -0,0 +1,39 @@ +package proxy + +import ( + "testing" + + "github.com/Silo-Server/silo-server/internal/netaccess" +) + +// /health carries the provider report the API stores on this proxy's row. +// Without a status source the field is absent — the same body a build that +// predates the field serves — and with one it mirrors the cache. +func TestHealthReportsNetworkAccessFromTheStatusSource(t *testing.T) { + server := newMetricsProxyServer(t) + server.metrics = newFakeSampler() + + if got := decodeProxyHealth(t, server).NetworkAccess; got != nil { + t.Fatalf("health without a status source reported %#v", got) + } + + cache := netaccess.NewStatusCache() + server.SetNetworkAccessStatus(cache) + if got := decodeProxyHealth(t, server).NetworkAccess; got != nil { + t.Fatalf("health with an empty cache reported %#v", got) + } + + cache.Report(netaccess.Status{ + InstallationID: 1, Provider: "tailscale", State: netaccess.StateConnected, + Origin: "https://proxy-1.tail1234.ts.net", Hostname: "proxy-1.tail1234.ts.net", + AuthURL: "https://login.tailscale.com/a/secret", + }) + got := decodeProxyHealth(t, server).NetworkAccess + entry, ok := got["tailscale"] + if !ok || entry.State != netaccess.StateConnected || entry.Origin != "https://proxy-1.tail1234.ts.net" || entry.Hostname != "proxy-1.tail1234.ts.net" || entry.UpdatedAt.IsZero() { + t.Fatalf("health network_access = %#v", got) + } + if origin, ok := got.ConnectedOrigin("tailscale"); !ok || origin != "https://proxy-1.tail1234.ts.net" { + t.Fatalf("ConnectedOrigin over the wire form = %q, %v", origin, ok) + } +} diff --git a/internal/proxy/network_access_test.go b/internal/proxy/network_access_test.go new file mode 100644 index 0000000000..3b7c99863b --- /dev/null +++ b/internal/proxy/network_access_test.go @@ -0,0 +1,275 @@ +package proxy + +import ( + "context" + "encoding/json" + "net/http" + "net/http/httptest" + "testing" + "time" + + "github.com/Silo-Server/silo-server/internal/netaccess" +) + +// fakeProviderHost stands in for the plugin service: one provider, "stub", +// whose connect and disconnect flip an in-memory state, and a "down" provider +// whose process is not running. +type fakeProviderHost struct { + connected bool + calls []string +} + +func (f *fakeProviderHost) status() netaccess.Status { + status := netaccess.Status{InstallationID: 5, Provider: "stub", State: netaccess.StateDisconnected, ProviderVersion: "stub 0.1.0"} + if f.connected { + status.State = netaccess.StateConnected + status.Origin = "https://proxy-1.stub.test" + status.Hostname = "proxy-1.stub.test" + status.AuthURL = "" + } + return status +} + +func (f *fakeProviderHost) HostNetworkAccessStatus(context.Context) (netaccess.HostStatusReport, error) { + f.calls = append(f.calls, "status") + return netaccess.HostStatusReport{Providers: []netaccess.Status{ + f.status(), + {InstallationID: 9, Provider: "down", State: netaccess.StateUnavailable, Error: "plugin process is backoff: plugin process exited"}, + }}, nil +} + +func (f *fakeProviderHost) HostNetworkAccessProviderStatus(_ context.Context, provider string) (netaccess.Status, error) { + f.calls = append(f.calls, "status:"+provider) + if provider != "stub" { + return netaccess.Status{}, netaccess.ErrProviderNotFound + } + return f.status(), nil +} + +func (f *fakeProviderHost) HostNetworkAccessConnect(_ context.Context, provider string) (netaccess.Status, error) { + f.calls = append(f.calls, "connect:"+provider) + if provider != "stub" { + return netaccess.Status{}, netaccess.ErrProviderNotFound + } + f.connected = true + return f.status(), nil +} + +func (f *fakeProviderHost) HostNetworkAccessDisconnect(_ context.Context, provider string) (netaccess.Status, error) { + f.calls = append(f.calls, "disconnect:"+provider) + if provider != "stub" { + return netaccess.Status{}, netaccess.ErrProviderNotFound + } + f.connected = false + return f.status(), nil +} + +func networkAccessRequest(t *testing.T, server *Server, method, path, secret string) *httptest.ResponseRecorder { + t.Helper() + request := httptest.NewRequest(method, path, nil) + if secret != "" { + request.Header.Set("Authorization", "Bearer "+secret) + } + recorder := httptest.NewRecorder() + server.Handler().ServeHTTP(recorder, request) + return recorder +} + +// The bearer network-access routes drive the provider host: status lists +// every instance with the full status (auth URL and error included, since +// the caller holds the node bearer), connect and disconnect answer the state +// reached, an unknown provider is 404, and all of them refuse without the +// bearer with the listener's plain-text failure. +func TestProxyNetworkAccessRoutesDriveTheProviderHost(t *testing.T) { + const secret = "network-access-proxy-secret" + server := newDownloadProxyServer(t, secret) + host := &fakeProviderHost{} + server.SetNetworkAccessProviderHost(host) + + for _, route := range []struct{ method, path string }{ + {http.MethodGet, "/network-access/status"}, + {http.MethodGet, "/network-access/stub/status"}, + {http.MethodPost, "/network-access/stub/connect"}, + {http.MethodPost, "/network-access/stub/disconnect"}, + } { + denied := networkAccessRequest(t, server, route.method, route.path, "") + if denied.Code != http.StatusUnauthorized || denied.Header().Get("Content-Type") != "text/plain; charset=utf-8" { + t.Fatalf("%s %s without bearer: %d %s", route.method, route.path, denied.Code, denied.Body) + } + } + if len(host.calls) != 0 { + t.Fatalf("refused requests reached the host: %v", host.calls) + } + + status := networkAccessRequest(t, server, http.MethodGet, "/network-access/status", secret) + if status.Code != http.StatusOK || status.Header().Get("Content-Type") != "application/json" || status.Header().Get("Cache-Control") != "no-store" { + t.Fatalf("status: %d %s %q", status.Code, status.Body, status.Header()) + } + var report netaccess.HostStatusReport + if err := json.Unmarshal(status.Body.Bytes(), &report); err != nil { + t.Fatal(err) + } + if len(report.Providers) != 2 || report.Providers[0].Provider != "stub" || report.Providers[0].State != netaccess.StateDisconnected || + report.Providers[1].Provider != "down" || report.Providers[1].State != netaccess.StateUnavailable || report.Providers[1].Error == "" { + t.Fatalf("status report = %+v", report) + } + providerStatus := networkAccessRequest(t, server, http.MethodGet, "/network-access/stub/status", secret) + if providerStatus.Code != http.StatusOK || providerStatus.Header().Get("Cache-Control") != "no-store" { + t.Fatalf("provider status: %d %s", providerStatus.Code, providerStatus.Body) + } + var oneProvider netaccess.Status + if err := json.Unmarshal(providerStatus.Body.Bytes(), &oneProvider); err != nil { + t.Fatal(err) + } + if oneProvider.Provider != "stub" || oneProvider.State != netaccess.StateDisconnected { + t.Fatalf("provider status = %+v", oneProvider) + } + + connect := networkAccessRequest(t, server, http.MethodPost, "/network-access/stub/connect", secret) + if connect.Code != http.StatusOK { + t.Fatalf("connect: %d %s", connect.Code, connect.Body) + } + var connected netaccess.Status + if err := json.Unmarshal(connect.Body.Bytes(), &connected); err != nil { + t.Fatal(err) + } + if connected.State != netaccess.StateConnected || connected.Origin != "https://proxy-1.stub.test" || connected.InstallationID != 5 { + t.Fatalf("connect status = %+v", connected) + } + + disconnect := networkAccessRequest(t, server, http.MethodPost, "/network-access/stub/disconnect", secret) + if disconnect.Code != http.StatusOK { + t.Fatalf("disconnect: %d %s", disconnect.Code, disconnect.Body) + } + var disconnected netaccess.Status + if err := json.Unmarshal(disconnect.Body.Bytes(), &disconnected); err != nil { + t.Fatal(err) + } + if disconnected.State != netaccess.StateDisconnected || disconnected.Origin != "" { + t.Fatalf("disconnect status = %+v", disconnected) + } + + unknown := networkAccessRequest(t, server, http.MethodPost, "/network-access/netbird/connect", secret) + if unknown.Code != http.StatusNotFound { + t.Fatalf("unknown provider connect: %d %s", unknown.Code, unknown.Body) + } + unknownStatus := networkAccessRequest(t, server, http.MethodGet, "/network-access/netbird/status", secret) + if unknownStatus.Code != http.StatusNotFound { + t.Fatalf("unknown provider status: %d %s", unknownStatus.Code, unknownStatus.Body) + } + want := []string{"status", "status:stub", "connect:stub", "disconnect:stub", "connect:netbird", "status:netbird"} + if len(host.calls) != len(want) { + t.Fatalf("host calls = %v, want %v", host.calls, want) + } + for i := range want { + if host.calls[i] != want[i] { + t.Fatalf("host calls = %v, want %v", host.calls, want) + } + } +} + +// A proxy without a plugin host (no providers wired) answers 503 so the API's +// fan-out can say why the host is unavailable, and never 404, which would +// read as "provider not installed here". +func TestProxyNetworkAccessRoutesWithoutAHostAnswer503(t *testing.T) { + const secret = "network-access-proxy-secret" + server := newDownloadProxyServer(t, secret) + for _, route := range []struct{ method, path string }{ + {http.MethodGet, "/network-access/status"}, + {http.MethodGet, "/network-access/stub/status"}, + {http.MethodPost, "/network-access/stub/connect"}, + {http.MethodPost, "/network-access/stub/disconnect"}, + } { + got := networkAccessRequest(t, server, route.method, route.path, secret) + if got.Code != http.StatusServiceUnavailable { + t.Fatalf("%s %s without a host: %d %s", route.method, route.path, got.Code, got.Body) + } + } +} + +type blockedProviderHost struct { + fakeProviderHost + started chan struct{} +} + +func (f *blockedProviderHost) HostNetworkAccessProviderStatus(ctx context.Context, provider string) (netaccess.Status, error) { + if provider == "slow" { + close(f.started) + <-ctx.Done() + return netaccess.Status{Provider: provider, State: netaccess.StateUnavailable}, nil + } + return f.fakeProviderHost.HostNetworkAccessProviderStatus(ctx, provider) +} + +func TestProxyNetworkAccessProviderStatusDoesNotWaitForOtherProviders(t *testing.T) { + const secret = "network-access-proxy-secret" + server := newDownloadProxyServer(t, secret) + host := &blockedProviderHost{started: make(chan struct{})} + server.SetNetworkAccessProviderHost(host) + router := server.Handler() + ctx, cancel := context.WithCancel(context.Background()) + slowDone := make(chan struct{}) + t.Cleanup(func() { + cancel() + <-slowDone + }) + go func() { + defer close(slowDone) + request := httptest.NewRequest(http.MethodGet, "/network-access/slow/status", nil).WithContext(ctx) + request.Header.Set("Authorization", "Bearer "+secret) + router.ServeHTTP(httptest.NewRecorder(), request) + }() + select { + case <-host.started: + case <-time.After(5 * time.Second): + t.Fatal("slow provider status was not requested") + } + request := httptest.NewRequest(http.MethodGet, "/network-access/stub/status", nil) + request.Header.Set("Authorization", "Bearer "+secret) + recorder := httptest.NewRecorder() + router.ServeHTTP(recorder, request) + if recorder.Code != http.StatusOK { + t.Fatalf("healthy provider beside blocked provider: %d %s", recorder.Code, recorder.Body) + } + var status netaccess.Status + if err := json.Unmarshal(recorder.Body.Bytes(), &status); err != nil { + t.Fatal(err) + } + if status.Provider != "stub" || status.State != netaccess.StateDisconnected { + t.Fatalf("healthy provider status = %+v", status) + } + select { + case <-slowDone: + t.Fatal("slow provider finished before it was released") + default: + } +} + +// The proxy listener validates the ingress token a provider stamps on the +// requests it forwards, so an unknown token is refused before any handler and +// a valid one is stripped. +func TestProxyListenerValidatesIngressTokens(t *testing.T) { + server := newDownloadProxyServer(t, "secret") + registry := netaccess.NewRegistry() + server.SetIngressTokens(registry) + token, err := registry.Issue(5, "stub") + if err != nil { + t.Fatal(err) + } + + forged := httptest.NewRequest(http.MethodGet, "/api/v1/health", nil) + forged.Header.Set(netaccess.IngressTokenHeader, "not-the-token") + recorder := httptest.NewRecorder() + server.Handler().ServeHTTP(recorder, forged) + if recorder.Code != http.StatusForbidden { + t.Fatalf("forged token: %d %s", recorder.Code, recorder.Body) + } + + genuine := httptest.NewRequest(http.MethodGet, "/api/v1/health", nil) + genuine.Header.Set(netaccess.IngressTokenHeader, token) + recorder = httptest.NewRecorder() + server.Handler().ServeHTTP(recorder, genuine) + if recorder.Code != http.StatusOK { + t.Fatalf("genuine token: %d %s", recorder.Code, recorder.Body) + } +} diff --git a/internal/proxy/protocol_registry.go b/internal/proxy/protocol_registry.go index 9813a222a1..a3682fbae2 100644 --- a/internal/proxy/protocol_registry.go +++ b/internal/proxy/protocol_registry.go @@ -1,6 +1,9 @@ package proxy import ( + "net/http" + + "github.com/Silo-Server/silo-server/internal/netaccess" "github.com/Silo-Server/silo-server/internal/playback" "github.com/Silo-Server/silo-server/internal/workerprotocol" "github.com/danielgtaylor/huma/v2" @@ -27,3 +30,32 @@ func ProtocolControls(schemas huma.Registry) []workerprotocol.Operation { reprobe, } } + +// ProtocolNetworkAccess describes the bearer routes the API server fans its +// network access admin operations out to: one read of every provider instance +// on this proxy, and per-provider status, connect and disconnect. They are worker +// contracts, never native API aliases. +func ProtocolNetworkAccess(schemas huma.Registry) []workerprotocol.Operation { + const listener = "proxy" + const providerPathIn = "path" + status := workerprotocol.JSONRead[netaccess.HostStatusReport](schemas, listener, "/network-access/status", "(*internal/proxy.Server).handleNetworkAccessStatus", 401, 500, 503) + status.Description = "Live status of every network access provider plugin instance running on this proxy, read from the plugin with a ten-second timeout each. Carries auth_url and error because the caller holds the node bearer. 503 when the proxy hosts no plugins." + provider := []*huma.Param{{Name: "provider", In: providerPathIn, Required: true, Schema: &huma.Schema{Type: huma.TypeString}, Description: "Provider slug declared by the plugin manifest, e.g. tailscale."}} + providerStatus := workerprotocol.JSONRead[netaccess.Status](schemas, listener, "/network-access/{provider}/status", "(*internal/proxy.Server).handleNetworkAccessProviderStatus", 401, 404, 500, 503) + providerStatus.Parameters = provider + providerStatus.Description = "Live status of the named network access provider on this proxy, with a ten-second timeout. Reads only this provider, so another provider's timeout does not hide its status. 404 for a provider no enabled installation declares; 503 when the proxy hosts no plugins." + command := func(path, handler, description string) workerprotocol.Operation { + op := workerprotocol.JSONRead[netaccess.Status](schemas, listener, "/network-access/{provider}"+path, handler, 401, 404, 500, 503) + op.Method = http.MethodPost + op.Parameters = provider + op.RetrySafety = "natural_idempotent" + op.Description = description + return op + } + return []workerprotocol.Operation{ + status, + providerStatus, + command("/connect", "(*internal/proxy.Server).handleNetworkAccessConnect", "Ask the provider on this proxy to bring its overlay identity up; answers the state reached within ten seconds. Repeating converges on one connected instance. 404 for a provider no enabled installation declares."), + command("/disconnect", "(*internal/proxy.Server).handleNetworkAccessDisconnect", "Ask the provider on this proxy to tear its overlay listener down; answers the state reached within ten seconds. Repeating converges on disconnected. 404 for a provider no enabled installation declares."), + } +} diff --git a/internal/proxy/server.go b/internal/proxy/server.go index 485423fc8a..435153c3de 100644 --- a/internal/proxy/server.go +++ b/internal/proxy/server.go @@ -28,6 +28,7 @@ import ( "github.com/Silo-Server/silo-server/internal/downloadprepare" "github.com/Silo-Server/silo-server/internal/downloads" "github.com/Silo-Server/silo-server/internal/httpstream" + "github.com/Silo-Server/silo-server/internal/netaccess" "github.com/Silo-Server/silo-server/internal/nodeconfig" "github.com/Silo-Server/silo-server/internal/nodemetrics" "github.com/Silo-Server/silo-server/internal/noderouting" @@ -92,6 +93,37 @@ type Server struct { // countProbesInFlight overrides the detached-probe count the re-probe route // refuses on. Tests set it; production leaves it nil. countProbesInFlight func() int + + // networkAccess reports the network access provider plugins running beside + // this proxy, for the API's health pull. Nil until a plugin host is wired + // (SetNetworkAccessStatus), which leaves the health field absent — the same + // as a build that predates it. + networkAccess NetworkAccessStatusSource + // networkAccessHost drives the provider instances for the bearer + // network-access routes; nil answers 503 there. See network_access.go. + networkAccessHost NetworkAccessProviderHost + // ingressTokens validates the X-Silo-Ingress-Token a provider plugin + // stamps on requests it proxies to this listener, so the access path is + // known here too. Nil accepts no tokens (the header is still stripped). + ingressTokens *netaccess.Registry +} + +// SetIngressTokens wires the ingress-token registry the listener validates +// provider-stamped requests against. Call it during construction. +func (s *Server) SetIngressTokens(registry *netaccess.Registry) { + s.ingressTokens = registry +} + +// NetworkAccessStatusSource answers what a proxy reports about its network +// access providers. *netaccess.StatusCache satisfies it. +type NetworkAccessStatusSource interface { + NodeNetworkAccess() netaccess.NodeNetworkAccess +} + +// SetNetworkAccessStatus wires the provider status source /health reports +// from. Call it during construction; nil leaves the field absent. +func (s *Server) SetNetworkAccessStatus(source NetworkAccessStatusSource) { + s.networkAccess = source } type remoteArtifactMissReporter interface { @@ -228,6 +260,9 @@ func (s *Server) router() chi.Router { if s.clientIP != nil { r.Use(clientip.Middleware(s.clientIP)) } + if s.ingressTokens != nil { + r.Use(netaccess.Middleware(s.ingressTokens)) + } // hls.js uses XHR for manifest/segment fetches which are subject to // CORS when the proxy runs on a different origin than the web app. r.Use(cors.Handler(cors.Options{ @@ -289,6 +324,12 @@ func (s *Server) router() chi.Router { r.Post("/admin/reload-config", s.handleReloadConfig) r.Post("/admin/reprobe-capabilities", s.handleReprobeCapabilities) r.Get("/status", s.handleStatus) + // Network access providers running beside this proxy; the API fans its + // admin status/connect/disconnect out to these with the node bearer. + r.Get("/network-access/status", s.handleNetworkAccessStatus) + r.Get("/network-access/{provider}/status", s.handleNetworkAccessProviderStatus) + r.Post("/network-access/{provider}/connect", s.handleNetworkAccessConnect) + r.Post("/network-access/{provider}/disconnect", s.handleNetworkAccessDisconnect) }) return r } @@ -474,6 +515,12 @@ type healthResponse struct { SampledAt time.Time `json:"sampled_at,omitzero"` // Build identifies the binary this proxy runs; see transcodenode.HealthResponse. Build buildinfo.Info `json:"build"` + // NetworkAccess is the last status of each network access provider plugin + // running beside this proxy, keyed by provider slug. The API stores it on + // the node row and hands overlay clients the matching origin. Absent when no + // provider runs here; carries no auth URL or error text, since this route + // takes no credential. + NetworkAccess netaccess.NodeNetworkAccess `json:"network_access,omitempty"` } func (s *Server) handleHealth(w http.ResponseWriter, _ *http.Request) { @@ -482,6 +529,10 @@ func (s *Server) handleHealth(w http.ResponseWriter, _ *http.Request) { activeJobs = s.tracker.ActiveCount() } snapshot := s.metrics.Snapshot().RedactPaths() + var networkAccess netaccess.NodeNetworkAccess + if s.networkAccess != nil { + networkAccess = s.networkAccess.NodeNetworkAccess() + } w.Header().Set("Content-Type", "application/json") json.NewEncoder(w).Encode(healthResponse{ Status: "ok", @@ -493,6 +544,7 @@ func (s *Server) handleHealth(w http.ResponseWriter, _ *http.Request) { Attribution: snapshot.Attribution, SampledAt: snapshot.SampledAt, Build: buildinfo.Current(), + NetworkAccess: networkAccess, }) } diff --git a/internal/proxy/testdata/media_routes.txt b/internal/proxy/testdata/media_routes.txt index d6bdc4a30f..7087f96f95 100644 --- a/internal/proxy/testdata/media_routes.txt +++ b/internal/proxy/testdata/media_routes.txt @@ -7,6 +7,10 @@ GET /downloads/file/{token} media transfer viewer_egress false true HEAD /downloads/file/{token} media transfer viewer_egress false true GET /hw-capabilities non-media GET /metrics non-media +GET /network-access/status non-media +POST /network-access/{provider}/connect non-media +POST /network-access/{provider}/disconnect non-media +GET /network-access/{provider}/status non-media GET /status non-media GET /stream/direct/{token} media playback viewer_egress true true HEAD /stream/direct/{token} media playback viewer_egress true true @@ -33,6 +37,10 @@ GET /downloads/file/{token} media transfer viewer_egress false true HEAD /downloads/file/{token} media transfer viewer_egress false true GET /hw-capabilities non-media GET /metrics non-media +GET /network-access/status non-media +POST /network-access/{provider}/connect non-media +POST /network-access/{provider}/disconnect non-media +GET /network-access/{provider}/status non-media GET /status non-media GET /stream/direct/{token} media playback viewer_egress true true HEAD /stream/direct/{token} media playback viewer_egress true true diff --git a/internal/routeinventory/classify.go b/internal/routeinventory/classify.go index 38fd6509e9..f78fb28667 100644 --- a/internal/routeinventory/classify.go +++ b/internal/routeinventory/classify.go @@ -822,6 +822,10 @@ var traitOnlyRules = []authRule{ var infrastructureMiddleware = []string{ "apimw.RequestID", mwRequestID, "middleware.Recoverer", "apimw.RequestLogger", "apimw.Metrics", "httpstream.CompressExcept", "httpstream.CompressWithExclusions", "clientip.Middleware", "activitylog.NewMiddleware", + // netaccess.Middleware validates and strips the network access ingress + // token on every listener; it records the access path and never grants + // or changes authorization. + "netaccess.Middleware", } func classifyAuth(middleware []string) (string, []string) { diff --git a/internal/streamtoken/token.go b/internal/streamtoken/token.go index 70c3e7d005..ffe77714ea 100644 --- a/internal/streamtoken/token.go +++ b/internal/streamtoken/token.go @@ -40,23 +40,24 @@ const ( // (uid/pid/mfid) are lookup keys re-resolved against the authority on // reconstruct; they are never trusted on their own. type Claims struct { - SessionID string `json:"sid"` - MediaPath string `json:"path"` - PlayMethod string `json:"method"` - TranscodeAudio bool `json:"ta,omitempty"` - TranscodeNode string `json:"tnode,omitempty"` - TranscodeTransportID string `json:"tid,omitempty"` - RoutingWorkload string `json:"rwl,omitempty"` - RoutingExecution string `json:"rex,omitempty"` - RoutingExecutionNodeID int `json:"rxnid,omitzero"` - RoutingEgress string `json:"reg,omitempty"` - RoutingEgressNodeID int `json:"renid,omitempty"` - TargetCodec string `json:"tc,omitempty"` - TargetRes string `json:"tres,omitempty"` - AudioCodec string `json:"ac,omitempty"` - AudioChannels int `json:"ach,omitempty"` - AudioTrackIndex int `json:"ati,omitempty"` - AudioOnly bool `json:"ao,omitempty"` + SessionID string `json:"sid"` + MediaPath string `json:"path"` + PlayMethod string `json:"method"` + TranscodeAudio bool `json:"ta,omitempty"` + TranscodeNode string `json:"tnode,omitempty"` + TranscodeTransportID string `json:"tid,omitempty"` + RoutingNetworkProvider *string `json:"rnp,omitempty"` + RoutingWorkload string `json:"rwl,omitempty"` + RoutingExecution string `json:"rex,omitempty"` + RoutingExecutionNodeID int `json:"rxnid,omitzero"` + RoutingEgress string `json:"reg,omitempty"` + RoutingEgressNodeID int `json:"renid,omitempty"` + TargetCodec string `json:"tc,omitempty"` + TargetRes string `json:"tres,omitempty"` + AudioCodec string `json:"ac,omitempty"` + AudioChannels int `json:"ach,omitempty"` + AudioTrackIndex int `json:"ati,omitempty"` + AudioOnly bool `json:"ao,omitempty"` // DVProfile is the file's Dolby Vision profile (0 = none); remux nodes // use it to strip dangling profile 7 RPUs. Absent in older tokens, which // decodes as 0 (no strip — the pre-existing behavior). diff --git a/internal/worker/reconciler.go b/internal/worker/reconciler.go index 6f54bea987..39f01777ce 100644 --- a/internal/worker/reconciler.go +++ b/internal/worker/reconciler.go @@ -46,6 +46,7 @@ type SessionSync struct { TargetBitrateKbps int TranscodeHWAccel string ToneMapMode string + RoutingNetworkProvider *string RoutingWorkload string RoutingExecution string RoutingExecutionNodeID int @@ -199,8 +200,8 @@ func (r *Reconciler) ReconcileNodeSessions(ctx context.Context, reportingNode st transcode_hw_accel, tone_map_mode, routing_workload, routing_execution, routing_execution_node_id, routing_execution_node_url, routing_egress, routing_egress_node_id, routing_egress_node_url, - position_seconds, is_paused, has_websocket, compat_origin) - VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9, NOW(), $10::inet, $11, $12, $13, $14, $15, $16, $17, $18, $19, $20, $21, $22, $23, $24, $25, $26, $27, $28, $29, $30, $31, $32, $33, $34, $35, $36, $37) + position_seconds, is_paused, has_websocket, compat_origin, routing_network_provider) + VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9, NOW(), $10::inet, $11, $12, $13, $14, $15, $16, $17, $18, $19, $20, $21, $22, $23, $24, $25, $26, $27, $28, $29, $30, $31, $32, $33, $34, $35, $36, $37, $38) ON CONFLICT (session_id) DO UPDATE SET user_id = EXCLUDED.user_id, profile_id = EXCLUDED.profile_id, @@ -238,6 +239,7 @@ func (r *Reconciler) ReconcileNodeSessions(ctx context.Context, reportingNode st is_paused = EXCLUDED.is_paused, has_websocket = EXCLUDED.has_websocket, compat_origin = EXCLUDED.compat_origin, + routing_network_provider = EXCLUDED.routing_network_provider, last_sync_at = NOW() `, s.SessionID, s.UserID, s.ProfileID, s.MediaFileID, nullableInt(s.RequestedMediaFileID), s.PlayMethod, sessionNode, s.StartedAt, s.UpdatedAt, nullableIP(s.ClientIP), @@ -250,7 +252,7 @@ func (r *Reconciler) ReconcileNodeSessions(ctx context.Context, reportingNode st nullableString(s.TranscodeHWAccel), nullableString(s.ToneMapMode), nullableString(s.RoutingWorkload), nullableString(s.RoutingExecution), nullableInt(s.RoutingExecutionNodeID), nullableString(s.RoutingExecutionNodeURL), nullableString(s.RoutingEgress), nullableInt(s.RoutingEgressNodeID), nullableString(s.RoutingEgressNodeURL), normalizePositionSeconds(s.PositionSeconds), - s.IsPaused, s.HasWebSocket, s.IsJellyfinCompat) + s.IsPaused, s.HasWebSocket, s.IsJellyfinCompat, s.RoutingNetworkProvider) if err != nil { return fmt.Errorf("upserting session %s: %w", s.SessionID, err) } @@ -345,7 +347,8 @@ func loadNodeSessionsSnapshot(ctx context.Context, tx pgx.Tx, reportingNode stri COALESCE(position_seconds, 0), COALESCE(is_paused, FALSE), COALESCE(has_websocket, FALSE), - COALESCE(compat_origin, FALSE) + COALESCE(compat_origin, FALSE), + routing_network_provider FROM playback_sessions_sync WHERE COALESCE(reporting_node, '') = $1 ORDER BY session_id @@ -396,6 +399,7 @@ func loadNodeSessionsSnapshot(ctx context.Context, tx pgx.Tx, reportingNode stri &s.IsPaused, &s.HasWebSocket, &s.IsJellyfinCompat, + &s.RoutingNetworkProvider, ); err != nil { return nil, err } @@ -452,6 +456,7 @@ func sessionSnapshotsEqual(left, right []SessionSync) bool { left[i].TargetBitrateKbps != right[i].TargetBitrateKbps || left[i].TranscodeHWAccel != right[i].TranscodeHWAccel || left[i].ToneMapMode != right[i].ToneMapMode || + !equalOptionalString(left[i].RoutingNetworkProvider, right[i].RoutingNetworkProvider) || left[i].RoutingWorkload != right[i].RoutingWorkload || left[i].RoutingExecution != right[i].RoutingExecution || left[i].RoutingExecutionNodeID != right[i].RoutingExecutionNodeID || @@ -621,3 +626,10 @@ func (r *Reconciler) syncOnce(ctx context.Context) error { func (r *Reconciler) Stop() { close(r.stop) } + +func equalOptionalString(a, b *string) bool { + if a == nil || b == nil { + return a == b + } + return *a == *b +} diff --git a/internal/worker/reconciler_postgres_test.go b/internal/worker/reconciler_postgres_test.go index 21918c1c6d..0fa8341ad5 100644 --- a/internal/worker/reconciler_postgres_test.go +++ b/internal/worker/reconciler_postgres_test.go @@ -46,6 +46,27 @@ INSERT INTO users VALUES(1),(2);`); err != nil { if err = r.ReconcileNodeSessions(t.Context(), "node", sessions); err != nil { t.Fatal(err) } + + // Default, overlay and unknown must survive SQL independently, including + // a replan from an overlay back to the default network. + for _, provider := range []*string{new("tailscale"), new(""), nil} { + sessions[1].RoutingNetworkProvider = provider + if err := r.ReconcileNodeSessions(t.Context(), "node", sessions); err != nil { + t.Fatal(err) + } + tx, err := pool.Begin(t.Context()) + if err != nil { + t.Fatal(err) + } + snapshot, err := loadNodeSessionsSnapshot(t.Context(), tx, "node") + _ = tx.Rollback(t.Context()) + if err != nil { + t.Fatal(err) + } + if len(snapshot) != 2 || !equalOptionalString(snapshot[1].RoutingNetworkProvider, provider) { + t.Fatalf("network route round trip: %#v", snapshot) + } + } if _, err = pool.Exec(t.Context(), "DELETE FROM users WHERE id=1"); err != nil { t.Fatal(err) } diff --git a/internal/worker/reconciler_test.go b/internal/worker/reconciler_test.go index 582e86425f..aea30e4b93 100644 --- a/internal/worker/reconciler_test.go +++ b/internal/worker/reconciler_test.go @@ -102,3 +102,15 @@ func TestSyncNowCoalescesPendingPass(t *testing.T) { t.Fatal("no follow-up snapshot capture ran; the coalesced sync was lost") } } + +func TestSessionSnapshotsCompareNetworkProviderValues(t *testing.T) { + for _, a := range []*string{nil, new(""), new("tailscale")} { + for _, b := range []*string{nil, new(""), new("tailscale")} { + want := a == nil && b == nil || a != nil && b != nil && *a == *b + got := sessionSnapshotsEqual([]SessionSync{{SessionID: "s", RoutingNetworkProvider: a}}, []SessionSync{{SessionID: "s", RoutingNetworkProvider: b}}) + if got != want { + t.Fatalf("provider comparison = %v, want %v (%v, %v)", got, want, a, b) + } + } + } +} diff --git a/migrations/sql/20260914071026_plugin_instance_state.sql b/migrations/sql/20260914071026_plugin_instance_state.sql new file mode 100644 index 0000000000..490643d761 --- /dev/null +++ b/migrations/sql/20260914071026_plugin_instance_state.sql @@ -0,0 +1,22 @@ +-- +goose Up +-- +goose StatementBegin +-- Per-instance private state for resident plugins (network access providers +-- keep their overlay node keys here). One row per installation, host scope and +-- key. host_scope is derived by the host ('api' or 'node:'), +-- never supplied by the plugin. state_value is an encrypted GCM envelope +-- bound to the row: AAD = RowAAD("plugin_instance_state", "state_value", +-- "::"). The API never returns it. +CREATE TABLE plugin_instance_state ( + plugin_installation_id BIGINT NOT NULL REFERENCES plugin_installations(id) ON DELETE CASCADE, + host_scope TEXT NOT NULL, + state_key TEXT NOT NULL, + state_value BYTEA NOT NULL, + updated_at TIMESTAMPTZ NOT NULL DEFAULT NOW(), + PRIMARY KEY (plugin_installation_id, host_scope, state_key) +); +-- +goose StatementEnd + +-- +goose Down +-- +goose StatementBegin +DROP TABLE IF EXISTS plugin_instance_state; +-- +goose StatementEnd diff --git a/migrations/sql/20260914084130_node_network_access.sql b/migrations/sql/20260914084130_node_network_access.sql new file mode 100644 index 0000000000..bd97676bc3 --- /dev/null +++ b/migrations/sql/20260914084130_node_network_access.sql @@ -0,0 +1,17 @@ +-- +goose Up +-- +goose StatementBegin +-- Last network access provider status a proxy node reported on its health +-- check, keyed by provider slug: +-- {"tailscale": {"state": "connected", "origin": "https://proxy-1.tail1234.ts.net", +-- "hostname": "proxy-1.tail1234.ts.net", "updated_at": "..."}} +-- Rewritten by every health sweep from the node's own report; an empty object +-- means the node reports no providers (or predates the field). Stream URLs +-- handed to a client that arrived through a provider are built on the matching +-- connected origin, never on the node's LAN/public URL. +ALTER TABLE stream_nodes ADD COLUMN network_access JSONB NOT NULL DEFAULT '{}'::jsonb; +-- +goose StatementEnd + +-- +goose Down +-- +goose StatementBegin +ALTER TABLE stream_nodes DROP COLUMN IF EXISTS network_access; +-- +goose StatementEnd diff --git a/migrations/sql/20260914210643_plugin_runtime_generation.sql b/migrations/sql/20260914210643_plugin_runtime_generation.sql new file mode 100644 index 0000000000..f73d4b4815 --- /dev/null +++ b/migrations/sql/20260914210643_plugin_runtime_generation.sql @@ -0,0 +1,6 @@ +-- +goose Up +ALTER TABLE plugin_installations + ADD COLUMN runtime_generation BIGINT NOT NULL DEFAULT 0; + +-- +goose Down +ALTER TABLE plugin_installations DROP COLUMN runtime_generation; diff --git a/migrations/sql/20260920210643_add_session_network_provider.sql b/migrations/sql/20260920210643_add_session_network_provider.sql new file mode 100644 index 0000000000..882d02f5fb --- /dev/null +++ b/migrations/sql/20260920210643_add_session_network_provider.sql @@ -0,0 +1,7 @@ +-- +goose Up +ALTER TABLE playback_sessions_sync ADD COLUMN routing_network_provider TEXT; +COMMENT ON COLUMN playback_sessions_sync.routing_network_provider IS + 'Validated access provider selected for playback; empty means default network, NULL means unknown.'; + +-- +goose Down +ALTER TABLE playback_sessions_sync DROP COLUMN routing_network_provider; diff --git a/web/src/api/types.ts b/web/src/api/types.ts index 0ba5bf53db..8f65a73de4 100644 --- a/web/src/api/types.ts +++ b/web/src/api/types.ts @@ -2523,6 +2523,8 @@ export interface AdminSession { /** Resolved playback workload and route. Node IDs/names are omitted when * that phase runs on the integrated API process (or direct play has no * executor). */ + /** Empty means default network; absent means unknown (older session). */ + routing_network_provider?: string; routing_workload?: string; routing_execution?: string; routing_execution_node_id?: number; @@ -3587,6 +3589,22 @@ export interface PluginCatalogEntry { metadata?: Record; } +export type PluginRuntimeState = "stopped" | "starting" | "running" | "backoff" | "failed"; + +/** + * Process state of one installation. `resident` marks a plugin the server + * supervises (started at boot, restarted after a crash); `backoff` and + * `failed` only occur for those. + */ +export interface PluginRuntime { + resident: boolean; + state: PluginRuntimeState; + restart_count: number; + last_error?: string; + last_started_at?: string; + next_restart_at?: string; +} + export interface PluginInstallation { id: number; repository_id?: number | null; @@ -3594,6 +3612,7 @@ export interface PluginInstallation { version: string; install_path: string; enabled: boolean; + runtime: PluginRuntime; capabilities: PluginCapability[]; global_config_schema: PluginConfigSchema[]; user_config_schema: PluginConfigSchema[]; diff --git a/web/src/api/v2/operations.ts b/web/src/api/v2/operations.ts index 2d16dd7574..f27a23bb2a 100644 --- a/web/src/api/v2/operations.ts +++ b/web/src/api/v2/operations.ts @@ -153,6 +153,7 @@ export const v2Operations = { "GET /api/v2/admin/markers/history": "listAdminMarkerHistory", "GET /api/v2/admin/markers/items/{id}/history": "listAdminItemMarkerHistory", "GET /api/v2/admin/markers/providers": "listAdminMarkerProviders", + "GET /api/v2/admin/network-access/{provider}/status": "getAdminNetworkAccessStatus", "GET /api/v2/admin/node-sessions": "listAdminNodeSessions", "GET /api/v2/admin/nodes": "listAdminNodes", "GET /api/v2/admin/notifications/server-channels": "listAdminNotificationServerChannels", @@ -324,6 +325,7 @@ export const v2Operations = { "GET /api/v2/library/{id}/user-collections": "listLibraryUserCollections", "GET /api/v2/markers/files/{file_id}": "getFileMarkers", "GET /api/v2/markers/items/{item_id}": "getItemMarkers", + "GET /api/v2/network-access/capabilities": "getNetworkAccessCapabilities", "GET /api/v2/notifications": "listNotifications", "GET /api/v2/notifications/capabilities": "getNotificationCapabilities", "GET /api/v2/notifications/discord-preferences": "getNotificationDiscordPreferences", @@ -524,6 +526,8 @@ export const v2Operations = { "POST /api/v2/admin/literary-works/matches/ignore": "ignoreAdminLiteraryMatch", "POST /api/v2/admin/logs/ws-ticket": "createAdminLogsSocketTicket", "POST /api/v2/admin/markers/providers/{provider}/validate": "validateAdminMarkerProvider", + "POST /api/v2/admin/network-access/{provider}/connect": "connectNetworkAccess", + "POST /api/v2/admin/network-access/{provider}/disconnect": "disconnectNetworkAccess", "POST /api/v2/admin/nodes": "createAdminNode", "POST /api/v2/admin/nodes/force-reload": "forceReloadAdminNodes", "POST /api/v2/admin/nodes/{id}/check": "checkAdminNode", @@ -541,6 +545,7 @@ export const v2Operations = { "POST /api/v2/admin/people/{id}/refresh": "refreshAdminPerson", "POST /api/v2/admin/plugins/installations": "createAdminPluginInstallation", "POST /api/v2/admin/plugins/installations/{id}/config/test": "testAdminPluginInstallationConfig", + "POST /api/v2/admin/plugins/installations/{id}/restart": "restartAdminPluginInstallation", "POST /api/v2/admin/plugins/installations/{id}/update": "applyAdminPluginUpdate", "POST /api/v2/admin/plugins/repositories": "createAdminPluginRepository", "POST /api/v2/admin/plugins/uploads": "uploadAdminPluginInstallation", diff --git a/web/src/api/v2/schema.ts b/web/src/api/v2/schema.ts index 69bb2c6f8b..cb7a4212c9 100644 --- a/web/src/api/v2/schema.ts +++ b/web/src/api/v2/schema.ts @@ -2178,6 +2178,57 @@ export interface paths { patch?: never; trace?: never; }; + "/api/v2/admin/network-access/{provider}/connect": { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + get?: never; + put?: never; + /** Ask the provider on the named hosts (every host when hosts is omitted) to bring its overlay identity up and start proxying. Answers 202 with the state each host reached within ten seconds; enrollment may continue in the background (awaiting_authorization carries the auth_url), so poll status for the final state. Repeating the request converges on one connected instance per host. Hosts not named answer their current status; an unknown host id is 422. */ + post: operations["connectNetworkAccess"]; + delete?: never; + options?: never; + head?: never; + patch?: never; + trace?: never; + }; + "/api/v2/admin/network-access/{provider}/disconnect": { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + get?: never; + put?: never; + /** Ask the provider on the named hosts (every host when hosts is omitted) to tear its overlay listener down and clear its desired-connected intent. Answers 202 with the state each host reached within ten seconds. Repeating the request converges on disconnected. Hosts not named answer their current status; an unknown host id is 422. */ + post: operations["disconnectNetworkAccess"]; + delete?: never; + options?: never; + head?: never; + patch?: never; + trace?: never; + }; + "/api/v2/admin/network-access/{provider}/status": { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + /** Read the provider's live state on every host that runs it, asking each plugin instance directly with a ten-second timeout. A host whose plugin process is not running answers state unavailable with the supervisor's last error. An unknown provider slug is 404. */ + get: operations["getAdminNetworkAccessStatus"]; + put?: never; + post?: never; + delete?: never; + options?: never; + head?: never; + patch?: never; + trace?: never; + }; "/api/v2/admin/node-sessions": { parameters: { query?: never; @@ -2661,6 +2712,23 @@ export interface paths { patch?: never; trace?: never; }; + "/api/v2/admin/plugins/installations/{id}/restart": { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + get?: never; + put?: never; + /** Stop the installation's process and, for a resident plugin (one the server supervises, such as a network access provider), start it again with a fresh failure budget; the response's runtime reports the outcome, including a launch that failed. A non-resident plugin is only stopped and launches on its next use. A disabled installation is 409. Repeating the request converges on one running process. */ + post: operations["restartAdminPluginInstallation"]; + delete?: never; + options?: never; + head?: never; + patch?: never; + trace?: never; + }; "/api/v2/admin/plugins/installations/{id}/task-bindings/{capability_id}": { parameters: { query?: never; @@ -7122,6 +7190,23 @@ export interface paths { patch?: never; trace?: never; }; + "/api/v2/network-access/capabilities": { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + /** List the installed network access providers (overlay networks such as Tailscale that reach this server without port forwarding). available when an enabled plugin declares one; not_configured otherwise. Read from plugin manifests; no plugin is launched and no health is implied. */ + get: operations["getNetworkAccessCapabilities"]; + put?: never; + post?: never; + delete?: never; + options?: never; + head?: never; + patch?: never; + trace?: never; + }; "/api/v2/notifications": { parameters: { query?: never; @@ -12549,6 +12634,10 @@ export interface components { /** Format: int64 */ max_jobs: number | null; name: string; + /** @description Last network access provider status the node reported on its health check, keyed by provider slug. Omitted when the node reports no providers. */ + network_access?: { + [key: string]: components["schemas"]["AdminNodeNetworkAccess"]; + }; physical_gpu_keys?: string[]; public_url?: string; /** @enum {string} */ @@ -12577,6 +12666,19 @@ export interface components { health_persisted: boolean; healthy: boolean; }; + AdminNodeNetworkAccess: { + /** @description Overlay DNS name of the node. */ + hostname?: string; + /** @description scheme://host[:port] clients on the provider's overlay use to reach this node. Only used while state is connected. */ + origin?: string; + /** @description disconnected | awaiting_authorization | connecting | connected | error */ + state: string; + /** + * Format: date-time + * @description When the node last heard from the provider, on the node's clock. + */ + updated_at?: string; + }; AdminNodeReloadOutputBody: { results: components["schemas"]["AdminNodeReloadResult"][]; }; @@ -12937,6 +13039,8 @@ export interface components { */ routing_execution_node_id?: string; routing_execution_node_name?: string; + /** @description Access network selected for playback: empty means default; absent means unknown; otherwise the validated provider identifier. */ + routing_network_provider?: string; routing_workload?: string; /** Format: int64 */ season_number?: number; @@ -12992,6 +13096,7 @@ export interface components { effective_play_method: boolean; effective_play_method_values: string[]; is_jellyfin_client: boolean; + network_access_route: boolean; node_observations: boolean; node_routing: boolean; /** @description Opaque revision of this document */ @@ -13273,6 +13378,7 @@ export interface components { repository_id?: string; repository_name?: string; routes: components["schemas"]["PluginRoute"][]; + runtime: components["schemas"]["AdminPluginRuntime"]; /** @enum {string} */ source_kind: "silo" | "approved_community" | "external"; task_bindings: components["schemas"]["AdminPluginTaskBinding"][]; @@ -13351,6 +13457,32 @@ export interface components { enabled?: boolean; url?: string; }; + AdminPluginRuntime: { + /** @description Why the process last stopped or failed to start */ + last_error?: string; + /** + * Format: date-time + * @description RFC 3339 instant in UTC with millisecond precision + */ + last_started_at?: string; + /** + * Format: date-time + * @description Scheduled automatic restart while in backoff + */ + next_restart_at?: string; + /** @description True when the server supervises this plugin's process */ + resident: boolean; + /** + * Format: int64 + * @description Automatic restarts since the plugin last ran stably or was restarted by an administrator + */ + restart_count: number; + /** + * @description Process state; backoff and failed occur only for resident plugins + * @enum {string} + */ + state: "stopped" | "starting" | "running" | "backoff" | "failed"; + }; AdminPluginTaskBinding: { capability_id: string; /** @@ -19940,6 +20072,75 @@ export interface components { */ present: boolean; }; + NetworkAccessCapabilities: { + /** @description Whether the current principal may use the capability */ + allowed: boolean; + providers: components["schemas"]["NetworkAccessProviderSummary"][]; + /** @description Opaque revision of this document */ + revision: string; + /** + * @description Support and configuration state, not health + * @enum {string} + */ + state: "available" | "disabled" | "not_configured" | "unsupported"; + }; + NetworkAccessCommand: { + /** @description Host ids to act on (api, node:); omitted means every host */ + hosts?: string[]; + }; + NetworkAccessHostRef: { + /** @description api for the API server; node: for a proxy node */ + id: string; + /** @description Server name for the API host; node name for a proxy */ + name: string; + /** @enum {string} */ + role: "api" | "proxy"; + }; + NetworkAccessHostStatus: { + /** @description Overlay IP addresses */ + addresses: string[]; + /** @description Interactive enrollment URL, only while awaiting_authorization; admin-only, never logged */ + auth_url?: string; + error?: string; + host: components["schemas"]["NetworkAccessHostRef"]; + /** @description Overlay DNS name */ + hostname?: string; + /** @description scheme://host[:port] clients reach the API listener at over the overlay */ + origin?: string; + provider_version?: string; + /** @description the provider's own state string when it is not one of the published states */ + raw_state?: string; + /** + * @description Published host state; provider states outside this vocabulary map to error + * @enum {string} + */ + state: + | "disconnected" + | "awaiting_authorization" + | "connecting" + | "connected" + | "error" + | "unavailable"; + /** + * Format: date-time + * @description When the host last heard from the provider; absent while unavailable + */ + updated_at?: string; + }; + NetworkAccessProviderSummary: { + display_name: string; + /** + * @description Plugin installation declaring network_access_provider.v1 + * @example 1 + */ + installation_id: string; + /** @description Stable provider slug used in the admin routes, e.g. tailscale */ + provider: string; + }; + NetworkAccessStatus: { + hosts: components["schemas"]["NetworkAccessHostStatus"][]; + provider: string; + }; NodeHWAccel: { error?: string; node_name?: string; @@ -46225,6 +46426,392 @@ export interface operations { }; }; }; + connectNetworkAccess: { + parameters: { + query?: never; + header?: { + /** @description Optional. When present, it must name the authenticated account's primary profile; an absent header is accepted. */ + "X-Profile-Id"?: string; + /** @description Verification proof for a PIN-locked profile, issued by POST /api/v2/profiles/{id}/verify-pin; required only when the declared profile is locked */ + "X-Profile-Token"?: string; + }; + path: { + provider: string; + }; + cookie?: never; + }; + requestBody?: { + content: { + "application/json": components["schemas"]["NetworkAccessCommand"]; + }; + }; + responses: { + /** @description Accepted */ + 202: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["NetworkAccessStatus"]; + }; + }; + /** @description Bad Request */ + 400: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + /** @description Unauthorized */ + 401: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + /** @description Forbidden */ + 403: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + /** @description Not Found */ + 404: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + /** @description Not Acceptable */ + 406: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + /** @description Request Timeout */ + 408: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + /** @description Request Entity Too Large */ + 413: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + /** @description Unsupported Media Type */ + 415: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + /** @description Unprocessable Entity */ + 422: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + /** @description Too Many Requests */ + 429: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + /** @description Internal Server Error */ + 500: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + /** @description Service Unavailable */ + 503: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + }; + }; + disconnectNetworkAccess: { + parameters: { + query?: never; + header?: { + /** @description Optional. When present, it must name the authenticated account's primary profile; an absent header is accepted. */ + "X-Profile-Id"?: string; + /** @description Verification proof for a PIN-locked profile, issued by POST /api/v2/profiles/{id}/verify-pin; required only when the declared profile is locked */ + "X-Profile-Token"?: string; + }; + path: { + provider: string; + }; + cookie?: never; + }; + requestBody?: { + content: { + "application/json": components["schemas"]["NetworkAccessCommand"]; + }; + }; + responses: { + /** @description Accepted */ + 202: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["NetworkAccessStatus"]; + }; + }; + /** @description Bad Request */ + 400: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + /** @description Unauthorized */ + 401: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + /** @description Forbidden */ + 403: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + /** @description Not Found */ + 404: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + /** @description Not Acceptable */ + 406: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + /** @description Request Timeout */ + 408: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + /** @description Request Entity Too Large */ + 413: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + /** @description Unsupported Media Type */ + 415: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + /** @description Unprocessable Entity */ + 422: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + /** @description Too Many Requests */ + 429: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + /** @description Internal Server Error */ + 500: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + /** @description Service Unavailable */ + 503: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + }; + }; + getAdminNetworkAccessStatus: { + parameters: { + query?: never; + header?: { + /** @description Optional. When present, it must name the authenticated account's primary profile; an absent header is accepted. */ + "X-Profile-Id"?: string; + /** @description Verification proof for a PIN-locked profile, issued by POST /api/v2/profiles/{id}/verify-pin; required only when the declared profile is locked */ + "X-Profile-Token"?: string; + }; + path: { + provider: string; + }; + cookie?: never; + }; + requestBody?: never; + responses: { + /** @description OK */ + 200: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["NetworkAccessStatus"]; + }; + }; + /** @description Bad Request */ + 400: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + /** @description Unauthorized */ + 401: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + /** @description Forbidden */ + 403: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + /** @description Not Found */ + 404: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + /** @description Not Acceptable */ + 406: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + /** @description Unprocessable Entity */ + 422: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + /** @description Too Many Requests */ + 429: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + /** @description Internal Server Error */ + 500: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + /** @description Service Unavailable */ + 503: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + }; + }; listAdminNodeSessions: { parameters: { query?: { @@ -50625,6 +51212,124 @@ export interface operations { }; }; }; + restartAdminPluginInstallation: { + parameters: { + query?: never; + header?: { + /** @description Optional. When present, it must name the authenticated account's primary profile; an absent header is accepted. */ + "X-Profile-Id"?: string; + /** @description Verification proof for a PIN-locked profile, issued by POST /api/v2/profiles/{id}/verify-pin; required only when the declared profile is locked */ + "X-Profile-Token"?: string; + }; + path: { + /** @description Opaque identifier */ + id: string; + }; + cookie?: never; + }; + requestBody?: never; + responses: { + /** @description OK */ + 200: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["AdminPluginInstallation"]; + }; + }; + /** @description Bad Request */ + 400: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + /** @description Unauthorized */ + 401: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + /** @description Forbidden */ + 403: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + /** @description Not Found */ + 404: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + /** @description Not Acceptable */ + 406: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + /** @description Conflict */ + 409: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + /** @description Unprocessable Entity */ + 422: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + /** @description Too Many Requests */ + 429: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + /** @description Internal Server Error */ + 500: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + /** @description Service Unavailable */ + 503: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + }; + }; updateAdminPluginTaskBinding: { parameters: { query?: never; @@ -90758,6 +91463,116 @@ export interface operations { }; }; }; + getNetworkAccessCapabilities: { + parameters: { + query?: never; + header?: { + /** @description Optional first precondition, evaluated before If-None-Match: a tag that does not match the current representation is 412 precondition_failed. */ + "If-Match"?: string; + "If-None-Match"?: string; + }; + path?: never; + cookie?: never; + }; + requestBody?: never; + responses: { + /** @description OK */ + 200: { + headers: { + "Cache-Control"?: string; + /** @description The strong, opaque validator of the representation; send it back in If-Match on a guarded mutation or If-None-Match on a conditional read. */ + ETag?: string; + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["NetworkAccessCapabilities"]; + }; + }; + /** @description The representation named by If-None-Match is current; no body. */ + 304: { + headers: { + /** @description The strong, opaque validator of the representation; send it back in If-Match on a guarded mutation or If-None-Match on a conditional read. */ + ETag?: string; + [name: string]: unknown; + }; + content?: never; + }; + /** @description Bad Request */ + 400: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + /** @description Unauthorized */ + 401: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + /** @description Not Acceptable */ + 406: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + /** @description Precondition Failed */ + 412: { + headers: { + /** @description The strong, opaque validator of the representation; send it back in If-Match on a guarded mutation or If-None-Match on a conditional read. */ + ETag?: string; + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + /** @description Unprocessable Entity */ + 422: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + /** @description Too Many Requests */ + 429: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + /** @description Internal Server Error */ + 500: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + /** @description Service Unavailable */ + 503: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["Problem"]; + }; + }; + }; + }; listNotifications: { parameters: { query?: { diff --git a/web/src/components/PlaybackRouteBadges.test.tsx b/web/src/components/PlaybackRouteBadges.test.tsx new file mode 100644 index 0000000000..6b560879cf --- /dev/null +++ b/web/src/components/PlaybackRouteBadges.test.tsx @@ -0,0 +1,68 @@ +import { render, screen } from "@testing-library/react"; +import { describe, expect, it, vi } from "vitest"; +import type { AdminSession } from "@/api/types"; +import { PlaybackRouteBadges } from "./PlaybackRouteBadges"; + +vi.mock("@/hooks/queries/admin/networkAccess", () => ({ + useNetworkAccessCapabilities: () => ({ + data: { providers: [{ provider: "tailscale", display_name: "Tailscale" }] }, + }), +})); + +const session = { + reporting_node: "70709751cdba", + routing_workload: "remux", + routing_execution: "api", + routing_egress: "api", +} as AdminSession; + +describe("PlaybackRouteBadges", () => { + it("names the overlay and API server, retaining container identity in the tooltip", () => { + render(); + expect(screen.getByLabelText("Network: Tailscale")).toBeVisible(); + expect(screen.getByText("API server")).toBeVisible(); + expect(screen.queryByText(session.reporting_node)).not.toBeInTheDocument(); + expect(screen.getByTitle(`API server: ${session.reporting_node} (serves media)`)).toBeVisible(); + }); + + it("shows both a transcode executor and its proxy", () => { + render( + , + ); + expect(screen.getByLabelText("Network: Tailscale")).toBeVisible(); + expect(screen.getByLabelText("Transcode: GPU worker")).toBeVisible(); + expect(screen.getByLabelText("Proxy: Edge proxy")).toBeVisible(); + expect(screen.queryByText("API server")).not.toBeInTheDocument(); + }); + + it("distinguishes the default network from missing route telemetry", () => { + const { rerender } = render( + , + ); + expect(screen.getByLabelText("Network: Default")).toHaveAttribute( + "title", + expect.stringContaining("LAN, public URL, or reverse proxy"), + ); + rerender(); + expect(screen.getByLabelText("Network: Unknown")).toBeVisible(); + expect(screen.queryByLabelText("Network: Tailscale")).not.toBeInTheDocument(); + }); + + it("retains the provider identifier when its plugin is no longer installed", () => { + render( + , + ); + expect(screen.getByLabelText("Network: other-provider")).toBeVisible(); + }); +}); diff --git a/web/src/components/PlaybackRouteBadges.tsx b/web/src/components/PlaybackRouteBadges.tsx index 9af79fb362..03fc03fa89 100644 --- a/web/src/components/PlaybackRouteBadges.tsx +++ b/web/src/components/PlaybackRouteBadges.tsx @@ -1,3 +1,4 @@ +import { useNetworkAccessCapabilities } from "@/hooks/queries/admin/networkAccess"; import type { AdminSession } from "@/api/types"; import { getSessionRouteNodes, type ActivityRouteNode } from "@/pages/adminActivityPresentation"; @@ -8,22 +9,67 @@ const BADGE_COLORS: Record = { legacy: "border-primary/10 bg-primary/5 text-primary", }; -/** Compact, role-labeled badges for each participant in a playback route. */ -export function PlaybackRouteBadges({ session }: { session: AdminSession }) { - return getSessionRouteNodes(session).map((node) => ( +function OverlayNetworkBadge({ provider }: { provider: string }) { + const capabilities = useNetworkAccessCapabilities(); + const name = + capabilities.data?.providers.find((entry) => entry.provider === provider)?.display_name || + provider; + return ; +} + +function NetworkBadge({ name, title }: { name: string; title: string }) { + return ( - {node.kind === "transcode" || node.kind === "proxy" ? ( - <> - {node.label} - - - ) : null} - {node.name} + Network + + {name} - )); + ); +} + +/** The selected access network and serving nodes, shared by all activity views. */ +export function PlaybackRouteBadges({ session }: { session: AdminSession }) { + const provider = session.routing_network_provider; + return ( + <> + {provider ? ( + + ) : ( + + )} + {getSessionRouteNodes(session).map((node) => { + const title = + node.kind === "server" + ? `API server${session.reporting_node ? `: ${session.reporting_node}` : ""}${session.routing_egress === "api" ? " (serves media)" : ""}` + : `${node.label}: ${node.name}`; + return ( + + {node.kind === "transcode" || node.kind === "proxy" ? ( + <> + {node.label} + + + ) : null} + {node.kind === "server" ? "API server" : node.name} + + ); + })} + + ); } diff --git a/web/src/hooks/admin/useSettingsOverview.test.ts b/web/src/hooks/admin/useSettingsOverview.test.ts index 8d0ff6be6a..6d30764a2d 100644 --- a/web/src/hooks/admin/useSettingsOverview.test.ts +++ b/web/src/hooks/admin/useSettingsOverview.test.ts @@ -25,7 +25,7 @@ describe("buildSettingsOverview health tiles", () => { const model = buildSettingsOverview({}); expect(model.tiles).toHaveLength(5); - expect(model.cards).toHaveLength(12); + expect(model.cards).toHaveLength(13); expect(tile({}, "storage").stateText).toBe("Not set up"); expect(card({}, "general")).toEqual({ id: "general" }); }); @@ -296,6 +296,7 @@ describe("buildSettingsOverview groups", () => { "ai", "notifications", "compatibility", + "network-access", ]); }); diff --git a/web/src/hooks/admin/useSettingsOverview.ts b/web/src/hooks/admin/useSettingsOverview.ts index 3dc608788a..659626459c 100644 --- a/web/src/hooks/admin/useSettingsOverview.ts +++ b/web/src/hooks/admin/useSettingsOverview.ts @@ -29,6 +29,7 @@ export const ADMIN_SETTINGS_PAGE_IDS = [ "ai", "notifications", "compatibility", + "network-access", ] as const; export type AdminSettingsPageID = (typeof ADMIN_SETTINGS_PAGE_IDS)[number]; diff --git a/web/src/hooks/queries/admin/networkAccess.test.ts b/web/src/hooks/queries/admin/networkAccess.test.ts new file mode 100644 index 0000000000..20129682d6 --- /dev/null +++ b/web/src/hooks/queries/admin/networkAccess.test.ts @@ -0,0 +1,109 @@ +import { createElement, type ReactNode } from "react"; +import { QueryClient, QueryClientProvider } from "@tanstack/react-query"; +import { act, cleanup, renderHook, waitFor } from "@testing-library/react"; +import { afterEach, beforeEach, expect, it, vi } from "vitest"; +import { + captureProfileRequestContext, + setAccessToken, + setProfileId, + setProfileToken, + setRefreshToken, +} from "@/api/client"; +import { adminKeys } from "../keys"; +import { useAdminNetworkAccessStatus, useConnectNetworkAccess } from "./networkAccess"; + +function fixture() { + const client = new QueryClient({ defaultOptions: { queries: { retry: false } } }); + return { + client, + wrapper: ({ children }: { children: ReactNode }) => + createElement(QueryClientProvider, { client }, children), + }; +} + +const status = { + provider: "tailscale", + hosts: [ + { + host: { id: "api", role: "api", name: "API server" }, + state: "awaiting_authorization", + auth_url: "https://login.example.test/enroll/secret", + addresses: [], + }, + ], +}; + +beforeEach(() => { + localStorage.clear(); + sessionStorage.clear(); + setAccessToken("synthetic-admin"); + setRefreshToken(null); + setProfileId("profile-a"); + setProfileToken("pin-a"); +}); + +it("writes command results only to the requesting administrator's cache", async () => { + const fetchMock = vi + .fn() + .mockResolvedValue( + new Response(JSON.stringify(status), { headers: { "Content-Type": "application/json" } }), + ); + vi.stubGlobal("fetch", fetchMock); + const context = fixture(); + const authority = captureProfileRequestContext()!; + const oldKey = [ + ...adminKeys.networkAccessStatus("tailscale"), + authority.serverOrigin, + authority.authContextVersion, + authority.profileId, + authority.profileTokenGeneration, + ]; + const oldStatus = { provider: "tailscale", hosts: [] }; + context.client.setQueryData(oldKey, oldStatus); + setProfileToken("pin-b"); + const { result } = renderHook(useConnectNetworkAccess, context); + act(() => result.current.mutate({ provider: "tailscale", hosts: ["api"] })); + await waitFor(() => expect(result.current.isSuccess).toBe(true)); + expect(context.client.getQueryData(oldKey)).toEqual(oldStatus); + const currentKey = [ + ...oldKey.slice(0, -1), + captureProfileRequestContext()!.profileTokenGeneration, + ]; + expect(context.client.getQueryData(currentKey)).toEqual(status); +}); + +afterEach(() => { + cleanup(); + vi.unstubAllGlobals(); +}); + +it.each([null, "pin-b", "pin-a"])( + "hides cached enrollment URLs after PIN transition %s", + async (pin) => { + const fetchMock = vi + .fn() + .mockResolvedValueOnce( + new Response(JSON.stringify(status), { headers: { "Content-Type": "application/json" } }), + ) + .mockImplementation(() => new Promise(() => {})); + vi.stubGlobal("fetch", fetchMock); + const { client, wrapper } = fixture(); + const { result, rerender } = renderHook(() => useAdminNetworkAccessStatus("tailscale"), { + wrapper, + }); + await waitFor(() => expect(result.current.isSuccess).toBe(true)); + expect(result.current.data?.hosts[0]?.auth_url).toBe(status.hosts[0]?.auth_url); + act(() => setProfileToken(pin)); + rerender(); + expect(result.current.data).toBeUndefined(); + await waitFor(() => expect(fetchMock).toHaveBeenCalledTimes(2)); + expect( + JSON.stringify( + client + .getQueryCache() + .getAll() + .map((query) => query.queryKey), + ), + ).not.toMatch(/pin-a|pin-b/); + }, +); diff --git a/web/src/hooks/queries/admin/networkAccess.ts b/web/src/hooks/queries/admin/networkAccess.ts new file mode 100644 index 0000000000..d1b2e0305f --- /dev/null +++ b/web/src/hooks/queries/admin/networkAccess.ts @@ -0,0 +1,150 @@ +import { useMutation, useQuery, useQueryClient } from "@tanstack/react-query"; +import { toast } from "sonner"; + +import { + captureProfileRequestContext, + isCapturedProfileAuthorityActive, + StaleApiRequestContextError, + type ProfileRequestContextSnapshot, +} from "@/api/client"; +import { v2 } from "@/api/v2/request"; +import type { components } from "@/api/v2/schema"; +import { usePageActivity } from "@/hooks/usePageActivity"; + +import { adminKeys } from "../keys"; + +export type NetworkAccessCapabilities = components["schemas"]["NetworkAccessCapabilities"]; +export type NetworkAccessProviderSummary = components["schemas"]["NetworkAccessProviderSummary"]; +export type NetworkAccessStatus = components["schemas"]["NetworkAccessStatus"]; +export type NetworkAccessHostStatus = components["schemas"]["NetworkAccessHostStatus"]; +export type NetworkAccessHostState = NetworkAccessHostStatus["state"]; + +/** + * Provider status is asked of the plugin on every read, so polling costs a + * plugin RPC per host each time; 15 s follows enrollment closely enough + * (the auth URL appears, then connected) without hammering the plugin. + */ +const STATUS_POLL_INTERVAL = 15_000; + +function networkAccessStatusKey(provider: string, context: ProfileRequestContextSnapshot | null) { + return [ + ...adminKeys.networkAccessStatus(provider), + context?.serverOrigin, + context?.authContextVersion, + context?.profileId, + context?.profileTokenGeneration, + ] as const; +} + +/** The installed overlay-network providers, read from plugin manifests. */ +export function useNetworkAccessCapabilities() { + const profileContext = captureProfileRequestContext(); + return useQuery({ + queryKey: [...adminKeys.networkAccessCapabilities(), profileContext?.authContextVersion], + enabled: profileContext !== null, + queryFn: async (): Promise => { + if (!profileContext || !isCapturedProfileAuthorityActive(profileContext)) + throw new StaleApiRequestContextError(); + const result = await v2("GET /api/v2/network-access/capabilities", { profileContext }); + if (!isCapturedProfileAuthorityActive(profileContext)) + throw new StaleApiRequestContextError(); + return result; + }, + staleTime: 30_000, + }); +} + +/** One provider's state on every host, polled while the page is in front. */ +export function useAdminNetworkAccessStatus(provider: string | null) { + const pageActivity = usePageActivity(); + const profileContext = captureProfileRequestContext(); + return useQuery({ + queryKey: networkAccessStatusKey(provider ?? "", profileContext), + enabled: profileContext !== null && provider !== null, + queryFn: async (): Promise => { + if (!profileContext || !provider || !isCapturedProfileAuthorityActive(profileContext)) + throw new StaleApiRequestContextError(); + const result = await v2("GET /api/v2/admin/network-access/{provider}/status", { + path: { provider }, + profileContext, + }); + if (!isCapturedProfileAuthorityActive(profileContext)) + throw new StaleApiRequestContextError(); + return result; + }, + staleTime: STATUS_POLL_INTERVAL, + refetchInterval: pageActivity.canApplyRealtimeUpdates ? STATUS_POLL_INTERVAL : false, + }); +} + +export interface NetworkAccessCommandRequest { + provider: string; + /** Host ids to act on; omitted means every host. */ + hosts?: string[]; +} +type NetworkAccessCommandIntent = NetworkAccessCommandRequest & { + profileContext: ProfileRequestContextSnapshot | null; +}; + +/** + * Connect and disconnect are acknowledgements: the server answers 202 with + * the state each host reached within its timeout and enrollment may go on in + * the background, so the result is written straight into the status query and + * polling carries it from there. The intent captures its authority at click + * time, sends once without authentication replay, and refuses to run after + * the authority changed. + */ +function useNetworkAccessCommand(kind: "connect" | "disconnect") { + const queryClient = useQueryClient(); + const failed = kind === "connect" ? "Failed to connect" : "Failed to disconnect"; + const mutation = useMutation({ + retry: false, + mutationFn: async (intent: NetworkAccessCommandIntent): Promise => { + if (!intent.profileContext || !isCapturedProfileAuthorityActive(intent.profileContext)) + throw new StaleApiRequestContextError(); + const options = { + path: { provider: intent.provider }, + body: intent.hosts ? { hosts: intent.hosts } : {}, + profileContext: intent.profileContext, + retryAuthentication: false, + }; + const result = + kind === "connect" + ? await v2("POST /api/v2/admin/network-access/{provider}/connect", options) + : await v2("POST /api/v2/admin/network-access/{provider}/disconnect", options); + if (!isCapturedProfileAuthorityActive(intent.profileContext)) + throw new StaleApiRequestContextError(); + return result; + }, + onSuccess: (result, intent) => { + if (!intent.profileContext || !isCapturedProfileAuthorityActive(intent.profileContext)) + return; + const queryKey = networkAccessStatusKey(intent.provider, intent.profileContext); + queryClient.setQueryData(queryKey, result); + void queryClient.invalidateQueries({ + queryKey, + exact: true, + }); + }, + onError: (err, intent) => { + if (!intent.profileContext || !isCapturedProfileAuthorityActive(intent.profileContext)) + return; + toast.error(err instanceof Error ? err.message : failed); + }, + }); + return { + ...mutation, + mutate: (request: NetworkAccessCommandRequest) => + mutation.mutate({ ...request, profileContext: captureProfileRequestContext() }), + mutateAsync: (request: NetworkAccessCommandRequest) => + mutation.mutateAsync({ ...request, profileContext: captureProfileRequestContext() }), + }; +} + +export function useConnectNetworkAccess() { + return useNetworkAccessCommand("connect"); +} + +export function useDisconnectNetworkAccess() { + return useNetworkAccessCommand("disconnect"); +} diff --git a/web/src/hooks/queries/admin/pluginLifecycle.test.ts b/web/src/hooks/queries/admin/pluginLifecycle.test.ts index 6993477fda..74f925b430 100644 --- a/web/src/hooks/queries/admin/pluginLifecycle.test.ts +++ b/web/src/hooks/queries/admin/pluginLifecycle.test.ts @@ -7,8 +7,10 @@ import { useApplyPluginUpdate, useDeletePluginInstallation, useInstallPlugin, + useRestartPluginInstallation, useUpdatePluginInstallation, } from "./plugins"; +import { adminKeys } from "../keys"; const installation = (id: string, extra: Record = {}) => ({ id, @@ -112,6 +114,55 @@ it("assigns update_policy through PUT once and keeps the projected row", async ( expect(result.current.data?.update_policy).toBe("notify"); }); +it("restarts a failed resident and refreshes installation and network status", async () => { + const fetchMock = vi + .fn() + .mockResolvedValue( + json(installation("7", { runtime: { resident: true, state: "running", restart_count: 0 } })), + ); + vi.stubGlobal("fetch", fetchMock); + const context = fixture(); + const invalidate = vi.spyOn(context.client, "invalidateQueries"); + const { result } = renderHook(useRestartPluginInstallation, context); + act(() => result.current.mutate(7)); + await waitFor(() => expect(result.current.isSuccess).toBe(true)); + expect(fetchMock).toHaveBeenCalledOnce(); + const [url, init] = fetchMock.mock.calls[0] as [string, RequestInit]; + expect(String(url)).toBe("/api/v2/admin/plugins/installations/7/restart"); + expect(init.method).toBe("POST"); + expect((init.headers as Record)["X-Profile-Id"]).toBe("profile-a"); + expect(result.current.data?.runtime.state).toBe("running"); + expect(invalidate).toHaveBeenCalledWith({ queryKey: adminKeys.pluginInstallations() }); + expect(invalidate).toHaveBeenCalledWith({ queryKey: adminKeys.networkAccessStatusRoot() }); +}); + +it.each(["401", "network"])("does not replay a restart after %s failure", async (failure) => { + const fetchMock = + failure === "network" + ? vi.fn().mockRejectedValue(new Error("connection lost")) + : vi.fn().mockResolvedValue(problem("authentication_required", 401)); + vi.stubGlobal("fetch", fetchMock); + const { result } = renderHook(useRestartPluginInstallation, fixture()); + act(() => result.current.mutate(7)); + await waitFor(() => expect(result.current.isError).toBe(true)); + expect(fetchMock).toHaveBeenCalledOnce(); +}); + +it("does not restart after an offline request's administrator authority changes", async () => { + onlineManager.setOnline(false); + const fetchMock = vi.fn(); + vi.stubGlobal("fetch", fetchMock); + const { result } = renderHook(useRestartPluginInstallation, fixture()); + act(() => result.current.mutate(7)); + await waitFor(() => expect(result.current.isPaused).toBe(true)); + act(() => { + setProfileToken("pin-b"); + onlineManager.setOnline(true); + }); + await waitFor(() => expect(result.current.isError).toBe(true)); + expect(fetchMock).not.toHaveBeenCalled(); +}); + it.each(["401", "network"])( "never replays an uncertain %s apply-update under global retry3", async (failure) => { diff --git a/web/src/hooks/queries/admin/plugins.ts b/web/src/hooks/queries/admin/plugins.ts index 3fbb8a819a..c874d6ac11 100644 --- a/web/src/hooks/queries/admin/plugins.ts +++ b/web/src/hooks/queries/admin/plugins.ts @@ -734,6 +734,50 @@ export function useApplyPluginUpdate() { }; } +export function useRestartPluginInstallation() { + const queryClient = useQueryClient(); + const mutation = useMutation({ + retry: false, + mutationFn: async ({ id, profileContext }: PluginLifecycleIntent) => { + if (!isCapturedProfileAuthorityActive(profileContext)) + throw new StaleApiRequestContextError(); + const row = await v2("POST /api/v2/admin/plugins/installations/{id}/restart", { + path: { id: String(id) }, + profileContext, + retryAuthentication: false, + }); + if (!isCapturedProfileAuthorityActive(profileContext)) + throw new StaleApiRequestContextError(); + return pluginInstallationOfV2(row); + }, + onSuccess: (_result, intent) => { + if (!isCapturedProfileAuthorityActive(intent.profileContext)) return; + toast.success("Plugin restart requested"); + invalidatePluginQueries(queryClient); + void queryClient.invalidateQueries({ queryKey: adminKeys.networkAccessStatusRoot() }); + }, + onError: (error, intent) => { + if (!isCapturedProfileAuthorityActive(intent.profileContext)) return; + toast.error( + lifecycleFailure( + error, + "Plugin restart could not be confirmed. Refresh installations before trying again.", + ), + ); + }, + }); + return { + ...mutation, + mutate: (id: number) => { + try { + mutation.mutate(captureInstallation(id)); + } catch { + toast.error("Select an administrator profile before restarting a plugin."); + } + }, + }; +} + export function useDeletePluginInstallation() { const queryClient = useQueryClient(); const mutation = useMutation({ diff --git a/web/src/hooks/queries/keys.ts b/web/src/hooks/queries/keys.ts index 8a8e7b4d6e..89edb6cea2 100644 --- a/web/src/hooks/queries/keys.ts +++ b/web/src/hooks/queries/keys.ts @@ -391,6 +391,10 @@ export const adminKeys = { restartKeys: () => ["admin", "restartKeys"] as const, catalogSearchStatus: () => ["admin", "catalogSearchStatus"] as const, jellyfinCompatStatus: () => ["admin", "jellyfinCompatStatus"] as const, + networkAccessCapabilities: () => ["admin", "networkAccess", "capabilities"] as const, + networkAccessStatusRoot: () => ["admin", "networkAccess", "status"] as const, + networkAccessStatus: (provider: string) => + ["admin", "networkAccess", "status", provider] as const, requestsRoot: () => ["admin", "requests"] as const, requests: (params: Record) => ["admin", "requests", params] as const, requestSettings: () => ["admin", "requests", "settings"] as const, diff --git a/web/src/lib/adminSettingsSearch.ts b/web/src/lib/adminSettingsSearch.ts index b5a11ecda2..2141afe7c9 100644 --- a/web/src/lib/adminSettingsSearch.ts +++ b/web/src/lib/adminSettingsSearch.ts @@ -4,6 +4,7 @@ import { Database, Download, Library, + Network, Paintbrush, PlayCircle, Plug, @@ -538,6 +539,26 @@ export const ADMIN_SETTINGS_GROUPS: AdminSettingsSearchGroup[] = [ ), icon: Plug, }, + { + id: "network-access", + label: "Network Access", + description: + "Overlay-network providers such as Tailscale that reach this server without port forwarding.", + groups: ["Providers"], + keywords: [ + "tailscale", + "netbird", + "overlay", + "tailnet", + "vpn", + "remote access", + "port forwarding", + "network access provider", + "plugin", + ], + settings: settingIndex("Providers", "Connect", "Disconnect", "Authorization"), + icon: Network, + }, ], }, ]; diff --git a/web/src/lib/pluginStatusIndicator.ts b/web/src/lib/pluginStatusIndicator.ts new file mode 100644 index 0000000000..40c0b445fb --- /dev/null +++ b/web/src/lib/pluginStatusIndicator.ts @@ -0,0 +1,35 @@ +import type { PluginInstallation } from "@/api/types"; + +/** + * Status dot for an installation. A resident plugin (one the server keeps + * running) reports its supervisor state; everything else reads enabled. + */ +export function pluginStatusIndicator(installation: PluginInstallation): { + dotClass: string; + label: string; + title?: string; +} { + if (!installation.enabled) { + return { dotClass: "bg-muted-foreground", label: "Inactive" }; + } + const runtime = installation.runtime; + if (!runtime.resident) { + return { dotClass: "bg-success", label: "Active" }; + } + switch (runtime.state) { + case "running": + return { dotClass: "bg-success", label: "Running" }; + case "starting": + return { dotClass: "bg-warning", label: "Starting" }; + case "backoff": + return { + dotClass: "bg-warning", + label: `Restarting (${runtime.restart_count})`, + title: runtime.last_error, + }; + case "failed": + return { dotClass: "bg-destructive", label: "Failed", title: runtime.last_error }; + default: + return { dotClass: "bg-muted-foreground", label: "Stopped", title: runtime.last_error }; + } +} diff --git a/web/src/pages/AdminPlugins.test.tsx b/web/src/pages/AdminPlugins.test.tsx index 843aec61f1..a20447c742 100644 --- a/web/src/pages/AdminPlugins.test.tsx +++ b/web/src/pages/AdminPlugins.test.tsx @@ -5,11 +5,15 @@ import { beforeEach, describe, expect, it, vi } from "vitest"; import type { PluginCatalogEntry, PluginInstallation } from "@/api/types"; +import { pluginStatusIndicator } from "@/lib/pluginStatusIndicator"; + import AdminPlugins from "./AdminPlugins"; const useAdminPluginsMock = vi.fn(); const checkPluginUpdatesMutateMock = vi.fn(); const updatePluginCatalogSettingsMutateMock = vi.fn(); +const restartPluginInstallationMutateMock = vi.fn(); +let restartPluginInstallationPending = false; const capturedButtonProps: Array> = []; const capturedSwitchProps: Array> = []; @@ -57,6 +61,7 @@ function makeInstallation(index: number, displayName: string): PluginInstallatio version: "1.0.0", install_path: `/plugins/installed-${suffix}`, enabled: true, + runtime: { resident: false, state: "stopped", restart_count: 0 }, source_kind: "silo", repository_name: "Silo plugins", updates_paused: false, @@ -131,6 +136,10 @@ vi.mock("@/hooks/queries/admin/plugins", () => ({ usePluginUpload: () => ({ upload: vi.fn(), progress: null, isPending: false }), useUpdatePluginInstallation: () => ({ mutate: vi.fn(), isPending: false }), useApplyPluginUpdate: () => ({ mutate: vi.fn(), isPending: false }), + useRestartPluginInstallation: () => ({ + mutate: restartPluginInstallationMutateMock, + isPending: restartPluginInstallationPending, + }), useDeletePluginInstallation: () => ({ mutate: vi.fn(), isPending: false }), useSavePluginConfig: () => ({ mutate: vi.fn(), isPending: false }), useTestPluginConfig: () => ({ mutate: vi.fn(), isPending: false }), @@ -148,6 +157,8 @@ describe("AdminPlugins", () => { capturedSwitchProps.length = 0; checkPluginUpdatesMutateMock.mockReset(); updatePluginCatalogSettingsMutateMock.mockReset(); + restartPluginInstallationMutateMock.mockReset(); + restartPluginInstallationPending = false; useAdminPluginsMock.mockReturnValue({ repositories: [], catalog: [], @@ -157,6 +168,98 @@ describe("AdminPlugins", () => { }); }); + it("reads runtime.state for the status dot only when the plugin is resident", () => { + const base = makeInstallation(1, "Resident"); + expect(pluginStatusIndicator(base)).toMatchObject({ dotClass: "bg-success", label: "Active" }); + expect( + pluginStatusIndicator({ + ...base, + runtime: { resident: false, state: "stopped", restart_count: 0 }, + }), + ).toMatchObject({ dotClass: "bg-success", label: "Active" }); + expect( + pluginStatusIndicator({ + ...base, + runtime: { resident: true, state: "running", restart_count: 0 }, + }), + ).toMatchObject({ dotClass: "bg-success", label: "Running" }); + expect( + pluginStatusIndicator({ + ...base, + runtime: { resident: true, state: "backoff", restart_count: 3, last_error: "exited" }, + }), + ).toMatchObject({ dotClass: "bg-warning", label: "Restarting (3)", title: "exited" }); + expect( + pluginStatusIndicator({ + ...base, + runtime: { resident: true, state: "failed", restart_count: 9, last_error: "boom" }, + }), + ).toMatchObject({ dotClass: "bg-destructive", label: "Failed", title: "boom" }); + expect( + pluginStatusIndicator({ + ...base, + enabled: false, + runtime: { resident: true, state: "failed", restart_count: 9 }, + }), + ).toMatchObject({ dotClass: "bg-muted-foreground", label: "Inactive" }); + }); + + it("renders the resident runtime state on the installed card", () => { + useAdminPluginsMock.mockReturnValue({ + repositories: [], + catalog: [], + installations: [ + { + ...makeInstallation(1, "Overlay"), + runtime: { resident: true, state: "failed", restart_count: 10, last_error: "exited" }, + }, + ], + catalogSettings: undefined, + isLoading: false, + }); + const markup = renderToStaticMarkup( + + + , + ); + expect(markup).toContain("bg-destructive"); + expect(markup).toContain("Failed"); + expect(markup).not.toContain(">Active<"); + const restart = capturedButtonProps.find((props) => props["aria-label"] === "Restart Overlay"); + expect(restart).toBeDefined(); + expect(restart?.disabled).toBe(false); + (restart?.onClick as () => void)(); + expect(restartPluginInstallationMutateMock).toHaveBeenCalledWith(1); + }); + + it.each([ + { enabled: false, resident: true, pending: false, visible: false }, + { enabled: true, resident: false, pending: false, visible: false }, + { enabled: true, resident: true, pending: true, visible: true }, + ])("only offers restart to enabled residents and waits for the response: %j", (testCase) => { + restartPluginInstallationPending = testCase.pending; + useAdminPluginsMock.mockReturnValue({ + repositories: [], + catalog: [], + installations: [ + { + ...makeInstallation(1, "Overlay"), + enabled: testCase.enabled, + runtime: { resident: testCase.resident, state: "failed", restart_count: 10 }, + }, + ], + isLoading: false, + }); + renderToStaticMarkup( + + + , + ); + const restart = capturedButtonProps.find((props) => props["aria-label"] === "Restart Overlay"); + expect(Boolean(restart)).toBe(testCase.visible); + if (restart) expect(restart.disabled).toBe(testCase.pending); + }); + it("starts the shared plugin update check task from the plugins page", () => { renderToStaticMarkup( @@ -340,6 +443,7 @@ describe("AdminPlugins", () => { version: "0.9.0", install_path: "/plugins/example", enabled: true, + runtime: { resident: false, state: "stopped", restart_count: 0 }, source_kind: "silo", repository_name: "Silo plugins", updates_paused: false, diff --git a/web/src/pages/AdminPlugins.tsx b/web/src/pages/AdminPlugins.tsx index 50e31fad36..fde6ba5b4a 100644 --- a/web/src/pages/AdminPlugins.tsx +++ b/web/src/pages/AdminPlugins.tsx @@ -9,6 +9,7 @@ import { Loader2, Package, Plus, + RotateCw, Search, Settings2, Shield, @@ -72,6 +73,7 @@ import { useDeletePluginRepository, useInstallPlugin, usePluginUpload, + useRestartPluginInstallation, useSavePluginAuthBinding, useSavePluginConfig, useSavePluginTaskBinding, @@ -83,6 +85,7 @@ import { import { useTask } from "@/hooks/queries/admin/tasks"; import { adminKeys } from "@/hooks/queries/keys"; import { pluginRouteHref } from "@/lib/pluginRouteHref"; +import { pluginStatusIndicator } from "@/lib/pluginStatusIndicator"; import { navigateToPluginRoute } from "@/lib/buildPluginHref"; const INSTALLED_PAGE_SIZE = 10; @@ -283,6 +286,7 @@ function InstalledPluginCard({ const updateInstallation = useUpdatePluginInstallation(); const deleteInstallation = useDeletePluginInstallation(); const applyUpdate = useApplyPluginUpdate(); + const restartInstallation = useRestartPluginInstallation(); const [confirmDelete, setConfirmDelete] = useState(false); const capabilities = installation.capabilities ?? []; const presentation = installation.presentation ?? catalogEntry?.presentation; @@ -291,6 +295,7 @@ function InstalledPluginCard({ const adminRoutes = routes.filter( (route) => route.navigable && route.navigation_kind === "admin", ); + const status = pluginStatusIndicator(installation); return ( <> @@ -352,12 +357,10 @@ function InstalledPluginCard({ )} )} - - + + - {installation.enabled ? "Active" : "Inactive"} + {status.label} @@ -385,6 +388,24 @@ function InstalledPluginCard({ {/* Right: actions */}
+ {installation.runtime.resident && installation.enabled ? ( + + ) : null} {adminRoutes.length > 0 ? ( <> {adminRoutes.map((route) => { diff --git a/web/src/pages/admin-settings/AdminSettingsLayout.tsx b/web/src/pages/admin-settings/AdminSettingsLayout.tsx index 7e27da0ceb..be5130dd51 100644 --- a/web/src/pages/admin-settings/AdminSettingsLayout.tsx +++ b/web/src/pages/admin-settings/AdminSettingsLayout.tsx @@ -28,6 +28,7 @@ import WatchSyncSettings from "./WatchSyncSettings"; import AISettings from "./AISettings"; import NotificationsAdminSettings from "./NotificationsAdminSettings"; import CompatibilityProxiesSettings from "./CompatibilityProxiesSettings"; +import NetworkAccessSettings from "./NetworkAccessSettings"; import InfrastructureSettings from "./InfrastructureSettings"; import SettingsOverview from "./SettingsOverview"; import "@/styles/admin-settings.css"; @@ -48,6 +49,7 @@ const SETTINGS_COMPONENTS: Record = { ai: AISettings, notifications: NotificationsAdminSettings, compatibility: CompatibilityProxiesSettings, + "network-access": NetworkAccessSettings, infrastructure: InfrastructureSettings, }; diff --git a/web/src/pages/admin-settings/NetworkAccessSettings.test.tsx b/web/src/pages/admin-settings/NetworkAccessSettings.test.tsx new file mode 100644 index 0000000000..694fb2c7e1 --- /dev/null +++ b/web/src/pages/admin-settings/NetworkAccessSettings.test.tsx @@ -0,0 +1,283 @@ +import { QueryClient, QueryClientProvider, useMutation } from "@tanstack/react-query"; +import { act, render, screen, waitFor, within } from "@testing-library/react"; +import userEvent from "@testing-library/user-event"; +import { beforeEach, describe, expect, it, vi } from "vitest"; + +import type { + NetworkAccessCapabilities, + NetworkAccessCommandRequest, + NetworkAccessStatus, +} from "@/hooks/queries/admin/networkAccess"; + +import NetworkAccessSettings from "./NetworkAccessSettings"; + +const mocks = vi.hoisted(() => ({ + capabilities: { data: undefined as NetworkAccessCapabilities | undefined, isLoading: false }, + status: { + data: undefined as NetworkAccessStatus | undefined, + isLoading: false, + isError: false, + error: null as unknown, + }, + statusCalls: [] as Array, + connect: vi.fn(), + disconnect: vi.fn(), +})); + +vi.mock("@/hooks/queries/admin/networkAccess", () => ({ + useNetworkAccessCapabilities: () => mocks.capabilities, + useAdminNetworkAccessStatus: (provider: string | null) => { + mocks.statusCalls.push(provider); + return mocks.status; + }, + useConnectNetworkAccess: () => + useMutation({ + mutationFn: async (request: NetworkAccessCommandRequest) => mocks.connect(request), + }), + useDisconnectNetworkAccess: () => + useMutation({ + mutationFn: async (request: NetworkAccessCommandRequest) => mocks.disconnect(request), + }), +})); + +function renderPage() { + const client = new QueryClient(); + return render( + + + , + ); +} + +function capabilities( + providers: NetworkAccessCapabilities["providers"] = [], +): NetworkAccessCapabilities { + return { + revision: "r1", + state: providers.length ? "available" : "not_configured", + allowed: providers.length > 0, + providers, + }; +} + +const tailscale = { provider: "tailscale", display_name: "Tailscale", installation_id: "7" }; + +function host( + overrides: Partial = {}, +): NetworkAccessStatus["hosts"][number] { + return { + host: { id: "api", role: "api", name: "Living Room" }, + state: "disconnected", + addresses: [], + ...overrides, + }; +} + +describe("NetworkAccessSettings", () => { + beforeEach(() => { + mocks.capabilities = { data: capabilities(), isLoading: false }; + mocks.status = { data: undefined, isLoading: false, isError: false, error: null }; + mocks.statusCalls = []; + mocks.connect.mockReset(); + mocks.disconnect.mockReset(); + }); + + it("heads the page and explains that providers are plugins when none is installed", () => { + renderPage(); + + expect(screen.getByRole("heading", { level: 1, name: "Network Access" })).toBeInTheDocument(); + expect(screen.getByRole("group", { name: "No providers installed" })).toBeInTheDocument(); + expect( + screen.getByText(/Install a network access provider from the Plugins page/), + ).toBeVisible(); + expect(mocks.statusCalls).toHaveLength(0); + }); + + it("lists each provider with a row per host, its state and origin", () => { + mocks.capabilities = { data: capabilities([tailscale]), isLoading: false }; + mocks.status = { + data: { + provider: "tailscale", + hosts: [ + host({ + state: "connected", + hostname: "silo.tail1234.ts.net", + origin: "https://silo.tail1234.ts.net", + addresses: ["100.64.0.7"], + provider_version: "tsnet 1.102.4", + updated_at: "2026-09-14T09:00:00.000Z", + }), + ], + }, + isLoading: false, + isError: false, + error: null, + }; + + renderPage(); + + expect(mocks.statusCalls).toEqual(["tailscale"]); + const group = screen.getByRole("group", { name: "Tailscale" }); + const row = within(group).getByTestId("network-access-host-tailscale-api"); + expect(row).toHaveAttribute("data-state", "connected"); + expect(within(row).getByText("Living Room · API server")).toBeInTheDocument(); + expect(within(row).getByText("Connected")).toBeInTheDocument(); + expect(within(row).getByRole("link", { name: "https://silo.tail1234.ts.net" })).toHaveAttribute( + "href", + "https://silo.tail1234.ts.net", + ); + expect(within(row).getByText("100.64.0.7")).toBeInTheDocument(); + expect(within(row).getByText("tsnet 1.102.4")).toBeInTheDocument(); + expect(within(row).getByRole("button", { name: "Disconnect" })).toBeEnabled(); + expect(within(row).queryByRole("button", { name: "Connect" })).not.toBeInTheDocument(); + // Internal enum names never reach the admin. + expect(group).not.toHaveTextContent("awaiting_authorization"); + }); + + it("links the authorization page while a host waits for enrollment", () => { + mocks.capabilities = { data: capabilities([tailscale]), isLoading: false }; + mocks.status = { + data: { + provider: "tailscale", + hosts: [ + host({ state: "awaiting_authorization", auth_url: "https://login.example.test/a/abc" }), + ], + }, + isLoading: false, + isError: false, + error: null, + }; + + renderPage(); + + const row = screen.getByTestId("network-access-host-tailscale-api"); + expect(within(row).getByText("Waiting for authorization")).toBeInTheDocument(); + const link = within(row).getByRole("link", { name: /Open the authorization page/ }); + expect(link).toHaveAttribute("href", "https://login.example.test/a/abc"); + expect(link).toHaveAttribute("target", "_blank"); + // Enrollment can be abandoned from here. + expect(within(row).getByRole("button", { name: "Disconnect" })).toBeEnabled(); + }); + + it("sends connect and disconnect for the row's host only", async () => { + const user = userEvent.setup(); + mocks.capabilities = { data: capabilities([tailscale]), isLoading: false }; + mocks.status = { + data: { provider: "tailscale", hosts: [host()] }, + isLoading: false, + isError: false, + error: null, + }; + + renderPage(); + + await user.click(screen.getByRole("button", { name: "Connect" })); + expect(mocks.connect).toHaveBeenCalledWith({ provider: "tailscale", hosts: ["api"] }); + expect(mocks.disconnect).not.toHaveBeenCalled(); + + mocks.status = { + data: { provider: "tailscale", hosts: [host({ state: "connected" })] }, + isLoading: false, + isError: false, + error: null, + }; + renderPage(); + await user.click(screen.getByRole("button", { name: "Disconnect" })); + expect(mocks.disconnect).toHaveBeenCalledWith({ provider: "tailscale", hosts: ["api"] }); + }); + + it.each(["Connect", "Disconnect"] as const)( + "tracks pending %s requests independently for each host", + async (action) => { + const user = userEvent.setup(); + let resolveFirst!: () => void; + let resolveSecond!: () => void; + const first = new Promise((resolve) => { + resolveFirst = resolve; + }); + const second = new Promise((resolve) => { + resolveSecond = resolve; + }); + const command = action === "Connect" ? mocks.connect : mocks.disconnect; + command.mockReturnValueOnce(first).mockReturnValueOnce(second); + mocks.capabilities = { data: capabilities([tailscale]), isLoading: false }; + mocks.status.data = { + provider: "tailscale", + hosts: ["api", "node:10", "node:12"].map((id) => + host({ + host: { id, role: id === "api" ? "api" : "proxy", name: id }, + state: action === "Connect" ? "disconnected" : "connected", + }), + ), + }; + renderPage(); + const buttonFor = (id: string) => + within(screen.getByTestId(`network-access-host-tailscale-${id}`)).getByRole("button", { + name: action, + }); + const apiButton = buttonFor("api"); + const firstProxyButton = buttonFor("node:10"); + const secondProxyButton = buttonFor("node:12"); + + await user.click(apiButton); + await waitFor(() => expect(apiButton).toBeDisabled()); + expect(apiButton.querySelector(".animate-spin")).not.toBeNull(); + expect(firstProxyButton).toBeEnabled(); + expect(firstProxyButton.querySelector(".animate-spin")).toBeNull(); + expect(secondProxyButton).toBeEnabled(); + expect(secondProxyButton.querySelector(".animate-spin")).toBeNull(); + + await user.click(firstProxyButton); + await waitFor(() => expect(firstProxyButton).toBeDisabled()); + expect(apiButton).toBeDisabled(); + expect(secondProxyButton).toBeEnabled(); + expect(command).toHaveBeenNthCalledWith(1, { provider: "tailscale", hosts: ["api"] }); + expect(command).toHaveBeenNthCalledWith(2, { provider: "tailscale", hosts: ["node:10"] }); + + await act(async () => resolveFirst()); + await waitFor(() => expect(apiButton).toBeEnabled()); + expect(firstProxyButton).toBeDisabled(); + expect(secondProxyButton).toBeEnabled(); + await act(async () => resolveSecond()); + await waitFor(() => expect(firstProxyButton).toBeEnabled()); + }, + ); + + it("explains a host whose plugin is not running and keeps Connect disabled there", () => { + mocks.capabilities = { data: capabilities([tailscale]), isLoading: false }; + mocks.status = { + data: { + provider: "tailscale", + hosts: [ + host({ state: "unavailable", error: "plugin process is backoff: plugin process exited" }), + ], + }, + isLoading: false, + isError: false, + error: null, + }; + + renderPage(); + + const row = screen.getByTestId("network-access-host-tailscale-api"); + expect(within(row).getByText("Plugin not running")).toBeInTheDocument(); + expect(within(row).getByRole("status")).toHaveTextContent("plugin process exited"); + expect(within(row).getByRole("button", { name: "Connect" })).toBeDisabled(); + expect(row).not.toHaveTextContent("unavailable"); + }); + + it("surfaces a status read failure without hiding the provider", () => { + mocks.capabilities = { data: capabilities([tailscale]), isLoading: false }; + mocks.status = { + data: undefined, + isLoading: false, + isError: true, + error: new Error("Network access provider not found."), + }; + + renderPage(); + + expect(screen.getByRole("group", { name: "Tailscale" })).toBeInTheDocument(); + expect(screen.getByRole("alert")).toHaveTextContent("Network access provider not found."); + }); +}); diff --git a/web/src/pages/admin-settings/NetworkAccessSettings.tsx b/web/src/pages/admin-settings/NetworkAccessSettings.tsx new file mode 100644 index 0000000000..7f84868e6e --- /dev/null +++ b/web/src/pages/admin-settings/NetworkAccessSettings.tsx @@ -0,0 +1,240 @@ +import { ExternalLink, Loader2 } from "lucide-react"; + +import { Badge } from "@/components/ui/badge"; +import { Button } from "@/components/ui/button"; +import { Skeleton } from "@/components/ui/skeleton"; +import { SettingsPageHeader } from "@/components/settings/SettingsPageHeader"; +import { + useAdminNetworkAccessStatus, + useConnectNetworkAccess, + useDisconnectNetworkAccess, + useNetworkAccessCapabilities, + type NetworkAccessHostState, + type NetworkAccessHostStatus, + type NetworkAccessProviderSummary, +} from "@/hooks/queries/admin/networkAccess"; +import { formatDateTime } from "@/lib/datetime"; +import { cn } from "@/lib/utils"; + +import { FieldGroup } from "./FieldGroup"; + +// `state` is the provider's own vocabulary plus the host-side `unavailable`; +// admins get plain wording. +const STATE_LABELS: Record = { + disconnected: "Disconnected", + awaiting_authorization: "Waiting for authorization", + connecting: "Connecting", + connected: "Connected", + error: "Error", + unavailable: "Plugin not running", +}; + +const STATE_DOT: Record = { + disconnected: "bg-muted-foreground/40", + awaiting_authorization: "bg-amber-500", + connecting: "bg-amber-500", + connected: "bg-emerald-500", + error: "bg-destructive", + unavailable: "bg-muted-foreground/40", +}; + +function stateLabel(state: string): string { + return STATE_LABELS[state as NetworkAccessHostState] ?? state; +} + +function StatePill({ state }: { state: string }) { + return ( + + + ); +} + +function hostLabel(host: NetworkAccessHostStatus["host"]): string { + const role = host.role === "api" ? "API server" : "Proxy node"; + return host.name ? `${host.name} · ${role}` : role; +} + +/** + * One host's row: who it is, the provider's state there, where clients + * reach it, and the connect/disconnect control. While a host waits for + * enrollment the auth URL is the one thing the admin needs, so it sits + * beside the state rather than behind a disclosure. + */ +function HostRow({ provider, host }: { provider: string; host: NetworkAccessHostStatus }) { + const connect = useConnectNetworkAccess(); + const disconnect = useDisconnectNetworkAccess(); + const busy = connect.isPending || disconnect.isPending; + const unavailable = host.state === "unavailable"; + const connected = host.state === "connected"; + const pending = host.state === "connecting" || host.state === "awaiting_authorization"; + const showDisconnect = connected || pending; + return ( +
  • +
    +
    + {hostLabel(host.host)} + + {host.provider_version ? ( + + {host.provider_version} + + ) : null} +
    + {host.origin ? ( +

    + Clients reach this host at{" "} + + {host.origin} + + {host.hostname && host.hostname !== host.origin ? ` (${host.hostname})` : ""} +

    + ) : host.hostname ? ( +

    {host.hostname}

    + ) : null} + {host.addresses.length > 0 ? ( +

    {host.addresses.join(", ")}

    + ) : null} + {host.state === "awaiting_authorization" && host.auth_url ? ( +

    + + Open the authorization page + + to finish enrolling this host. +

    + ) : null} + {host.error ? ( +

    + {host.error} +

    + ) : null} + {host.updated_at ? ( +

    + Last heard from {formatDateTime(host.updated_at, { seconds: false })} +

    + ) : null} +
    +
    + {showDisconnect ? ( + + ) : ( + + )} +
    +
  • + ); +} + +function ProviderGroup({ provider }: { provider: NetworkAccessProviderSummary }) { + const status = useAdminNetworkAccessStatus(provider.provider); + const hosts = status.data?.hosts ?? []; + + return ( + + {status.isLoading ? ( +
    + + +
    + ) : status.isError ? ( +

    + {status.error instanceof Error ? status.error.message : "Could not read provider status."} +

    + ) : hosts.length === 0 ? ( +

    No host runs this provider yet.

    + ) : ( +
      + {hosts.map((host) => ( + + ))} +
    + )} +
    + ); +} + +/** + * Settings → Network Access: every installed overlay-network provider + * (Tailscale and the like) with its state on each host that runs it. The + * providers themselves are plugins; installing one is done on the Plugins + * page, so this page only reads them and drives connect/disconnect. + */ +export default function NetworkAccessSettings() { + const capabilities = useNetworkAccessCapabilities(); + const providers = capabilities.data?.providers ?? []; + + return ( +
    + +

    + Network access providers give this server an identity on an overlay network such as + Tailscale, so clients can reach it without port forwarding or a public reverse proxy. + Providers are plugins: install one from the Plugins page and it appears here. +

    + {capabilities.isLoading ? ( +
    + + +
    + ) : capabilities.isError ? ( +

    + {capabilities.error instanceof Error + ? capabilities.error.message + : "Could not read network access providers."} +

    + ) : providers.length === 0 ? ( + +

    + No installed plugin provides network access. Install a network access provider from the + Plugins page to connect this server to an overlay network. +

    +
    + ) : ( + providers.map((provider) => ) + )} +
    + ); +} diff --git a/web/src/pages/adminActivityPresentation.test.ts b/web/src/pages/adminActivityPresentation.test.ts index 91ff1d54c9..c61356c40d 100644 --- a/web/src/pages/adminActivityPresentation.test.ts +++ b/web/src/pages/adminActivityPresentation.test.ts @@ -477,6 +477,7 @@ describe("adminActivityPresentation", () => { label: "Transcode", name: "silo-transcode-recovered", }, + { key: "server:local", kind: "server", label: "Server", name: "Local server" }, ]); }); diff --git a/web/src/pages/adminActivityPresentation.ts b/web/src/pages/adminActivityPresentation.ts index cbe3fc4002..4d903fb361 100644 --- a/web/src/pages/adminActivityPresentation.ts +++ b/web/src/pages/adminActivityPresentation.ts @@ -319,7 +319,7 @@ function reportingServerRouteNode(session: AdminSession): ActivityRouteNode { /** * Return registered playback nodes in work-to-viewer order. Routes without a - * registered node retain the reporting API server's identity, while rows from + * registered node identify the API server, while rows from * older servers fall back to their legacy node fields. */ export function getSessionRouteNodes(session: AdminSession): ActivityRouteNode[] { @@ -354,6 +354,9 @@ export function getSessionRouteNodes(session: AdminSession): ActivityRouteNode[] } } + if (egress === "api" || (execution === "api" && nodes.length === 0)) { + nodes.push(reportingServerRouteNode(session)); + } return nodes.length > 0 ? nodes : [reportingServerRouteNode(session)]; }