4cf3471398
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2092 lines
83 KiB
Go
2092 lines
83 KiB
Go
// Package dispatch ties together the Target-side components: it receives
|
||
// ControllerEnvelope messages off the gRPC stream, routes them to the
|
||
// module registry and the active Runtime, and emits AgentEnvelope state
|
||
// updates back to the Controller.
|
||
//
|
||
// Survivability:
|
||
//
|
||
// - On successful Create, dispatch writes a sidecar JSON in metaDir so
|
||
// the agent can re-attach to the running container after a restart.
|
||
// - On Stop, the sidecar is deleted.
|
||
// - On agent startup, Rehydrate() scans sidecars, verifies each
|
||
// container still exists, and rebuilds the in-memory instance map.
|
||
// - On every (re)connect, SetSender + Announce re-plumb the upward
|
||
// pipe and re-send current state so the controller's DB stays honest.
|
||
package dispatch
|
||
|
||
import (
|
||
"context"
|
||
"fmt"
|
||
"log/slog"
|
||
"net"
|
||
"os"
|
||
"path/filepath"
|
||
"strings"
|
||
"sync"
|
||
"sync/atomic"
|
||
"time"
|
||
|
||
"google.golang.org/protobuf/types/known/timestamppb"
|
||
|
||
agentmodule "github.com/dbledeez/panel/agent/internal/module"
|
||
"github.com/dbledeez/panel/agent/internal/runtime"
|
||
"github.com/dbledeez/panel/agent/internal/rcon"
|
||
"github.com/dbledeez/panel/agent/internal/state"
|
||
modulepkg "github.com/dbledeez/panel/pkg/module"
|
||
panelv1 "github.com/dbledeez/panel/proto/panel/v1"
|
||
)
|
||
|
||
// Sender is the subset of the gRPC stream we use to send events upstream.
|
||
type Sender interface {
|
||
Send(*panelv1.AgentEnvelope) error
|
||
}
|
||
|
||
// Dispatcher wires the gRPC stream to the runtime and module registry.
|
||
type Dispatcher struct {
|
||
log *slog.Logger
|
||
modules *modulepkg.Registry
|
||
runtime runtime.Runtime
|
||
dataRoot string
|
||
metaDir string
|
||
backupDir string
|
||
|
||
sendMu sync.Mutex
|
||
send Sender
|
||
// sendCh decouples envelope producers (handler goroutines, trackers,
|
||
// log streams) from the actual gRPC Send. A single drainer goroutine
|
||
// owns the stream's Send so concurrent producers never race on it
|
||
// (grpc ClientStream.Send is NOT safe for concurrent use) and never
|
||
// block the dispatch path when the controller is slow to drain. High-
|
||
// frequency telemetry (logs/stats/app-state/player/heartbeat) is
|
||
// DROPPED when the buffer is full rather than blocking — losing a log
|
||
// line is fine (the controller's durable state comes from Announce +
|
||
// InstanceState, not replayed history); freezing the agent is not.
|
||
// This is what stopped a chatty game (VEIN) from wedging the whole
|
||
// agent fleet-wide via a backpressured synchronous Send.
|
||
sendCh chan *panelv1.AgentEnvelope
|
||
sendClosed chan struct{}
|
||
|
||
mu sync.Mutex
|
||
instances map[string]*instanceRecord
|
||
// fsHelpers holds one long-lived alpine sidecar per stopped instance
|
||
// whose Files tab has been opened. Lets operators browse/edit files
|
||
// on a stopped container. Reaped after ~5 min idle; torn down when
|
||
// the main instance starts (frees up volumes for the game).
|
||
fsHelpers map[string]*fsHelper
|
||
// updaters tracks the active updater goroutine per instance so a
|
||
// stop / delete can cancel an in-flight steamcmd download. Without
|
||
// this the sidecar runs to completion (or its 30-min timeout)
|
||
// holding the volume mount, blocking the main delete from purging
|
||
// — which the operator sees as "stuck downloading" with the parent
|
||
// already gone. Each entry holds a unique pointer so the
|
||
// unregister path can compare against the slot it installed.
|
||
updaters map[string]*updaterSlot
|
||
|
||
// uploads holds in-progress chunked uploads keyed by upload_id.
|
||
// Each entry owns a temp file the chunk handler appends to; on the
|
||
// final chunk the file is shipped to the container, closed and
|
||
// unlinked. See files_chunked.go.
|
||
uploads map[string]*uploadSession
|
||
}
|
||
|
||
// updaterSlot pairs the cancel func with a unique pointer identity so
|
||
// concurrent unregisters from racing updater goroutines don't clobber
|
||
// each other. Comparing pointers is cheap and correct.
|
||
type updaterSlot struct {
|
||
cancel context.CancelFunc
|
||
}
|
||
|
||
// fsHelper tracks one file-ops sidecar: alpine mounting the instance's
|
||
// volumes read/write, running `sleep infinity`.
|
||
type fsHelper struct {
|
||
containerID string
|
||
lastAccess time.Time
|
||
}
|
||
|
||
type instanceRecord struct {
|
||
InstanceID string
|
||
ContainerID string
|
||
ModuleID string
|
||
// Branch is the normalized Steam branch this instance was installed from
|
||
// (e.g. "v2.6", "latest_experimental", "public"), derived at Create from
|
||
// the chosen update_provider. Used by warm-seed so a new instance only
|
||
// reuses an install from a sibling on the SAME branch — copying a 3.0
|
||
// (experimental) install into a server meant to run 2.6 would corrupt it.
|
||
// Empty on instances created before this feature (treated as the module
|
||
// default branch by branchOrDefault).
|
||
Branch string
|
||
DataPath string
|
||
// BrowseableRoot is the default (first) file-manager root —
|
||
// container-absolute. BrowseableRoots is the full list; when the
|
||
// module exposes more than one view (7DTD: saves + game files),
|
||
// safeJoinContainer accepts any of them. Always contains at least
|
||
// BrowseableRoot.
|
||
BrowseableRoot string
|
||
BrowseableRoots []modulepkg.BrowseableRoot
|
||
LogCancel context.CancelFunc
|
||
|
||
// ExitWatchCancel single-flights the watchExit goroutine per container
|
||
// lifetime. Before this guard, every activate() call — handleStart,
|
||
// Announce on EVERY controller reconnect, the post-warm-seed autostart —
|
||
// spawned a fresh watchExit, each parked in a ctx-unbounded
|
||
// ContainerWait: leaked goroutines + leaked docker connections (the same
|
||
// failure shape as the ExecCapture leak) + duplicate crashed/exited
|
||
// emissions when the container finally died. Guarded by d.mu like the
|
||
// other cancel fields.
|
||
ExitWatchCancel context.CancelFunc
|
||
|
||
// RCON wire-up, resolved at Create time, consumed at Start time.
|
||
RCONAddr string
|
||
RCONPassword string
|
||
|
||
// State tracker and its cancel, populated on Start.
|
||
Tracker *state.Tracker
|
||
TrackerCancel context.CancelFunc
|
||
TrackerDone chan struct{}
|
||
|
||
// stopping suppresses watchExit's CRASHED emission during graceful stop.
|
||
stopping atomic.Bool
|
||
|
||
// memGuardWarned latches true the first time the 45 GB warn path
|
||
// fires on this container. Both paths (warn + hard) check this to
|
||
// avoid stacking SIGKILL on top of an in-flight graceful save.
|
||
memGuardWarned atomic.Bool
|
||
|
||
// lastStatus tracks the most recent InstanceStatus the agent sent
|
||
// upstream for this instance. Read by EmitAppState to promote a stale
|
||
// CRASHED status back to RUNNING when state-source polls succeed —
|
||
// the docker-restart-policy auto-recovery flip in watchExit doesn't
|
||
// fire when a server segfaults during world-gen but stabilizes on
|
||
// retry, leaving the panel stuck on CRASHED until manual stop+start.
|
||
lastStatus atomic.Int32
|
||
|
||
// hangGuard is the per-instance state for the ark-sa hang-detection
|
||
// guardrail. Embedded by value so its zero value is usable. Methods
|
||
// on Dispatcher (arkHangGuardOn{LogLine,RconResult}, arkHangGuardRestart)
|
||
// own the field; nothing outside the guard touches it. See hangguard.go.
|
||
hangGuard hangGuardState
|
||
|
||
// provisionGuard single-flights the 7dtd cluster auto-provision watcher
|
||
// (one per instance at a time). See decoremap_provision.go.
|
||
provisionGuard atomic.Bool
|
||
}
|
||
|
||
// New constructs a dispatcher. The caller sets the sender via SetSender
|
||
// after the initial handshake (and on each reconnect).
|
||
func New(log *slog.Logger, modules *modulepkg.Registry, rt runtime.Runtime, dataRoot, metaDir, backupDir string) *Dispatcher {
|
||
d := &Dispatcher{
|
||
log: log,
|
||
modules: modules,
|
||
runtime: rt,
|
||
dataRoot: dataRoot,
|
||
metaDir: metaDir,
|
||
backupDir: backupDir,
|
||
instances: map[string]*instanceRecord{},
|
||
fsHelpers: map[string]*fsHelper{},
|
||
updaters: map[string]*updaterSlot{},
|
||
uploads: map[string]*uploadSession{},
|
||
// Buffer sized to absorb a burst of telemetry from a chatty game
|
||
// while the controller catches up. Beyond this, droppable envelopes
|
||
// are discarded (see sendEnv). 4096 is generous for ~dozens of
|
||
// instances; a full buffer means the controller is badly behind and
|
||
// dropping stale logs/stats is the right call.
|
||
sendCh: make(chan *panelv1.AgentEnvelope, 4096),
|
||
}
|
||
go d.fsHelperReaper()
|
||
go d.uploadReaper()
|
||
go d.sendLoop()
|
||
return d
|
||
}
|
||
|
||
// sendLoop is the SOLE owner of stream.Send. It drains sendCh and writes each
|
||
// envelope to the current sender. Running on one goroutine guarantees Sends
|
||
// are serialized (grpc ClientStream.Send is not safe for concurrent use) and
|
||
// that a slow/backpressured controller can never block a producer past the
|
||
// buffer. Envelopes enqueued while disconnected (send == nil) are dropped —
|
||
// the controller rebuilds durable state from Announce on reconnect.
|
||
func (d *Dispatcher) sendLoop() {
|
||
for env := range d.sendCh {
|
||
d.sendMu.Lock()
|
||
s := d.send
|
||
d.sendMu.Unlock()
|
||
if s == nil {
|
||
continue
|
||
}
|
||
_ = s.Send(env)
|
||
}
|
||
}
|
||
|
||
// Modules exposes the registry so main() can summarize it into the
|
||
// AgentHello envelope. Read-only from the caller's perspective.
|
||
func (d *Dispatcher) Modules() *modulepkg.Registry { return d.modules }
|
||
|
||
// SetSender installs (or replaces) the upstream sender after a
|
||
// (re)connect. Safe to call from any goroutine.
|
||
func (d *Dispatcher) SetSender(s Sender) {
|
||
d.sendMu.Lock()
|
||
d.send = s
|
||
d.sendMu.Unlock()
|
||
}
|
||
|
||
// sendEnv enqueues an envelope for the sendLoop drainer instead of calling
|
||
// stream.Send inline. This keeps producers (handler goroutines, trackers, log
|
||
// streams) off the gRPC send path entirely: they never race on Send and never
|
||
// block when the controller is slow to drain.
|
||
//
|
||
// Backpressure policy: high-frequency telemetry (logs / stats / app-state /
|
||
// player / heartbeat) is DROPPED if the buffer is full — losing a log line is
|
||
// harmless, wedging the agent is not. Low-frequency CRITICAL envelopes
|
||
// (command replies, instance-state, FS/backup/etc. results) are never dropped:
|
||
// if the buffer is somehow full they block briefly until space frees. Because
|
||
// critical envelopes are rare, this can't cause the flood-driven wedge that a
|
||
// chatty game's telemetry would.
|
||
//
|
||
// Messages emitted while disconnected (send == nil) are still effectively
|
||
// dropped by sendLoop — intentional, the controller rebuilds durable state
|
||
// from Announce() on reconnect, not replayed history.
|
||
func (d *Dispatcher) sendEnv(env *panelv1.AgentEnvelope) {
|
||
if isDroppableEnvelope(env) {
|
||
select {
|
||
case d.sendCh <- env:
|
||
default:
|
||
// Buffer full — drop this telemetry frame rather than block.
|
||
}
|
||
return
|
||
}
|
||
// Critical envelope: enqueue, blocking only if the buffer is full (rare).
|
||
d.sendCh <- env
|
||
}
|
||
|
||
// SendHeartbeat enqueues a heartbeat envelope on the droppable send queue.
|
||
// Routing heartbeats through sendLoop keeps it the SOLE owner of stream.Send:
|
||
// previously main.go's session ticker called stream.Send directly, racing the
|
||
// drainer goroutine (grpc ClientStream.Send is NOT safe for concurrent use).
|
||
// Heartbeats are droppable telemetry — a skipped beat under backpressure is
|
||
// harmless. Note the tradeoff: a broken stream is now detected by the recv
|
||
// loop (stream.Recv errors when the transport dies), not by a failed
|
||
// heartbeat send.
|
||
func (d *Dispatcher) SendHeartbeat(t time.Time) {
|
||
d.sendEnv(&panelv1.AgentEnvelope{
|
||
SentAt: timestamppb.New(t),
|
||
Payload: &panelv1.AgentEnvelope_Heartbeat{
|
||
Heartbeat: &panelv1.Heartbeat{At: timestamppb.New(t)},
|
||
},
|
||
})
|
||
}
|
||
|
||
// isDroppableEnvelope reports whether an envelope is high-frequency telemetry
|
||
// that's safe to discard under backpressure. Everything not listed here is
|
||
// treated as critical (command replies, instance state, FS/backup results,
|
||
// hello/announce) and is never dropped.
|
||
func isDroppableEnvelope(env *panelv1.AgentEnvelope) bool {
|
||
switch env.GetPayload().(type) {
|
||
case *panelv1.AgentEnvelope_Log,
|
||
*panelv1.AgentEnvelope_InstanceStats,
|
||
*panelv1.AgentEnvelope_AppState,
|
||
*panelv1.AgentEnvelope_Player,
|
||
*panelv1.AgentEnvelope_Heartbeat:
|
||
return true
|
||
default:
|
||
return false
|
||
}
|
||
}
|
||
|
||
// Handle routes a single inbound envelope.
|
||
func (d *Dispatcher) Handle(ctx context.Context, env *panelv1.ControllerEnvelope) {
|
||
switch p := env.Payload.(type) {
|
||
case *panelv1.ControllerEnvelope_Ping:
|
||
d.sendEnv(&panelv1.AgentEnvelope{
|
||
CorrelationId: env.CorrelationId,
|
||
SentAt: timestamppb.Now(),
|
||
Payload: &panelv1.AgentEnvelope_Pong{Pong: &panelv1.Pong{Nonce: p.Ping.Nonce}},
|
||
})
|
||
case *panelv1.ControllerEnvelope_InstanceCreate:
|
||
d.handleCreate(ctx, env.CorrelationId, p.InstanceCreate)
|
||
case *panelv1.ControllerEnvelope_InstanceStart:
|
||
d.handleStart(ctx, env.CorrelationId, p.InstanceStart)
|
||
case *panelv1.ControllerEnvelope_InstanceStop:
|
||
d.handleStop(ctx, env.CorrelationId, p.InstanceStop)
|
||
case *panelv1.ControllerEnvelope_InstanceDelete:
|
||
d.handleDelete(ctx, env.CorrelationId, p.InstanceDelete)
|
||
case *panelv1.ControllerEnvelope_Backup:
|
||
d.handleBackup(ctx, env.CorrelationId, p.Backup)
|
||
case *panelv1.ControllerEnvelope_Restore:
|
||
d.handleRestore(ctx, env.CorrelationId, p.Restore)
|
||
case *panelv1.ControllerEnvelope_BackupListEntries:
|
||
d.handleBackupListEntries(env.CorrelationId, p.BackupListEntries)
|
||
case *panelv1.ControllerEnvelope_BackupReadFile:
|
||
d.handleBackupReadFile(env.CorrelationId, p.BackupReadFile)
|
||
case *panelv1.ControllerEnvelope_Exec:
|
||
// No exec implementation yet — tell the caller instead of logging
|
||
// and silently succeeding (the controller's Await would otherwise
|
||
// hang until timeout with no signal why).
|
||
d.log.Warn("exec requested but not implemented", "instance_id", p.Exec.InstanceId)
|
||
d.replyError(env.CorrelationId, p.Exec.InstanceId, "exec_not_implemented", "exec is not implemented on this agent")
|
||
case *panelv1.ControllerEnvelope_Rcon:
|
||
d.handleRCON(ctx, env.CorrelationId, p.Rcon)
|
||
case *panelv1.ControllerEnvelope_FsList:
|
||
d.handleFsList(env.CorrelationId, p.FsList)
|
||
case *panelv1.ControllerEnvelope_FsRead:
|
||
d.handleFsRead(env.CorrelationId, p.FsRead)
|
||
case *panelv1.ControllerEnvelope_FsWrite:
|
||
d.handleFsWrite(env.CorrelationId, p.FsWrite)
|
||
case *panelv1.ControllerEnvelope_FsDelete:
|
||
d.handleFsDelete(env.CorrelationId, p.FsDelete)
|
||
case *panelv1.ControllerEnvelope_FsSymlink:
|
||
d.handleFsSymlink(env.CorrelationId, p.FsSymlink)
|
||
case *panelv1.ControllerEnvelope_FsExtract:
|
||
d.handleFsExtract(env.CorrelationId, p.FsExtract)
|
||
case *panelv1.ControllerEnvelope_FsCompress:
|
||
d.handleFsCompress(env.CorrelationId, p.FsCompress)
|
||
case *panelv1.ControllerEnvelope_FsRename:
|
||
d.handleFsRename(env.CorrelationId, p.FsRename)
|
||
case *panelv1.ControllerEnvelope_FsWriteChunk:
|
||
d.handleFsWriteChunk(env.CorrelationId, p.FsWriteChunk)
|
||
case *panelv1.ControllerEnvelope_ArkSaveRestore:
|
||
d.handleArkSaveRestore(env.CorrelationId, p.ArkSaveRestore)
|
||
case *panelv1.ControllerEnvelope_DayzModInstall:
|
||
d.handleDayzModInstall(env.CorrelationId, p.DayzModInstall)
|
||
case *panelv1.ControllerEnvelope_DayzModUninstall:
|
||
d.handleDayzModUninstall(env.CorrelationId, p.DayzModUninstall)
|
||
case *panelv1.ControllerEnvelope_Update:
|
||
d.handleUpdate(env.CorrelationId, p.Update)
|
||
case *panelv1.ControllerEnvelope_EmpyrionScenarioInstall:
|
||
d.handleEmpyrionScenarioInstall(env.CorrelationId, p.EmpyrionScenarioInstall)
|
||
case *panelv1.ControllerEnvelope_EmpyrionDiscoveries:
|
||
d.handleEmpyrionDiscoveries(env.CorrelationId, p.EmpyrionDiscoveries)
|
||
case *panelv1.ControllerEnvelope_EmpyrionPlayerSummary:
|
||
d.handleEmpyrionPlayerSummary(env.CorrelationId, p.EmpyrionPlayerSummary)
|
||
case *panelv1.ControllerEnvelope_RegionScan:
|
||
d.handleRegionScan(ctx, env.CorrelationId, p.RegionScan)
|
||
case *panelv1.ControllerEnvelope_RegionHeal:
|
||
d.handleRegionHeal(ctx, env.CorrelationId, p.RegionHeal)
|
||
case *panelv1.ControllerEnvelope_InstanceRenderConfig:
|
||
d.handleRenderConfig(ctx, env.CorrelationId, p.InstanceRenderConfig)
|
||
case *panelv1.ControllerEnvelope_SeedMods:
|
||
d.handleSeedMods(ctx, env.CorrelationId, p.SeedMods)
|
||
case *panelv1.ControllerEnvelope_UpdateCheck:
|
||
d.handleUpdateCheck(env.CorrelationId, p.UpdateCheck)
|
||
default:
|
||
d.log.Debug("unhandled controller message", "type", fmt.Sprintf("%T", env.Payload))
|
||
}
|
||
}
|
||
|
||
// Shutdown winds down agent-internal goroutines on SIGINT/SIGTERM but
|
||
// LEAVES the game containers running. A `systemctl restart panel-agent`
|
||
// shouldn't stop every ARK / 7DTD / V Rising the operator has up — the
|
||
// agent is a control-plane process; the game containers are workloads
|
||
// that should survive control-plane restarts the same way they survive
|
||
// gRPC blips. Earlier behaviour (Stop every instance) caused half a
|
||
// dozen game containers to drop out from under players any time the
|
||
// agent binary was redeployed.
|
||
//
|
||
// What we still tear down: state trackers + log-stream pumps, since
|
||
// those are agent-side goroutines holding open RCON dials and Docker
|
||
// log subscriptions that the next agent process will re-establish on
|
||
// rehydrate. The Docker container itself stays untouched.
|
||
func (d *Dispatcher) Shutdown(_ context.Context) {
|
||
// Snapshot the cancel funcs under d.mu (they're mutated under the same
|
||
// mutex by activate/handleStop), then invoke them outside the lock.
|
||
d.mu.Lock()
|
||
cancels := make([]context.CancelFunc, 0, len(d.instances)*3)
|
||
n := 0
|
||
for _, r := range d.instances {
|
||
n++
|
||
for _, c := range []context.CancelFunc{r.TrackerCancel, r.LogCancel, r.ExitWatchCancel} {
|
||
if c != nil {
|
||
cancels = append(cancels, c)
|
||
}
|
||
}
|
||
r.TrackerCancel = nil
|
||
r.LogCancel = nil
|
||
r.ExitWatchCancel = nil
|
||
}
|
||
d.mu.Unlock()
|
||
for _, c := range cancels {
|
||
c()
|
||
}
|
||
d.log.Info("agent shutdown — game containers left running for next agent rehydrate", "instance_count", n)
|
||
}
|
||
|
||
// Rehydrate loads each sidecar in metaDir, verifies the underlying
|
||
// container still exists, and rebuilds the in-memory instance record.
|
||
// Stale sidecars (container gone) are deleted. Intended for agent cold
|
||
// start. Activation (log stream + tracker) is deferred to Announce so
|
||
// the sender is in place before we start emitting events.
|
||
func (d *Dispatcher) Rehydrate(ctx context.Context) error {
|
||
metas, err := d.loadAllMeta()
|
||
if err != nil {
|
||
return fmt.Errorf("load meta: %w", err)
|
||
}
|
||
haveMeta := make(map[string]bool, len(metas))
|
||
for _, m := range metas {
|
||
haveMeta[m.InstanceID] = true
|
||
}
|
||
for _, m := range metas {
|
||
state, err := d.runtime.InspectByName(ctx, "panel-"+m.InstanceID)
|
||
if err != nil {
|
||
d.log.Warn("rehydrate: container gone, removing sidecar", "instance_id", m.InstanceID, "err", err)
|
||
d.deleteMeta(m.InstanceID)
|
||
continue
|
||
}
|
||
// If the sidecar predates this feature, BrowseableRoot is empty.
|
||
// Fill it from the current manifest so rehydrated instances gain
|
||
// container-path file ops without needing recreate.
|
||
browseable := m.BrowseableRoot
|
||
var roots []modulepkg.BrowseableRoot
|
||
if mf, ok := d.modules.Get(m.ModuleID); ok {
|
||
// Always re-read roots from current manifest — lets operators
|
||
// add new browseable_roots entries and have existing instances
|
||
// pick them up on next agent restart without a container rebuild.
|
||
roots = resolveBrowseableRoots(mf)
|
||
if browseable == "" {
|
||
browseable = resolveBrowseableRoot(mf)
|
||
}
|
||
}
|
||
d.mu.Lock()
|
||
d.instances[m.InstanceID] = &instanceRecord{
|
||
InstanceID: m.InstanceID,
|
||
ContainerID: state.ContainerID,
|
||
ModuleID: m.ModuleID,
|
||
Branch: m.Branch,
|
||
DataPath: m.DataPath,
|
||
BrowseableRoot: browseable,
|
||
BrowseableRoots: roots,
|
||
RCONAddr: m.RCONAddr,
|
||
RCONPassword: m.RCONPassword,
|
||
}
|
||
d.mu.Unlock()
|
||
d.log.Info("rehydrated instance",
|
||
"instance_id", m.InstanceID,
|
||
"module_id", m.ModuleID,
|
||
"container_status", state.Status,
|
||
)
|
||
// Reap orphan sidecar containers from a previous agent crash.
|
||
// steamcmd/backup/wipe sidecars are usually torn down by their
|
||
// goroutine's defer — but if the agent dies mid-run, the defer
|
||
// never fires and the sidecar lingers. It then holds the
|
||
// instance's volumes "in use," blocking future delete-with-purge.
|
||
//
|
||
// We only sweep EXITED containers here; running ones might be
|
||
// legitimately in use (e.g. an fs-helper sidecar an operator
|
||
// kept open). Force-removed exited orphans are safe — they
|
||
// already finished whatever they were doing.
|
||
if dr, ok := d.runtime.(*runtime.DockerRuntime); ok {
|
||
holders, err := dr.ListContainersHoldingInstanceVolumes(ctx, m.InstanceID)
|
||
if err == nil {
|
||
for _, hid := range holders {
|
||
if hid == state.ContainerID {
|
||
continue // the main instance container itself
|
||
}
|
||
// InspectByName accepts an ID too — Docker's ContainerInspect
|
||
// resolves either form. Status tells us if this is an
|
||
// exited orphan vs a still-running fs-helper.
|
||
hstate, ierr := dr.InspectByName(ctx, hid)
|
||
if ierr != nil {
|
||
continue
|
||
}
|
||
if hstate.Status == "exited" || hstate.Status == "dead" {
|
||
if err := dr.ForceRemoveContainer(ctx, hid); err == nil {
|
||
d.log.Info("rehydrate: reaped orphan sidecar", "instance_id", m.InstanceID, "container_id", short(hid))
|
||
}
|
||
}
|
||
}
|
||
}
|
||
}
|
||
}
|
||
|
||
// Orphan detection: a main game-server container (panel.role=instance)
|
||
// that exists on the runtime but has NO sidecar metadata. This is the
|
||
// fingerprint of an interrupted delete — e.g. a delete racing with
|
||
// agent shutdown, where the sidecar got removed but the container
|
||
// remove failed because the Docker connection died mid-call. Such an
|
||
// instance is fully invisible to the agent (every file/RCON/stop op
|
||
// returns "not on this target") even though the controller still
|
||
// routes to it and the world keeps running. Surface it loudly so it
|
||
// can be recovered instead of silently lost.
|
||
if dr, ok := d.runtime.(*runtime.DockerRuntime); ok {
|
||
names, lerr := dr.ListMainInstanceNames(ctx)
|
||
if lerr != nil {
|
||
d.log.Warn("rehydrate: orphan scan failed", "err", lerr)
|
||
} else {
|
||
for _, id := range names {
|
||
if haveMeta[id] {
|
||
continue
|
||
}
|
||
d.mu.Lock()
|
||
_, tracked := d.instances[id]
|
||
d.mu.Unlock()
|
||
if tracked {
|
||
continue
|
||
}
|
||
d.log.Error("rehydrate: ORPHAN container has no sidecar metadata — instance is invisible to the agent (likely an interrupted delete). Recover by recreating its sidecar or re-issuing create/delete from the controller.",
|
||
"instance_id", id, "container", "panel-"+id)
|
||
}
|
||
}
|
||
}
|
||
|
||
return nil
|
||
}
|
||
|
||
// Announce re-sends the current state of every tracked instance to the
|
||
// controller. For instances whose containers are currently running, it
|
||
// also (re)activates log streaming + the state tracker. Called after
|
||
// each (re)connect.
|
||
func (d *Dispatcher) Announce(ctx context.Context) {
|
||
d.mu.Lock()
|
||
recs := make([]*instanceRecord, 0, len(d.instances))
|
||
for _, r := range d.instances {
|
||
recs = append(recs, r)
|
||
}
|
||
d.mu.Unlock()
|
||
|
||
for _, rec := range recs {
|
||
rtState, err := d.runtime.InspectByName(ctx, "panel-"+rec.InstanceID)
|
||
if err != nil {
|
||
d.log.Warn("announce: inspect failed", "instance_id", rec.InstanceID, "err", err)
|
||
continue
|
||
}
|
||
status := rtStatusToProto(rtState.Status)
|
||
d.sendInstanceState(rec.InstanceID, status, rtState.ExitCode, "announced:"+rtState.Status)
|
||
if rtState.Status == "running" {
|
||
// On every reconnect, tear down and rebuild the log + stats
|
||
// streams. The reason: the controller's in-memory log ring buffer
|
||
// is wiped on its restart, so when the user opens the Console tab
|
||
// after a controller restart, history is blank. Docker's
|
||
// ContainerLogs API replays Tail=400 lines on every fresh attach,
|
||
// so rebuilding the streams here repopulates the controller's
|
||
// ring buffer with recent history (including the per-module
|
||
// "ready" log line, which lets the dashboard's auto-scan clear
|
||
// the "starting" overlay without needing the operator to open
|
||
// the console). Old streams keep flowing in the background until
|
||
// their goroutines notice the cancelled context and exit.
|
||
d.mu.Lock()
|
||
prevLogCancel := rec.LogCancel
|
||
rec.LogCancel = nil
|
||
d.mu.Unlock()
|
||
if prevLogCancel != nil {
|
||
prevLogCancel()
|
||
}
|
||
d.activate(rec)
|
||
}
|
||
}
|
||
}
|
||
|
||
// ---- Emitter (used by state.Tracker) ----
|
||
|
||
// EmitAppState forwards an AppStateUpdate upstream.
|
||
//
|
||
// Side effect: a successful state-source poll (the only thing that
|
||
// produces AppStateUpdate) implies the server is responsive. If the
|
||
// last status we sent was CRASHED, promote to RUNNING — handles the
|
||
// "segfault on first world-gen, healthy on retry" pattern where docker
|
||
// restart-policy auto-recovery in watchExit doesn't fire because the
|
||
// container's restart was driven by a different cause.
|
||
func (d *Dispatcher) EmitAppState(a *panelv1.AppStateUpdate) {
|
||
if a != nil && a.InstanceId != "" {
|
||
d.mu.Lock()
|
||
rec := d.instances[a.InstanceId]
|
||
d.mu.Unlock()
|
||
if rec != nil {
|
||
last := panelv1.InstanceStatus(rec.lastStatus.Load())
|
||
if last == panelv1.InstanceStatus_INSTANCE_STATUS_CRASHED {
|
||
d.sendInstanceState(a.InstanceId, panelv1.InstanceStatus_INSTANCE_STATUS_RUNNING, 0,
|
||
"running (recovered — state source healthy)")
|
||
}
|
||
}
|
||
}
|
||
d.sendEnv(&panelv1.AgentEnvelope{
|
||
SentAt: timestamppb.Now(),
|
||
Payload: &panelv1.AgentEnvelope_AppState{AppState: a},
|
||
})
|
||
}
|
||
|
||
// EmitPlayerEvent forwards a PlayerEvent upstream.
|
||
func (d *Dispatcher) EmitPlayerEvent(p *panelv1.PlayerEvent) {
|
||
d.sendEnv(&panelv1.AgentEnvelope{
|
||
SentAt: timestamppb.Now(),
|
||
Payload: &panelv1.AgentEnvelope_Player{Player: p},
|
||
})
|
||
}
|
||
|
||
// ---- Lifecycle handlers ----
|
||
|
||
// handleCreate resolves the manifest, creates the container, and records it.
|
||
// The container is NOT started here — that's InstanceStart's job.
|
||
func (d *Dispatcher) handleCreate(ctx context.Context, corrID string, req *panelv1.InstanceCreate) {
|
||
log := d.log.With("instance_id", req.InstanceId, "module_id", req.ModuleId, "correlation_id", corrID)
|
||
|
||
manifest, ok := d.modules.Get(req.ModuleId)
|
||
if !ok {
|
||
d.replyError(corrID, req.InstanceId, "module_not_found", fmt.Sprintf("module %q not in registry", req.ModuleId))
|
||
log.Warn("module not found")
|
||
return
|
||
}
|
||
|
||
if !manifest.HasMode("docker") {
|
||
d.replyError(corrID, req.InstanceId, "mode_unsupported", "module does not support docker mode")
|
||
return
|
||
}
|
||
|
||
if req.DataPath == "" {
|
||
req.DataPath = filepath.Join(d.dataRoot, req.InstanceId)
|
||
}
|
||
if err := os.MkdirAll(req.DataPath, 0o755); err != nil {
|
||
d.replyError(corrID, req.InstanceId, "data_path_mkdir", err.Error())
|
||
log.Error("mkdir data_path", "err", err)
|
||
return
|
||
}
|
||
|
||
// Expand templated paths in any mount_overrides the controller sent,
|
||
// then mkdir the resolved host path so Docker's bind-mount won't
|
||
// silently create an empty root-owned dir elsewhere.
|
||
//
|
||
// Templates understood (both are absolute after expansion):
|
||
// $AGENT_DATA_ROOT → agent's --data-root flag (where per-instance
|
||
// data lives; default ./data/instances)
|
||
// $PANEL_DATA_ROOT → parent of AGENT_DATA_ROOT — the panel's data
|
||
// root (./data on the default setup). This is
|
||
// where cross-instance shared dirs live (like
|
||
// ARK cluster transfer dirs, alongside
|
||
// instances/ and backups/).
|
||
if len(req.MountOverrides) > 0 {
|
||
panelRoot := filepath.Dir(d.dataRoot)
|
||
for k, v := range req.MountOverrides {
|
||
exp := v
|
||
exp = strings.ReplaceAll(exp, "$PANEL_DATA_ROOT", panelRoot)
|
||
exp = strings.ReplaceAll(exp, "$AGENT_DATA_ROOT", d.dataRoot)
|
||
req.MountOverrides[k] = exp
|
||
if filepath.IsAbs(exp) {
|
||
if err := os.MkdirAll(exp, 0o755); err != nil {
|
||
d.replyError(corrID, req.InstanceId, "mount_mkdir", err.Error())
|
||
log.Error("mkdir mount override", "path", exp, "err", err)
|
||
return
|
||
}
|
||
}
|
||
}
|
||
}
|
||
|
||
values, err := mergeValuesAndSecrets(manifest, req.ConfigValues)
|
||
if err != nil {
|
||
d.replyError(corrID, req.InstanceId, "secrets", err.Error())
|
||
log.Error("generate secrets", "err", err)
|
||
return
|
||
}
|
||
if err := modulepkg.RenderForBranch(manifest, req.DataPath, values, resolveCreateBranch(manifest, values)); err != nil {
|
||
d.replyError(corrID, req.InstanceId, "render", err.Error())
|
||
log.Error("render config files", "err", err)
|
||
return
|
||
}
|
||
req.ConfigValues = values
|
||
|
||
// _PANEL_SUPPRESS_AUTOSTART sentinel — controller-set marker for
|
||
// recreate dances (env-config / change-ports / cluster / mods) that
|
||
// preserve the prior running state. When true, skip the agent's
|
||
// post-create auto-start path so a stopped instance stays stopped
|
||
// after the recreate. The controller will issue an explicit Start
|
||
// itself when the prior state was "running".
|
||
suppressAutoStart := false
|
||
if v := req.ConfigValues["_PANEL_SUPPRESS_AUTOSTART"]; v == "true" {
|
||
suppressAutoStart = true
|
||
delete(req.ConfigValues, "_PANEL_SUPPRESS_AUTOSTART")
|
||
}
|
||
|
||
// _PANEL_SKIP_REINSTALL sentinel — controller-set marker for recreate
|
||
// dances that PRESERVE the install volume (env-config / change-ports /
|
||
// cluster / mods). On a recreate the game files + the operator's custom
|
||
// Mods are already on the volume, so the create-time warm-seed +
|
||
// first-update must NOT run: warm-seed does `find /game -delete` and
|
||
// re-copies from an ARBITRARY sibling, which silently drops a 7DTD
|
||
// server's custom Mods (the whole point of preserving the volume is to
|
||
// keep them). A first-ever create never carries this flag, so brand-new
|
||
// installs still warm-seed / download normally. Invariant for callers:
|
||
// any path that sets this must also set _PANEL_SUPPRESS_AUTOSTART when the
|
||
// prior state was stopped AND issue its own Start when it was running —
|
||
// the agent does neither for a skip-reinstall recreate.
|
||
skipReinstall := false
|
||
if v := req.ConfigValues["_PANEL_SKIP_REINSTALL"]; v == "true" {
|
||
skipReinstall = true
|
||
delete(req.ConfigValues, "_PANEL_SKIP_REINSTALL")
|
||
}
|
||
|
||
// _PANEL_SEED_MODS_FROM sentinel — controller-set marker for a cluster
|
||
// join that should clone the cluster master's Mods set onto this server.
|
||
// Value is the master's instance id. Handled in the skip-reinstall branch
|
||
// below (a cluster join is always a volume-preserving recreate): we run a
|
||
// Mods-only sidecar copy before the controller starts the server.
|
||
seedModsFrom := ""
|
||
if v := req.ConfigValues["_PANEL_SEED_MODS_FROM"]; v != "" {
|
||
seedModsFrom = v
|
||
delete(req.ConfigValues, "_PANEL_SEED_MODS_FROM")
|
||
}
|
||
|
||
// _PANEL_WIPE_PATHS sentinel — comma-separated list of paths
|
||
// (relative to BrowseableRoot) the controller wants nuked before
|
||
// the new container starts. Used by the ARK map-switch flow when
|
||
// the operator checks "wipe existing world data" — clean slate
|
||
// for the new map without keeping the old map's saves on disk.
|
||
//
|
||
// Strip the key so it doesn't pollute the container env. Wipe runs
|
||
// AFTER the previous container was deleted (volumes preserved) and
|
||
// BEFORE we create the new one, in an alpine sidecar that mounts
|
||
// the same volumes we're about to mount.
|
||
if wipeCSV := req.ConfigValues["_PANEL_WIPE_PATHS"]; wipeCSV != "" {
|
||
delete(req.ConfigValues, "_PANEL_WIPE_PATHS")
|
||
paths := []string{}
|
||
for _, p := range strings.Split(wipeCSV, ",") {
|
||
p = strings.TrimSpace(p)
|
||
if p != "" {
|
||
paths = append(paths, p)
|
||
}
|
||
}
|
||
if len(paths) > 0 {
|
||
if err := d.runWipeSidecar(ctx, manifest, req.InstanceId, req.DataPath, paths); err != nil {
|
||
d.replyError(corrID, req.InstanceId, "wipe", err.Error())
|
||
log.Error("wipe sidecar", "err", err)
|
||
return
|
||
}
|
||
log.Info("wiped paths before create", "instance_id", req.InstanceId, "paths", paths)
|
||
}
|
||
}
|
||
|
||
// Auto-bump host ports that collide with already-bound ports on the
|
||
// host. Without this, two ARK SA instances both try to bind 7777/UDP
|
||
// + 27020/TCP — the second container's port mapping silently drops
|
||
// (Docker creates the container without published ports), the agent's
|
||
// RCON tracker still dials the manifest default (27020), and ends up
|
||
// talking to whichever container won the binding. Result: both
|
||
// "running" servers share state from one container's RCON.
|
||
//
|
||
// We mutate req.Ports in place so both ResolveDocker (below) and
|
||
// resolveRCONAddress (after Create) see the bumped ports — keeping
|
||
// the RCON dial address aligned with the actual host binding.
|
||
if err := assignNonCollidingHostPorts(manifest, req); err != nil {
|
||
d.sendInstanceState(req.InstanceId, panelv1.InstanceStatus_INSTANCE_STATUS_CRASHED, -1, "port_alloc_failed: "+err.Error())
|
||
d.replyError(corrID, req.InstanceId, "port_alloc", err.Error())
|
||
log.Error("port allocation", "err", err)
|
||
return
|
||
}
|
||
|
||
// Mirror the assigned ports into ConfigValues for any port that
|
||
// declares an `env:` mapping. This is what makes a non-default
|
||
// host port reach the game binary — without it, two ARK instances
|
||
// pick distinct host ports but both ARKs internally bind 7777/27015
|
||
// because their ASA_PORT/QUERY_PORT envs are still the manifest
|
||
// defaults. Steam server browser would then list only one of them.
|
||
//
|
||
// Also force ContainerPort = HostPort for ALL ports. The game's
|
||
// env var tells it which port to bind INSIDE the container; that
|
||
// port has to match what Docker forwards from the outside, otherwise
|
||
// the agent's RCON dial (or a player's join attempt) hits an empty
|
||
// container port. Was previously a 1:1 mapping for non-internal
|
||
// ports only — but RCON's env var (e.g. RCON_PORT) also had to
|
||
// follow the host port to avoid the same mismatch internally.
|
||
if req.ConfigValues == nil {
|
||
req.ConfigValues = map[string]string{}
|
||
}
|
||
for i, p := range req.Ports {
|
||
if p == nil {
|
||
continue
|
||
}
|
||
decl := manifest.Port(p.Name)
|
||
if decl == nil {
|
||
continue
|
||
}
|
||
if decl.Env != "" {
|
||
req.ConfigValues[decl.Env] = fmt.Sprintf("%d", p.HostPort)
|
||
}
|
||
req.Ports[i].ContainerPort = p.HostPort
|
||
}
|
||
|
||
spec, err := agentmodule.ResolveDocker(manifest, req)
|
||
if err != nil {
|
||
d.replyError(corrID, req.InstanceId, "resolve", err.Error())
|
||
log.Error("resolve docker spec", "err", err)
|
||
return
|
||
}
|
||
|
||
// Forward image-acquisition progress (auto-build of `panel-*` images
|
||
// on first use, or registry pull progress) to the controller as log
|
||
// lines on the regular stdout stream — operator sees them in the
|
||
// Console tab while waiting on the create. Without this, first-time
|
||
// builds run silently for 30s–3min and the UI just sits on
|
||
// "installing".
|
||
spec.LogSink = func(line string) {
|
||
d.sendEnv(&panelv1.AgentEnvelope{
|
||
SentAt: timestamppb.Now(),
|
||
Payload: &panelv1.AgentEnvelope_Log{Log: &panelv1.LogLine{
|
||
InstanceId: req.InstanceId,
|
||
Stream: "stdout",
|
||
At: timestamppb.Now(),
|
||
Line: line,
|
||
}},
|
||
})
|
||
}
|
||
|
||
log.Info("pulling image and creating container", "image", spec.Image)
|
||
containerID, err := d.runtime.Create(ctx, spec)
|
||
if err != nil {
|
||
// Emit a terminal state so the controller's DB row moves out of
|
||
// "creating" and the operator can see what went wrong.
|
||
d.sendInstanceState(req.InstanceId, panelv1.InstanceStatus_INSTANCE_STATUS_CRASHED, -1, "create_failed: "+err.Error())
|
||
d.replyError(corrID, req.InstanceId, "create", err.Error())
|
||
log.Error("runtime create", "err", err)
|
||
return
|
||
}
|
||
|
||
rconAddr, rconPassword := resolveRCONAddress(manifest, req)
|
||
browseable := resolveBrowseableRoot(manifest)
|
||
roots := resolveBrowseableRoots(manifest)
|
||
|
||
rec := &instanceRecord{
|
||
InstanceID: req.InstanceId,
|
||
ContainerID: containerID,
|
||
ModuleID: req.ModuleId,
|
||
Branch: resolveCreateBranch(manifest, req.ConfigValues),
|
||
DataPath: req.DataPath,
|
||
BrowseableRoot: browseable,
|
||
BrowseableRoots: roots,
|
||
RCONAddr: rconAddr,
|
||
RCONPassword: rconPassword,
|
||
}
|
||
|
||
d.mu.Lock()
|
||
d.instances[req.InstanceId] = rec
|
||
d.mu.Unlock()
|
||
|
||
if err := d.writeMeta(rec); err != nil {
|
||
log.Warn("write sidecar meta failed (instance will not survive agent restart)", "err", err)
|
||
}
|
||
|
||
log.Info("instance created", "container_id", short(containerID), "rcon_addr", rconAddr)
|
||
// Detail string is parsed by the controller's UI to drive banners; for
|
||
// modules whose first update_provider needs Steam creds we tag the
|
||
// state so the dashboard can show a "Sign in + Update to install"
|
||
// banner directly on the card. Otherwise stay with the generic
|
||
// "created" detail.
|
||
createdDetail := "created"
|
||
if len(manifest.UpdateProviders) > 0 && manifest.UpdateProviders[0].RequiresSteamLogin {
|
||
createdDetail = "needs_steam_login"
|
||
}
|
||
d.sendInstanceState(req.InstanceId, panelv1.InstanceStatus_INSTANCE_STATUS_STOPPED, 0, createdDetail)
|
||
d.replyOK(corrID)
|
||
|
||
// Shared auto-start helper — used from both the update-success path and
|
||
// the "skip update but still auto-start" path so the Create flow lands
|
||
// on a running server regardless of which install model the module uses.
|
||
autoStart := func() {
|
||
startCorr := fmt.Sprintf("autostart-%d", time.Now().UnixNano())
|
||
d.log.Info("auto-starting after create", "instance_id", req.InstanceId, "correlation_id", startCorr)
|
||
startCtx, cancel := context.WithTimeout(context.Background(), 60*time.Second)
|
||
defer cancel()
|
||
d.handleStart(startCtx, startCorr, &panelv1.InstanceStart{InstanceId: req.InstanceId})
|
||
}
|
||
|
||
// Auto-fire the primary update provider so the operator doesn't have to
|
||
// click Update as a separate step. Opt-out via `auto_update_on_create:
|
||
// false` in module.yaml for modules whose image handles its own
|
||
// downloads (e.g. ark-sa's community image).
|
||
//
|
||
// Recreate-dance creates carry _PANEL_SUPPRESS_AUTOSTART=true; for those
|
||
// we skip BOTH the update kick AND the auto-start so a previously
|
||
// stopped instance stays stopped after an env-config edit. Volumes
|
||
// already hold the game files (recreate preserves them), so re-running
|
||
// the updater would just waste minutes and then auto-start at the end
|
||
// via onUpdateComplete. Controller will issue an explicit Start when
|
||
// the prior state was running.
|
||
if suppressAutoStart {
|
||
log.Info("create: auto-start + auto-update suppressed (recreate dance, prior state was stopped)", "instance_id", req.InstanceId)
|
||
} else if skipReinstall {
|
||
// Recreate that PRESERVES a populated install volume (env-config /
|
||
// cluster / ports / mods). Skip the create-time warm-seed +
|
||
// first-update — they wipe /game and re-copy from a sibling, which
|
||
// silently drops a 7DTD server's custom Mods. The controller drives
|
||
// the post-recreate start itself (explicit Start when the prior state
|
||
// was running; stays stopped otherwise via _PANEL_SUPPRESS_AUTOSTART),
|
||
// so the agent does nothing further here — EXCEPT an optional mod-seed.
|
||
if seedModsFrom != "" {
|
||
// Cluster join with "seed mods from master": clone the master's
|
||
// Mods set onto this server before the controller starts it.
|
||
// Synchronous so the seeded mods are in place by the time the
|
||
// create reply returns and the controller issues Start (~5-10s for
|
||
// a typical ~170 MB 7DTD mod set — well inside the 90s create
|
||
// budget). Best-effort: a failure leaves the existing mods intact.
|
||
seedCtx, cancel := context.WithTimeout(context.Background(), 75*time.Second)
|
||
if _, serr := d.seedModsFromInstance(seedCtx, seedModsFrom, req.InstanceId, req.DataPath, manifest); serr != nil {
|
||
log.Warn("mod-seed failed — server keeps its existing mods", "instance_id", req.InstanceId, "src", seedModsFrom, "err", serr)
|
||
}
|
||
cancel()
|
||
}
|
||
log.Info("create: install preserved (recreate dance) — skipping warm-seed + update", "instance_id", req.InstanceId)
|
||
} else if shouldAutoUpdate(manifest) {
|
||
// Default to the first declared provider, but allow a config_value
|
||
// to override it at create time. This is what lets Conan Exiles
|
||
// pick `enhanced` (UE5) vs `legacy` (UE4) from the create wizard's
|
||
// EDITION dropdown — the matching update_provider id fires on
|
||
// first install. Generic by design: any module can expose a
|
||
// `provider_id` config_value and have it auto-honored.
|
||
providerID := manifest.UpdateProviders[0].ID
|
||
if pref := pickProviderFromConfigValues(req.ConfigValues, manifest); pref != "" {
|
||
providerID = pref
|
||
}
|
||
log.Info("auto-triggering first update", "provider_id", providerID,
|
||
"chosen_via_config_values", providerID != manifest.UpdateProviders[0].ID)
|
||
autoCorr := fmt.Sprintf("autoupdate-%d", time.Now().UnixNano())
|
||
instanceID := req.InstanceId
|
||
// Surface the install as a server-authoritative state so any operator
|
||
// viewing the dashboard (not just the one who clicked Create) sees
|
||
// the "updating" pulse. The client was previously only flipping
|
||
// is-installing locally on its own Create button press — refreshing
|
||
// or joining a session mid-install showed "stopped" incorrectly.
|
||
// handleUpdateWithHook emits UPDATING again immediately below, but
|
||
// firing it here too closes the UI race between create-completion
|
||
// and the goroutine starting.
|
||
d.sendInstanceState(instanceID, panelv1.InstanceStatus_INSTANCE_STATUS_UPDATING, 0, "installing")
|
||
|
||
// When the first-time download finishes, auto-start the server. That
|
||
// way Create → "installing" → "starting" → "running" happens as one
|
||
// operator gesture instead of requiring a separate Start click. Mirrors
|
||
// AMP-style "create and run" behaviour. Guarded by the same flag as
|
||
// auto-update itself — a module that opts out of auto-update stays
|
||
// stopped after create as before.
|
||
onUpdateComplete := func(err error) {
|
||
if err != nil {
|
||
// Clear the "installing" pulse so the UI stops pretending the
|
||
// download is in flight. Detail carries the error for anyone
|
||
// inspecting the DB / logs.
|
||
d.sendInstanceState(instanceID, panelv1.InstanceStatus_INSTANCE_STATUS_CRASHED, -1, "install_failed: "+err.Error())
|
||
return
|
||
}
|
||
// Cluster join "seed mods from master" requested at CREATE time: on a
|
||
// FRESH create the install ran through this warm-seed/update path (not
|
||
// the skip-reinstall branch that also honors seedModsFrom), so clone
|
||
// the master's Mods set HERE — after the base install, before the
|
||
// server is startable. Best-effort: a failure keeps the base mods.
|
||
if seedModsFrom != "" {
|
||
d.sendInstanceState(instanceID, panelv1.InstanceStatus_INSTANCE_STATUS_UPDATING, 0, fmt.Sprintf("seeding mods from cluster master (%s)", seedModsFrom))
|
||
seedCtx, cancel := context.WithTimeout(context.Background(), 90*time.Second)
|
||
if _, serr := d.seedModsFromInstance(seedCtx, seedModsFrom, instanceID, req.DataPath, manifest); serr != nil {
|
||
log.Warn("create mod-seed failed — server keeps its base mods", "instance_id", instanceID, "src", seedModsFrom, "err", serr)
|
||
}
|
||
cancel()
|
||
}
|
||
// Drop the installing marker before kicking Start — handleStart
|
||
// sends its own STARTING event which will supersede, but emitting
|
||
// the transition explicitly avoids an awkward "installing" flash
|
||
// if Start races the next poll tick.
|
||
d.sendInstanceState(instanceID, panelv1.InstanceStatus_INSTANCE_STATUS_STOPPED, 0, "installed")
|
||
// Auto-start ONLY if the module opts in. 7DTD sets
|
||
// auto_start_on_create=false so the operator finishes world/RWG +
|
||
// cluster setup and starts manually (and to avoid the warm-seed →
|
||
// immediate-start memory race that locked up the agent host).
|
||
if shouldAutoStart(manifest) {
|
||
autoStart()
|
||
} else {
|
||
log.Info("create: install complete — auto-start suppressed (auto_start_on_create=false); operator starts manually", "instance_id", instanceID)
|
||
}
|
||
}
|
||
// Warm-seed shortcut: if a sibling instance of the same module
|
||
// already has the install volume populated, copy it instead of
|
||
// running the updater. Saves the operator a 30-90 min SteamCMD
|
||
// round-trip on the second/third/Nth instance of a game. If no
|
||
// sibling or the seed fails, fall through to the normal updater.
|
||
// The seed itself runs in a debian sidecar and takes ~2-5 min
|
||
// for ~16 GB on local SSD.
|
||
dataPath := req.DataPath
|
||
go func() {
|
||
seedCtx, cancel := context.WithTimeout(context.Background(), 15*time.Minute)
|
||
defer cancel()
|
||
seeded, err := d.tryWarmSeedFromSibling(seedCtx, instanceID, manifest, dataPath)
|
||
if err != nil {
|
||
log.Warn("warm-seed failed, falling through to updater", "instance_id", instanceID, "err", err)
|
||
} else if seeded {
|
||
log.Info("warm-seed succeeded — skipping initial updater run", "instance_id", instanceID, "module_id", manifest.ID)
|
||
onUpdateComplete(nil)
|
||
return
|
||
}
|
||
d.handleUpdateWithHook(autoCorr, &panelv1.UpdateRequest{
|
||
InstanceId: instanceID,
|
||
ProviderId: providerID,
|
||
}, onUpdateComplete)
|
||
}()
|
||
} else if shouldAutoStart(manifest) {
|
||
// Image-managed install path (ark-sa et al): no external updater but
|
||
// we still want Create → running in one gesture. The container's
|
||
// entrypoint handles first-boot SteamCMD itself, so all we do here
|
||
// is kick Start. UI sees STARTING (yellow pulse) until the game's
|
||
// ready-regex fires. Slight delay gives the agent's replyOK envelope
|
||
// time to reach the controller before the next state update.
|
||
log.Info("auto-starting (module handles install via container entrypoint)")
|
||
go func() {
|
||
time.Sleep(250 * time.Millisecond)
|
||
autoStart()
|
||
}()
|
||
}
|
||
}
|
||
|
||
// shouldAutoStart decides whether a just-created instance should boot
|
||
// automatically after the create (and first-time update, if any) lands.
|
||
// Default: yes. Modules that legitimately want "create then wait for the
|
||
// operator to configure before starting" opt out with
|
||
// `auto_start_on_create: false`. Independent of shouldAutoUpdate so that
|
||
// image-managed modules (ark-sa: no external updater, in-container
|
||
// SteamCMD) still end the Create flow on a running server.
|
||
//
|
||
// EXCEPTION — Steam-login-required modules: when the first update_provider
|
||
// has requires_steam_login, the create-time auto-update is skipped (no
|
||
// creds to use), so the game files were never downloaded. Auto-starting
|
||
// in that state would just bounce the container with an "executable not
|
||
// found" error and leave the instance crashed. We hold off until the
|
||
// operator has signed in to Steam + clicked Update; the update path's
|
||
// onUpdateComplete hook fires the start at the right moment.
|
||
func shouldAutoStart(m *modulepkg.Manifest) bool {
|
||
if m == nil {
|
||
return false
|
||
}
|
||
if m.AutoStartOnCreate != nil {
|
||
return *m.AutoStartOnCreate
|
||
}
|
||
// Don't auto-start before the Steam-gated install has run.
|
||
if len(m.UpdateProviders) > 0 && m.UpdateProviders[0].RequiresSteamLogin {
|
||
return false
|
||
}
|
||
return true
|
||
}
|
||
|
||
// shouldAutoUpdate decides whether a just-created instance should kick off
|
||
// its first update provider automatically. Default: yes, if there's any
|
||
// provider declared. Modules whose container image handles its own
|
||
// downloads (ark-sa via UPDATE_SERVER=TRUE) should set
|
||
// `auto_update_on_create: false` in their manifest.
|
||
func shouldAutoUpdate(m *modulepkg.Manifest) bool {
|
||
if m == nil || len(m.UpdateProviders) == 0 {
|
||
return false
|
||
}
|
||
if m.AutoUpdateOnCreate != nil {
|
||
return *m.AutoUpdateOnCreate
|
||
}
|
||
// Modules whose first provider requires a Steam login can't auto-update:
|
||
// the agent doesn't have credentials on the auto-create path (the
|
||
// controller-side Steam-login gate doesn't intercept create events).
|
||
// The operator clicks Update manually once creds are in place; that
|
||
// path flows through the controller and picks up the cached login.
|
||
if m.UpdateProviders[0].RequiresSteamLogin {
|
||
return false
|
||
}
|
||
return true
|
||
}
|
||
|
||
// pickProviderFromConfigValues lets a module expose a config_value that
|
||
// overrides which update_provider auto-fires at create time. Two keys are
|
||
// honored, in order of preference:
|
||
//
|
||
// 1. config_values["provider_id"] — explicit, generic. Matches an
|
||
// update_provider's `id` directly.
|
||
// 2. config_values["EDITION"] — module-specific convention used by
|
||
// Conan Exiles (enhanced/legacy → matching provider id).
|
||
//
|
||
// If neither is set, or the named provider doesn't exist, returns "" so
|
||
// the caller falls back to UpdateProviders[0]. Case-insensitive match
|
||
// against the provider id; trim whitespace defensively because
|
||
// `runtime.docker.env` defaults can sneak through with stray spaces.
|
||
func pickProviderFromConfigValues(cv map[string]string, m *modulepkg.Manifest) string {
|
||
if cv == nil || m == nil || len(m.UpdateProviders) == 0 {
|
||
return ""
|
||
}
|
||
candidates := []string{cv["provider_id"], cv["EDITION"]}
|
||
for _, want := range candidates {
|
||
want = strings.TrimSpace(strings.ToLower(want))
|
||
if want == "" {
|
||
continue
|
||
}
|
||
for _, p := range m.UpdateProviders {
|
||
if strings.ToLower(p.ID) == want {
|
||
return p.ID
|
||
}
|
||
}
|
||
}
|
||
return ""
|
||
}
|
||
|
||
// resolveCreateBranch returns the normalized Steam branch an instance is being
|
||
// installed from, derived from the update_provider it will use (the same
|
||
// selection logic as the create-time auto-update: config_values.provider_id
|
||
// via pickProviderFromConfigValues, else UpdateProviders[0]). The value tags
|
||
// the instance record so warm-seed only reuses an install from a sibling on
|
||
// the SAME branch. See normalizeBranch for the public-equivalent collapsing.
|
||
func resolveCreateBranch(m *modulepkg.Manifest, cv map[string]string) string {
|
||
if m == nil || len(m.UpdateProviders) == 0 {
|
||
return ""
|
||
}
|
||
id := m.UpdateProviders[0].ID
|
||
if pref := pickProviderFromConfigValues(cv, m); pref != "" {
|
||
id = pref
|
||
}
|
||
for i := range m.UpdateProviders {
|
||
if m.UpdateProviders[i].ID == id {
|
||
return normalizeBranch(m.UpdateProviders[i].Beta)
|
||
}
|
||
}
|
||
return ""
|
||
}
|
||
|
||
// handleStart starts a previously-created instance and begins log streaming.
|
||
func (d *Dispatcher) handleStart(ctx context.Context, corrID string, req *panelv1.InstanceStart) {
|
||
log := d.log.With("instance_id", req.InstanceId, "correlation_id", corrID)
|
||
|
||
d.mu.Lock()
|
||
rec, ok := d.instances[req.InstanceId]
|
||
d.mu.Unlock()
|
||
if !ok {
|
||
d.replyError(corrID, req.InstanceId, "not_found", "instance not created on this target")
|
||
return
|
||
}
|
||
|
||
d.sendInstanceState(req.InstanceId, panelv1.InstanceStatus_INSTANCE_STATUS_STARTING, 0, "starting")
|
||
|
||
// A Start racing an in-flight graceful stop is a trap: docker start on a
|
||
// container that's still shutting down is a no-op "success", and when the
|
||
// stop's SIGTERM grace finally lands it kills the instance the operator
|
||
// just asked to start (observed live: a UI restart left the server
|
||
// stranded stopped). Wait for the stop to settle before proceeding —
|
||
// bounded so a wedged stop can't park Start forever.
|
||
if rec.stopping.Load() {
|
||
log.Info("start racing an in-flight stop — waiting for it to settle")
|
||
d.sendInstanceState(req.InstanceId, panelv1.InstanceStatus_INSTANCE_STATUS_STARTING, 0, "waiting for in-flight stop to finish…")
|
||
deadline := time.Now().Add(90 * time.Second)
|
||
for rec.stopping.Load() && time.Now().Before(deadline) {
|
||
select {
|
||
case <-ctx.Done():
|
||
d.replyError(corrID, req.InstanceId, "cancelled", ctx.Err().Error())
|
||
return
|
||
case <-time.After(time.Second):
|
||
}
|
||
}
|
||
}
|
||
|
||
// Region Medic — for 7DTD, validate + heal the active world's regions while
|
||
// the container is still STOPPED, snapshot a fresh rolling backup, and ping
|
||
// Discord on heal. The CLI reads its per-server config (enable/keep/channel)
|
||
// from region-medic.json in the saves volume. Best-effort: never blocks start.
|
||
if rec.ModuleID == "7dtd" {
|
||
d.sendInstanceState(req.InstanceId, panelv1.InstanceStatus_INSTANCE_STATUS_STARTING, 0, "region medic: validating world…")
|
||
d.runRegionMedicOnStart(ctx, rec)
|
||
// Snapshot the shared cluster Player/ dir (character .ttp saves) to a
|
||
// rolling slot before boot. No-op for standalone servers. The panel
|
||
// tar backup can't capture this — the active Player dir is a symlink
|
||
// out to the /cluster bind — so this is the only restore path for
|
||
// per-character data on a clustered server.
|
||
d.snapshotClusterPlayerOnStart(ctx, rec)
|
||
}
|
||
|
||
// Retry start on transient port-bind races. When a container exits the
|
||
// kernel can hold its UDP bindings for a few seconds; the next Start
|
||
// sees "ports are not available" until the OS releases them. A short
|
||
// backoff is much friendlier than making the user click Start twice.
|
||
const maxAttempts = 4
|
||
var err error
|
||
for attempt := 1; attempt <= maxAttempts; attempt++ {
|
||
err = d.runtime.Start(ctx, rec.ContainerID)
|
||
if err == nil {
|
||
break
|
||
}
|
||
if !isTransientStartError(err) || attempt == maxAttempts {
|
||
break
|
||
}
|
||
wait := time.Duration(attempt*2) * time.Second
|
||
log.Warn("transient start error, retrying", "err", err, "attempt", attempt, "wait", wait)
|
||
d.sendInstanceState(req.InstanceId, panelv1.InstanceStatus_INSTANCE_STATUS_STARTING, 0,
|
||
fmt.Sprintf("retrying (port still held): attempt %d/%d", attempt+1, maxAttempts))
|
||
select {
|
||
case <-ctx.Done():
|
||
d.replyError(corrID, req.InstanceId, "cancelled", ctx.Err().Error())
|
||
return
|
||
case <-time.After(wait):
|
||
}
|
||
}
|
||
if err != nil {
|
||
d.replyError(corrID, req.InstanceId, "start", err.Error())
|
||
d.sendInstanceState(req.InstanceId, panelv1.InstanceStatus_INSTANCE_STATUS_CRASHED, -1, err.Error())
|
||
log.Error("runtime start", "err", err)
|
||
return
|
||
}
|
||
d.activate(rec)
|
||
d.sendInstanceState(req.InstanceId, panelv1.InstanceStatus_INSTANCE_STATUS_RUNNING, 0, "running")
|
||
d.replyOK(corrID)
|
||
log.Info("instance started")
|
||
|
||
// Auto-provision a freshly-generated clustered 7DTD world: watch (best-effort,
|
||
// off-thread) for the gen boot to bake native-banded decorations and bounce the
|
||
// container ONCE so the entrypoint aligns them to the cluster canonical. Strict
|
||
// no-op for already-canon worlds (season10/insane/pvp/creative) and non-cluster
|
||
// servers; single-flight. See decoremap_provision.go.
|
||
if rec.ModuleID == "7dtd" {
|
||
d.autoProvisionClusterWorldOnStart(rec)
|
||
}
|
||
|
||
// post_start lifecycle hooks — best-effort shell commands that run
|
||
// INSIDE the container after it's up. Used to patch image-generated
|
||
// config files that need a panel-side fixup (e.g. ark-sa's RCON port
|
||
// needs to be duplicated into [SessionSettings] of GameUserSettings.ini
|
||
// because ASA ignores AUTH responses when it's only under
|
||
// [ServerSettings]). Failures are logged but don't fail the Start —
|
||
// a broken hook shouldn't block an otherwise-healthy server.
|
||
if manifest, ok := d.modules.Get(rec.ModuleID); ok && manifest.Lifecycle != nil && len(manifest.Lifecycle.PostStart) > 0 {
|
||
go d.runPostStartHooks(rec, manifest.Lifecycle.PostStart)
|
||
}
|
||
}
|
||
|
||
// runPostStartHooks executes each post_start command inside the running
|
||
// container, via docker exec. Runs async so a slow or blocking hook (e.g.
|
||
// one that waits for a config file to be written by the entrypoint) never
|
||
// holds up the Start reply or the dashboard state transition to RUNNING.
|
||
func (d *Dispatcher) runPostStartHooks(rec *instanceRecord, commands []string) {
|
||
log := d.log.With("instance_id", rec.InstanceID, "phase", "post_start")
|
||
// Bound the whole hook suite at 10 minutes — ASA's ini fixup waits
|
||
// for the image's entrypoint to write the file on first boot, which
|
||
// on a cold SteamCMD install can take several minutes.
|
||
ctx, cancel := context.WithTimeout(context.Background(), 10*time.Minute)
|
||
defer cancel()
|
||
for i, cmd := range commands {
|
||
if cmd == "" {
|
||
continue
|
||
}
|
||
log.Info("running hook", "index", i, "cmd_preview", previewCmd(cmd))
|
||
stdout, stderr, code, err := d.runtime.ExecCapture(ctx, rec.ContainerID, []string{"sh", "-c", cmd})
|
||
if err != nil {
|
||
log.Warn("hook exec failed", "index", i, "err", err)
|
||
continue
|
||
}
|
||
if code != 0 {
|
||
log.Warn("hook non-zero exit", "index", i, "exit_code", code,
|
||
"stdout", trimForLog(stdout), "stderr", trimForLog(stderr))
|
||
continue
|
||
}
|
||
log.Info("hook ok", "index", i, "exit_code", code, "stdout", trimForLog(stdout))
|
||
}
|
||
}
|
||
|
||
func previewCmd(s string) string {
|
||
s = strings.TrimSpace(strings.ReplaceAll(s, "\n", " "))
|
||
if len(s) > 80 {
|
||
return s[:80] + "…"
|
||
}
|
||
return s
|
||
}
|
||
|
||
func trimForLog(b []byte) string {
|
||
s := strings.TrimSpace(string(b))
|
||
if len(s) > 400 {
|
||
s = s[:400] + "…"
|
||
}
|
||
return s
|
||
}
|
||
|
||
// extractINIValue pulls a single key from a specific section of an INI
|
||
// file. If section is empty the first matching key anywhere in the file
|
||
// wins. Lines with `;` or `#` as the first non-space character are
|
||
// comments. Whitespace around the value is trimmed.
|
||
//
|
||
// Used to read live RCON passwords out of game config files (ARK's
|
||
// [ServerSettings]ServerAdminPassword) so operator edits through the
|
||
// Config tab take effect without panel manual intervention.
|
||
func extractINIValue(content, section, key string) (string, error) {
|
||
inSection := section == ""
|
||
scanner := strings.Split(content, "\n")
|
||
for _, raw := range scanner {
|
||
line := strings.TrimSpace(raw)
|
||
if line == "" || strings.HasPrefix(line, ";") || strings.HasPrefix(line, "#") {
|
||
continue
|
||
}
|
||
if strings.HasPrefix(line, "[") && strings.HasSuffix(line, "]") {
|
||
sec := line[1 : len(line)-1]
|
||
if section == "" {
|
||
// No section filter — keep scanning across sections.
|
||
continue
|
||
}
|
||
inSection = sec == section
|
||
continue
|
||
}
|
||
if !inSection {
|
||
continue
|
||
}
|
||
eq := strings.IndexByte(line, '=')
|
||
if eq <= 0 {
|
||
continue
|
||
}
|
||
k := strings.TrimSpace(line[:eq])
|
||
if k == key {
|
||
return strings.TrimSpace(line[eq+1:]), nil
|
||
}
|
||
}
|
||
return "", fmt.Errorf("ini key [%s]%s not found", section, key)
|
||
}
|
||
|
||
// isTransientStartError detects Docker errors that typically resolve within
|
||
// a few seconds (port bind still held by kernel after previous container
|
||
// exit, daemon momentarily busy). We treat these as retryable so the user
|
||
// doesn't see a stale "ports unavailable" the first time they click Start.
|
||
func isTransientStartError(err error) bool {
|
||
if err == nil {
|
||
return false
|
||
}
|
||
s := err.Error()
|
||
return strings.Contains(s, "ports are not available") ||
|
||
strings.Contains(s, "port is already allocated") ||
|
||
strings.Contains(s, "address already in use") ||
|
||
strings.Contains(s, "bind: Only one usage")
|
||
}
|
||
|
||
// activate starts log streaming + state tracker for a running instance.
|
||
// Idempotent: if LogCancel is already set, this is a no-op.
|
||
func (d *Dispatcher) activate(rec *instanceRecord) {
|
||
d.mu.Lock()
|
||
alreadyActive := rec.LogCancel != nil
|
||
d.mu.Unlock()
|
||
if alreadyActive {
|
||
return
|
||
}
|
||
// Free the shared volumes for the game by tearing down the fs helper.
|
||
// Both containers on the same volume is technically allowed, but we
|
||
// prefer the game own it exclusively while it's running.
|
||
d.teardownFsHelper(rec.InstanceID)
|
||
manifest, ok := d.modules.Get(rec.ModuleID)
|
||
if !ok {
|
||
d.log.Warn("activate: module vanished", "instance_id", rec.InstanceID, "module_id", rec.ModuleID)
|
||
return
|
||
}
|
||
|
||
// Tear down any tracker still attached to this rec from a previous
|
||
// activation BEFORE building a new one. activate() is called from
|
||
// handleStart (which doesn't pre-cancel), Announce on every controller
|
||
// reconnect (definitely doesn't pre-cancel), and the post-warm-seed
|
||
// autostart in handleCreate. Without this guard each call leaves the
|
||
// previous goroutine alive with its frozen cfg.RCONAddr — visible as
|
||
// "rcon dial failed ... 127.0.0.1:<old_port>" forever after a recreate
|
||
// (the old tracker still references the old container's port allocation;
|
||
// only an agent restart cleared it). Wait briefly for the goroutine to
|
||
// exit so the fresh dial doesn't race the stale one.
|
||
d.mu.Lock()
|
||
prevCancel := rec.TrackerCancel
|
||
prevDone := rec.TrackerDone
|
||
rec.TrackerCancel = nil
|
||
rec.Tracker = nil
|
||
rec.TrackerDone = nil
|
||
d.mu.Unlock()
|
||
if prevCancel != nil {
|
||
prevCancel()
|
||
if prevDone != nil {
|
||
select {
|
||
case <-prevDone:
|
||
case <-time.After(5 * time.Second):
|
||
d.log.Warn("activate: prior tracker did not exit within 5s",
|
||
"instance_id", rec.InstanceID)
|
||
}
|
||
}
|
||
}
|
||
|
||
// On rehydrate, sidecar may have an empty RCONPassword (older agent
|
||
// versions didn't persist secrets, or the password lookup hadn't
|
||
// been wired for this module yet). Re-resolve from the live manifest
|
||
// so a manifest update (e.g. adding password_secret to ark-sa) takes
|
||
// effect on the next start without forcing a delete + recreate.
|
||
if rec.RCONPassword == "" && manifest.RCON != nil {
|
||
_, pw := resolveRCONAddress(manifest, &panelv1.InstanceCreate{InstanceId: rec.InstanceID})
|
||
if pw != "" {
|
||
rec.RCONPassword = pw
|
||
_ = d.writeMeta(rec) // best-effort persist
|
||
}
|
||
}
|
||
|
||
// If the module declares a way to source the RCON password at runtime
|
||
// (from a file or from a specific INI key), wire a PasswordFunc that
|
||
// re-reads on each dial attempt. Lets modules whose entrypoint
|
||
// generates secrets at first boot (7DTD's TelnetPassword via
|
||
// /game-saves/.panel-telnet-password) or whose config file is the
|
||
// source of truth (ARK: SA's [ServerSettings]ServerAdminPassword)
|
||
// work without manual wiring.
|
||
var passwordFunc func(context.Context) (string, error)
|
||
if manifest.RCON != nil {
|
||
if iniSpec := manifest.RCON.PasswordFromINI; iniSpec != nil && iniSpec.File != "" && iniSpec.Key != "" {
|
||
path := iniSpec.File
|
||
section := iniSpec.Section
|
||
key := iniSpec.Key
|
||
containerID := rec.ContainerID
|
||
rt := d.runtime
|
||
passwordFunc = func(ctx context.Context) (string, error) {
|
||
data, err := rt.CopyFileFromContainer(ctx, containerID, path)
|
||
if err != nil {
|
||
return "", err
|
||
}
|
||
return extractINIValue(string(data), section, key)
|
||
}
|
||
} else if manifest.RCON.PasswordFromFile != "" {
|
||
path := manifest.RCON.PasswordFromFile
|
||
containerID := rec.ContainerID
|
||
rt := d.runtime
|
||
passwordFunc = func(ctx context.Context) (string, error) {
|
||
data, err := rt.CopyFileFromContainer(ctx, containerID, path)
|
||
if err != nil {
|
||
return "", err
|
||
}
|
||
return strings.TrimSpace(string(data)), nil
|
||
}
|
||
}
|
||
}
|
||
|
||
tracker, err := state.New(d.log, state.Config{
|
||
InstanceID: rec.InstanceID,
|
||
Manifest: manifest,
|
||
RCONAddr: rec.RCONAddr,
|
||
RCONPassword: rec.RCONPassword,
|
||
PasswordFunc: passwordFunc,
|
||
Emitter: d,
|
||
// stdio adapter needs a handle to the runtime so it can attach to
|
||
// the container's stdin. Irrelevant for tcp adapters — tracker
|
||
// ignores these fields unless manifest.RCON.Adapter == "stdio".
|
||
ContainerID: "panel-" + rec.InstanceID,
|
||
Stdio: d.runtime.(rcon.StdioBackend),
|
||
// arkHangGuard hook — counts consecutive RCON failures per
|
||
// instance and triggers an auto-restart at threshold. No-op for
|
||
// non-ark-sa modules (the guard short-circuits inside).
|
||
OnRconResult: func(success bool) { d.arkHangGuardOnRconResult(rec, success) },
|
||
})
|
||
if err != nil {
|
||
d.log.Warn("state tracker init failed (continuing without it)", "instance_id", rec.InstanceID, "err", err)
|
||
} else {
|
||
trackerCtx, trackerCancel := context.WithCancel(context.Background())
|
||
done := make(chan struct{})
|
||
d.mu.Lock()
|
||
rec.Tracker = tracker
|
||
rec.TrackerCancel = trackerCancel
|
||
rec.TrackerDone = done
|
||
d.mu.Unlock()
|
||
go func() {
|
||
defer close(done)
|
||
tracker.Run(trackerCtx)
|
||
}()
|
||
}
|
||
|
||
// Same teardown discipline for the log + stats streams: every caller of
|
||
// activate (Announce, handleStart, post-warm-seed autostart) reaches
|
||
// here, and most pre-cancel LogCancel — but not all. Cancel anything
|
||
// lingering before installing fresh streams so we don't end up with
|
||
// two streamLogs goroutines duplicating every log line into the
|
||
// emitter.
|
||
d.mu.Lock()
|
||
prevLogCancel := rec.LogCancel
|
||
rec.LogCancel = nil
|
||
d.mu.Unlock()
|
||
if prevLogCancel != nil {
|
||
prevLogCancel()
|
||
}
|
||
|
||
logCtx, logCancel := context.WithCancel(context.Background())
|
||
d.mu.Lock()
|
||
rec.LogCancel = logCancel
|
||
d.mu.Unlock()
|
||
go d.streamLogs(logCtx, rec)
|
||
d.startExitWatch(rec)
|
||
// Stats poller shares the log-lifetime context: cancelled on stop,
|
||
// auto-closed if the container exits.
|
||
go d.streamStats(logCtx, rec)
|
||
}
|
||
|
||
// startExitWatch (re)arms the exit watcher for a record, single-flight:
|
||
// any watcher from a previous activation is cancelled first, so re-Announce
|
||
// on controller reconnects can't stack N watchExit goroutines each holding
|
||
// its own ContainerWait docker connection.
|
||
func (d *Dispatcher) startExitWatch(rec *instanceRecord) {
|
||
ctx, cancel := context.WithCancel(context.Background())
|
||
d.mu.Lock()
|
||
prev := rec.ExitWatchCancel
|
||
rec.ExitWatchCancel = cancel
|
||
d.mu.Unlock()
|
||
if prev != nil {
|
||
prev()
|
||
}
|
||
go d.watchExit(ctx, rec)
|
||
}
|
||
|
||
// handleStop stops the container gracefully and leaves everything else
|
||
// intact — the in-memory record, the sidecar JSON, the Docker container
|
||
// (in Exited state), the volumes, and the DB row all persist so the
|
||
// instance can be re-started, backed up, or restored. Full teardown is
|
||
// InstanceDelete's job.
|
||
func (d *Dispatcher) handleStop(ctx context.Context, corrID string, req *panelv1.InstanceStop) {
|
||
log := d.log.With("instance_id", req.InstanceId, "correlation_id", corrID)
|
||
|
||
d.mu.Lock()
|
||
rec, ok := d.instances[req.InstanceId]
|
||
d.mu.Unlock()
|
||
if !ok {
|
||
d.replyError(corrID, req.InstanceId, "not_found", "instance not on this target")
|
||
return
|
||
}
|
||
|
||
rec.stopping.Store(true)
|
||
d.sendInstanceState(req.InstanceId, panelv1.InstanceStatus_INSTANCE_STATUS_STOPPING, 0, "stopping")
|
||
grace := time.Duration(req.GraceSeconds) * time.Second
|
||
if grace == 0 {
|
||
grace = 30 * time.Second
|
||
}
|
||
// Cancel any in-flight updater (steamcmd download, github fetch).
|
||
// Otherwise Stop kills the main container but the sidecar keeps
|
||
// running, which the operator perceives as "stop didn't work — it's
|
||
// still downloading."
|
||
d.cancelUpdater(req.InstanceId)
|
||
|
||
// Graceful in-game save BEFORE signalling the container. A bare docker
|
||
// stop (SIGTERM→SIGKILL after grace) can tear a region file mid-write —
|
||
// the root cause of 7DTD .7rg "Wrong chunk header!" corruption that
|
||
// Region Medic exists to repair. If the module declares an RCON `save`
|
||
// command and the tracker is live, flush the world first. Best-effort:
|
||
// short timeout, and fall through to the normal stop on any error.
|
||
d.mu.Lock()
|
||
tracker := rec.Tracker
|
||
d.mu.Unlock()
|
||
if manifest, ok := d.modules.Get(rec.ModuleID); !req.Force && tracker != nil && ok && manifest.RCON != nil {
|
||
if saveCmd := manifest.RCON.Commands["save"]; saveCmd != "" {
|
||
saveCtx, cancelSave := context.WithTimeout(ctx, 15*time.Second)
|
||
if out, err := tracker.Exec(saveCtx, saveCmd); err != nil {
|
||
log.Warn("graceful save before stop failed (continuing to stop)", "cmd", saveCmd, "err", err)
|
||
} else {
|
||
log.Info("graceful save before stop", "cmd", saveCmd, "out", strings.TrimSpace(out))
|
||
time.Sleep(2 * time.Second) // let the engine flush region files to disk
|
||
}
|
||
cancelSave()
|
||
}
|
||
}
|
||
|
||
// Tear down the tracker, log/stats streams, and exit watcher. Mutations
|
||
// go under d.mu — activate/Announce/Shutdown touch the same fields from
|
||
// other goroutines; the cancel calls themselves happen outside the lock.
|
||
d.mu.Lock()
|
||
trackerCancel := rec.TrackerCancel
|
||
logCancel := rec.LogCancel
|
||
exitWatchCancel := rec.ExitWatchCancel
|
||
rec.TrackerCancel = nil
|
||
rec.Tracker = nil
|
||
rec.LogCancel = nil
|
||
rec.ExitWatchCancel = nil
|
||
d.mu.Unlock()
|
||
for _, c := range []context.CancelFunc{trackerCancel, logCancel, exitWatchCancel} {
|
||
if c != nil {
|
||
c()
|
||
}
|
||
}
|
||
if err := d.runtime.Stop(ctx, rec.ContainerID, grace); err != nil {
|
||
// Clear the flag on failure too — the container is still running, so
|
||
// a future exit is NOT part of a graceful stop, and handleStart's
|
||
// race guard must not wait on a stop that already gave up.
|
||
rec.stopping.Store(false)
|
||
d.replyError(corrID, req.InstanceId, "stop", err.Error())
|
||
log.Error("runtime stop", "err", err)
|
||
return
|
||
}
|
||
|
||
// Reset stopping so the exit watcher (if it fires again on a future
|
||
// restart) treats future exits normally.
|
||
rec.stopping.Store(false)
|
||
|
||
d.sendInstanceState(req.InstanceId, panelv1.InstanceStatus_INSTANCE_STATUS_STOPPED, 0, "stopped")
|
||
d.replyOK(corrID)
|
||
log.Info("instance stopped (container retained for restart/backup)")
|
||
}
|
||
|
||
// handleRCON runs one operator-supplied RCON command against the instance's
|
||
// tracker client and returns the result via an RCONResult envelope carrying
|
||
// the same correlation_id so the controller can match it to the original
|
||
// ExecRCON request.
|
||
func (d *Dispatcher) handleRCON(ctx context.Context, corrID string, req *panelv1.RCONCommand) {
|
||
d.mu.Lock()
|
||
rec, ok := d.instances[req.InstanceId]
|
||
var tracker *state.Tracker
|
||
if ok {
|
||
tracker = rec.Tracker
|
||
}
|
||
d.mu.Unlock()
|
||
if !ok {
|
||
d.sendRCONResult(corrID, "", "not_found", "instance not on this target")
|
||
return
|
||
}
|
||
if tracker == nil {
|
||
d.sendRCONResult(corrID, "", "no_tracker", "instance has no RCON tracker (module may not declare rcon)")
|
||
return
|
||
}
|
||
// Use tracker.Exec so we transparently redial on dead connections —
|
||
// Palworld's RCON closes sockets after ~60s idle, so the first operator
|
||
// command after a quiet period would otherwise always fail with EOF.
|
||
out, err := tracker.Exec(ctx, req.Command)
|
||
if err != nil {
|
||
d.sendRCONResult(corrID, out, "exec", err.Error())
|
||
return
|
||
}
|
||
d.sendRCONResult(corrID, out, "", "")
|
||
}
|
||
|
||
func (d *Dispatcher) sendRCONResult(corrID, output, _code, errMsg string) {
|
||
d.sendEnv(&panelv1.AgentEnvelope{
|
||
CorrelationId: corrID,
|
||
SentAt: timestamppb.Now(),
|
||
Payload: &panelv1.AgentEnvelope_RconResult{
|
||
RconResult: &panelv1.RCONResult{Output: output, Error: errMsg},
|
||
},
|
||
})
|
||
}
|
||
|
||
func (d *Dispatcher) streamLogs(ctx context.Context, rec *instanceRecord) {
|
||
// Coalesce progress-style bursts (SteamCMD download ticks, preallocation,
|
||
// LGSM step banners) into at most one update per 200ms per "key". Non-
|
||
// progress lines go through immediately so errors and unique content are
|
||
// never delayed. The Tracker still sees every line for event matching.
|
||
//
|
||
// The whole thing runs in a restart loop. Reason: docker's
|
||
// ContainerLogs(--follow) can return cleanly (EOF / pipe closed) for
|
||
// reasons that have nothing to do with the container exiting — daemon
|
||
// hiccups, log driver buffer flushes, occasional plain weirdness. Before
|
||
// this loop, when StreamLogs returned without ctx.Err(), the goroutine
|
||
// just exited and the instance went silent until the next agent restart.
|
||
// Symptom in the wild: a server's pill stuck on "starting" because no
|
||
// log lines were reaching the controller's ring buffer, even though the
|
||
// container itself was fine. Live-debugged on V Rising 2026-04-29.
|
||
for {
|
||
if ctx.Err() != nil {
|
||
return
|
||
}
|
||
// Skip if the container is no longer running — watchExit owns the
|
||
// terminal "stopped" / "crashed" transition, no point chasing the
|
||
// stream of a dead container.
|
||
if rec.ContainerID == "" {
|
||
return
|
||
}
|
||
|
||
co := newLogCoalescer(200*time.Millisecond, func(stream runtime.LogStream, line string, at time.Time) {
|
||
d.sendEnv(&panelv1.AgentEnvelope{
|
||
SentAt: timestamppb.Now(),
|
||
Payload: &panelv1.AgentEnvelope_Log{Log: &panelv1.LogLine{
|
||
InstanceId: rec.InstanceID,
|
||
Stream: string(stream),
|
||
At: timestamppb.New(at),
|
||
Line: line,
|
||
}},
|
||
})
|
||
})
|
||
err := d.runtime.StreamLogs(ctx, rec.ContainerID, func(stream runtime.LogStream, line string, at time.Time) {
|
||
co.offer(stream, line, at)
|
||
// Snapshot the tracker under d.mu — handleStop/activate swap it
|
||
// concurrently. Uncontended mutex cost is noise next to the
|
||
// regex matching OnLogLine does.
|
||
d.mu.Lock()
|
||
tracker := rec.Tracker
|
||
d.mu.Unlock()
|
||
if tracker != nil {
|
||
tracker.OnLogLine(line)
|
||
}
|
||
// ark-sa hang-detection guardrail: fires on the engine's
|
||
// "!!!HANG DETECTED!!!" self-detection token. No-op for
|
||
// other modules. See hangguard.go.
|
||
d.arkHangGuardOnLogLine(rec, line)
|
||
})
|
||
co.stop()
|
||
|
||
if ctx.Err() != nil {
|
||
return
|
||
}
|
||
if err != nil {
|
||
d.log.Warn("log stream ended with error, will restart",
|
||
"instance_id", rec.InstanceID, "err", err)
|
||
} else {
|
||
d.log.Info("log stream returned cleanly, restarting",
|
||
"instance_id", rec.InstanceID)
|
||
}
|
||
// Brief backoff so a wedged docker daemon doesn't peg CPU on tight
|
||
// retry. 2s is plenty quick for "the operator is watching the
|
||
// pill," far short of human-perceptible silence.
|
||
select {
|
||
case <-ctx.Done():
|
||
return
|
||
case <-time.After(2 * time.Second):
|
||
}
|
||
}
|
||
}
|
||
|
||
// watchExit blocks on ContainerWait and reports the container's exit (or
|
||
// crash/auto-restart cycle) upstream. ctx-bound: cancelled by startExitWatch
|
||
// on re-activation, by handleStop, and by Shutdown, so a stale watcher can't
|
||
// hold a docker connection past its container lifetime.
|
||
func (d *Dispatcher) watchExit(ctx context.Context, rec *instanceRecord) {
|
||
for {
|
||
code, err := d.runtime.Wait(ctx, rec.ContainerID)
|
||
if ctx.Err() != nil {
|
||
return
|
||
}
|
||
if err != nil {
|
||
if strings.Contains(err.Error(), "context") {
|
||
return
|
||
}
|
||
d.log.Warn("wait error", "instance_id", rec.InstanceID, "err", err)
|
||
}
|
||
if rec.stopping.Load() {
|
||
return
|
||
}
|
||
|
||
// Docker's restart policy (unless-stopped / on-failure / always) may
|
||
// relaunch the container process after the previous one dies. If the
|
||
// container is back in "running" state by the time Wait returns, this
|
||
// is a crash-then-auto-restart cycle, not a terminal exit — report the
|
||
// crash + recovery to the UI and re-arm our streams against the new
|
||
// process. Otherwise the Console goes dead and the status pill gets
|
||
// stuck on "stopped" even though the server is live again.
|
||
state, ierr := d.runtime.InspectByName(ctx, "panel-"+rec.InstanceID)
|
||
if ierr == nil && state.Status == "running" {
|
||
d.log.Warn("container auto-restarted by docker restart policy",
|
||
"instance_id", rec.InstanceID, "prior_exit_code", code)
|
||
d.sendInstanceState(rec.InstanceID, panelv1.InstanceStatus_INSTANCE_STATUS_CRASHED, int32(code),
|
||
fmt.Sprintf("crashed (exit %d) — docker auto-restarted", code))
|
||
d.sendInstanceState(rec.InstanceID, panelv1.InstanceStatus_INSTANCE_STATUS_RUNNING, 0,
|
||
"running (recovered after crash)")
|
||
d.rearmStreamsAfterAutoRestart(rec)
|
||
continue
|
||
}
|
||
|
||
// Terminal exit — no auto-restart.
|
||
status := panelv1.InstanceStatus_INSTANCE_STATUS_STOPPED
|
||
if code != 0 {
|
||
status = panelv1.InstanceStatus_INSTANCE_STATUS_CRASHED
|
||
}
|
||
d.sendInstanceState(rec.InstanceID, status, int32(code), "exited")
|
||
return
|
||
}
|
||
}
|
||
|
||
// rearmStreamsAfterAutoRestart cancels the log + stats streams left pointing
|
||
// at the dead process and spawns fresh ones against the newly-restarted
|
||
// container. Called from watchExit on a crash-recovery cycle so the Console
|
||
// tab and stats graph don't go dark between the docker `die` and the next
|
||
// `start`. The RCON tracker (if any) is left to reconnect on its next poll —
|
||
// no explicit rearm needed.
|
||
func (d *Dispatcher) rearmStreamsAfterAutoRestart(rec *instanceRecord) {
|
||
d.mu.Lock()
|
||
if rec.LogCancel != nil {
|
||
rec.LogCancel()
|
||
}
|
||
logCtx, logCancel := context.WithCancel(context.Background())
|
||
rec.LogCancel = logCancel
|
||
d.mu.Unlock()
|
||
go d.streamLogs(logCtx, rec)
|
||
go d.streamStats(logCtx, rec)
|
||
}
|
||
|
||
func (d *Dispatcher) sendInstanceState(instanceID string, status panelv1.InstanceStatus, exitCode int32, detail string) {
|
||
// Record the last-sent status on the instance record (when one exists)
|
||
// so EmitAppState can detect a stale CRASHED that needs promotion.
|
||
d.mu.Lock()
|
||
rec := d.instances[instanceID]
|
||
d.mu.Unlock()
|
||
if rec != nil {
|
||
rec.lastStatus.Store(int32(status))
|
||
}
|
||
d.sendEnv(&panelv1.AgentEnvelope{
|
||
SentAt: timestamppb.Now(),
|
||
Payload: &panelv1.AgentEnvelope_InstanceState{
|
||
InstanceState: &panelv1.InstanceStateUpdate{
|
||
InstanceId: instanceID,
|
||
Status: status,
|
||
ExitCode: exitCode,
|
||
At: timestamppb.Now(),
|
||
Detail: detail,
|
||
},
|
||
},
|
||
})
|
||
}
|
||
|
||
func (d *Dispatcher) replyOK(corrID string) {
|
||
if corrID == "" {
|
||
return
|
||
}
|
||
d.sendEnv(&panelv1.AgentEnvelope{
|
||
CorrelationId: corrID,
|
||
SentAt: timestamppb.Now(),
|
||
Payload: &panelv1.AgentEnvelope_Result{
|
||
Result: &panelv1.CommandResult{ExitCode: 0},
|
||
},
|
||
})
|
||
}
|
||
|
||
func (d *Dispatcher) replyError(corrID, instanceID, code, msg string) {
|
||
if corrID == "" && instanceID == "" {
|
||
return
|
||
}
|
||
d.sendEnv(&panelv1.AgentEnvelope{
|
||
CorrelationId: corrID,
|
||
SentAt: timestamppb.Now(),
|
||
Payload: &panelv1.AgentEnvelope_Result{
|
||
Result: &panelv1.CommandResult{
|
||
ExitCode: 1,
|
||
Error: &panelv1.Error{
|
||
Code: code,
|
||
Message: msg,
|
||
At: timestamppb.Now(),
|
||
},
|
||
},
|
||
},
|
||
})
|
||
}
|
||
|
||
// assignNonCollidingHostPorts walks the manifest's ports list and, for
|
||
// each declared port, picks an actually-available host port — auto-
|
||
// bumping if the manifest default (or operator-supplied override) is
|
||
// already in use by another process or container. Mutates req.Ports
|
||
// in place: every declared manifest port ends up with an explicit
|
||
// HostPort entry, so downstream code (ResolveDocker, resolveRCONAddress)
|
||
// sees the same chosen port.
|
||
//
|
||
// Probe strategy: bind a real socket on 0.0.0.0:port (or 127.0.0.1:port
|
||
// for internal ports) for ~50ms; if the bind succeeds, the port is
|
||
// available — Docker's published-port machinery will succeed there too.
|
||
// Avoids guessing from /proc/net or shelling to ss/lsof.
|
||
//
|
||
// Bump direction: increment by 1 starting from the requested port up
|
||
// to +200. We intentionally don't pick from a far-flung high range —
|
||
// keeping ports clustered makes operator firewall configs easier and
|
||
// the typical "default + N siblings" workload fits in <50.
|
||
func assignNonCollidingHostPorts(manifest *modulepkg.Manifest, req *panelv1.InstanceCreate) error {
|
||
if manifest == nil || manifest.Runtime.Docker == nil {
|
||
return nil
|
||
}
|
||
// Build by-name lookup of any existing operator overrides so we can
|
||
// preserve them and only mutate when they collide.
|
||
overrideByName := map[string]*panelv1.PortMap{}
|
||
for i, ov := range req.Ports {
|
||
if ov != nil && ov.Name != "" {
|
||
overrideByName[ov.Name] = req.Ports[i]
|
||
}
|
||
}
|
||
// Track ports we've claimed during THIS create — sibling ports in the
|
||
// same module (e.g. ark-sa game/raw/rcon) all bind on this same call,
|
||
// so the probe needs to know about ports we're about to take.
|
||
claimed := map[string]bool{} // key: "tcp:7777" / "udp:7777"
|
||
// Reserve list for predictable rollback if something goes wrong mid-call.
|
||
for _, p := range manifest.Ports {
|
||
if p.Name == "" || p.Default == 0 {
|
||
continue
|
||
}
|
||
proto := strings.ToLower(p.Proto)
|
||
if proto != "tcp" && proto != "udp" {
|
||
proto = "tcp"
|
||
}
|
||
// Operator-supplied override wins as the starting point.
|
||
startHost := uint16(p.Default)
|
||
if ov, ok := overrideByName[p.Name]; ok && ov.HostPort != 0 {
|
||
startHost = uint16(ov.HostPort)
|
||
}
|
||
chosen, err := findFreePort(startHost, proto, p.Internal, claimed)
|
||
if err != nil {
|
||
return fmt.Errorf("port %q: %w", p.Name, err)
|
||
}
|
||
claimed[fmt.Sprintf("%s:%d", proto, chosen)] = true
|
||
// Update or insert the PortMap entry. ContainerPort stays at the
|
||
// manifest default (or operator override) — only HostPort is
|
||
// adjusted to dodge the collision.
|
||
containerPort := uint32(p.Default)
|
||
if ov, ok := overrideByName[p.Name]; ok && ov.ContainerPort != 0 {
|
||
containerPort = ov.ContainerPort
|
||
}
|
||
entry := &panelv1.PortMap{
|
||
Name: p.Name,
|
||
Proto: proto,
|
||
ContainerPort: containerPort,
|
||
HostPort: uint32(chosen),
|
||
Internal: p.Internal,
|
||
}
|
||
if existing, ok := overrideByName[p.Name]; ok {
|
||
*existing = *entry
|
||
} else {
|
||
req.Ports = append(req.Ports, entry)
|
||
overrideByName[p.Name] = req.Ports[len(req.Ports)-1]
|
||
}
|
||
}
|
||
return nil
|
||
}
|
||
|
||
// findFreePort scans upward from start, returning the first port that
|
||
// can be bound on the appropriate interface. claimed tracks ports we've
|
||
// already chosen earlier in the same call. Caps at +200 — keeps panel-
|
||
// assigned ports in a tight cluster for operator firewall sanity.
|
||
func findFreePort(start uint16, proto string, internal bool, claimed map[string]bool) (uint16, error) {
|
||
host := "0.0.0.0"
|
||
if internal {
|
||
host = "127.0.0.1"
|
||
}
|
||
for offset := 0; offset < 200; offset++ {
|
||
port := uint32(start) + uint32(offset)
|
||
if port > 65535 {
|
||
break
|
||
}
|
||
key := fmt.Sprintf("%s:%d", proto, port)
|
||
if claimed[key] {
|
||
continue
|
||
}
|
||
addr := fmt.Sprintf("%s:%d", host, port)
|
||
if proto == "udp" {
|
||
l, err := net.ListenPacket("udp", addr)
|
||
if err != nil {
|
||
continue
|
||
}
|
||
_ = l.Close()
|
||
return uint16(port), nil
|
||
}
|
||
l, err := net.Listen("tcp", addr)
|
||
if err != nil {
|
||
continue
|
||
}
|
||
_ = l.Close()
|
||
return uint16(port), nil
|
||
}
|
||
return 0, fmt.Errorf("no free %s port in range %d-%d", proto, start, uint32(start)+200)
|
||
}
|
||
|
||
// resolveRCONAddress determines the 127.0.0.1:port the Agent will use to
|
||
// reach the instance's RCON and pulls the password from config_values.
|
||
// Returns ("", "") if the manifest declares no RCON.
|
||
//
|
||
// Password source precedence:
|
||
// 1. config_values["rcon_password"] — explicit override
|
||
// 2. config_values[manifest.RCON.PasswordSecret] — module-declared mapping
|
||
// (operator-set value via the Config tab)
|
||
// 3. manifest.Runtime.Docker.Env[manifest.RCON.PasswordSecret] —
|
||
// module manifest's own default for the secret-named env var (this
|
||
// is the line that fixes ARK SA: SERVER_ADMIN_PASSWORD lives in
|
||
// env defaults, never gets seeded into ConfigValues, and the
|
||
// acekorneya image doesn't write it back to GameUserSettings.ini)
|
||
// 4. manifest.RCON.PasswordLiteral — explicit literal in manifest
|
||
// 5. "" (tracker will still attempt connect; auth may fail)
|
||
func resolveRCONAddress(manifest *modulepkg.Manifest, req *panelv1.InstanceCreate) (addr, password string) {
|
||
if manifest.RCON == nil || manifest.RCON.HostPort == "" {
|
||
return "", ""
|
||
}
|
||
portDecl := manifest.Port(manifest.RCON.HostPort)
|
||
if portDecl == nil {
|
||
return "", ""
|
||
}
|
||
hostPort := uint16(portDecl.Default)
|
||
for _, ov := range req.Ports {
|
||
if ov != nil && ov.Name == manifest.RCON.HostPort && ov.HostPort != 0 {
|
||
hostPort = uint16(ov.HostPort)
|
||
break
|
||
}
|
||
}
|
||
if req.ConfigValues != nil {
|
||
if p := req.ConfigValues["rcon_password"]; p != "" {
|
||
password = p
|
||
} else if key := manifest.RCON.PasswordSecret; key != "" {
|
||
password = req.ConfigValues[key]
|
||
}
|
||
}
|
||
// Fall back to the manifest's env-defaults map when the secret
|
||
// hasn't been overridden via ConfigValues. The agent merges env
|
||
// defaults into the container's env at create time, but it does
|
||
// NOT seed ConfigValues with them — so without this, RCON sees
|
||
// an empty password and dials get RST'd at auth.
|
||
if password == "" && manifest.RCON.PasswordSecret != "" && manifest.Runtime.Docker != nil {
|
||
if v, ok := manifest.Runtime.Docker.Env[manifest.RCON.PasswordSecret]; ok {
|
||
password = v
|
||
}
|
||
}
|
||
// Last resort: explicit literal in the manifest.
|
||
if password == "" {
|
||
password = manifest.RCON.PasswordLiteral
|
||
}
|
||
return fmt.Sprintf("127.0.0.1:%d", hostPort), password
|
||
}
|
||
|
||
// resolveBrowseableRoot returns the container-absolute path the file
|
||
// manager should treat as the DEFAULT root (first view). Precedence:
|
||
// browseable_roots[0], explicit browseable_root, first declared volume's
|
||
// container path, or "" to fall back to bind mount.
|
||
func resolveBrowseableRoot(manifest *modulepkg.Manifest) string {
|
||
if manifest.Runtime.Docker == nil {
|
||
return ""
|
||
}
|
||
d := manifest.Runtime.Docker
|
||
if len(d.BrowseableRoots) > 0 && d.BrowseableRoots[0].Path != "" {
|
||
return d.BrowseableRoots[0].Path
|
||
}
|
||
if d.BrowseableRoot != "" {
|
||
return d.BrowseableRoot
|
||
}
|
||
if len(d.Volumes) > 0 {
|
||
return d.Volumes[0].Container
|
||
}
|
||
return ""
|
||
}
|
||
|
||
// resolveBrowseableRoots returns every root the operator is allowed to
|
||
// browse. Guarantees at least one entry (synthesized from browseable_root
|
||
// if the module only declares the singular form).
|
||
func resolveBrowseableRoots(manifest *modulepkg.Manifest) []modulepkg.BrowseableRoot {
|
||
if manifest.Runtime.Docker == nil {
|
||
return nil
|
||
}
|
||
d := manifest.Runtime.Docker
|
||
if len(d.BrowseableRoots) > 0 {
|
||
return d.BrowseableRoots
|
||
}
|
||
if d.BrowseableRoot != "" {
|
||
return []modulepkg.BrowseableRoot{{Name: "default", Path: d.BrowseableRoot}}
|
||
}
|
||
if len(d.Volumes) > 0 {
|
||
return []modulepkg.BrowseableRoot{{Name: "default", Path: d.Volumes[0].Container}}
|
||
}
|
||
return nil
|
||
}
|
||
|
||
// mergeValuesAndSecrets generates any module-declared secrets that are
|
||
// flagged Generated, then layers user-supplied config_values on top
|
||
// (user wins). Returns a fresh map suitable for template rendering.
|
||
func mergeValuesAndSecrets(manifest *modulepkg.Manifest, userValues map[string]string) (map[string]string, error) {
|
||
secrets, err := modulepkg.GenerateSecrets(manifest)
|
||
if err != nil {
|
||
return nil, err
|
||
}
|
||
out := make(map[string]string, len(secrets)+len(userValues))
|
||
for k, v := range secrets {
|
||
out[k] = v
|
||
}
|
||
for k, v := range userValues {
|
||
if v != "" {
|
||
out[k] = v
|
||
}
|
||
}
|
||
return out, nil
|
||
}
|
||
|
||
// rtStatusToProto converts a Docker container State.Status to our
|
||
// InstanceStatus enum. Unknown states map to UNSPECIFIED.
|
||
func rtStatusToProto(s string) panelv1.InstanceStatus {
|
||
switch s {
|
||
case "running":
|
||
return panelv1.InstanceStatus_INSTANCE_STATUS_RUNNING
|
||
case "restarting":
|
||
return panelv1.InstanceStatus_INSTANCE_STATUS_STARTING
|
||
case "created":
|
||
return panelv1.InstanceStatus_INSTANCE_STATUS_STOPPED
|
||
case "paused":
|
||
return panelv1.InstanceStatus_INSTANCE_STATUS_RUNNING
|
||
case "exited":
|
||
return panelv1.InstanceStatus_INSTANCE_STATUS_STOPPED
|
||
case "dead":
|
||
return panelv1.InstanceStatus_INSTANCE_STATUS_CRASHED
|
||
default:
|
||
return panelv1.InstanceStatus_INSTANCE_STATUS_UNSPECIFIED
|
||
}
|
||
}
|
||
|
||
func short(id string) string {
|
||
if len(id) > 12 {
|
||
return id[:12]
|
||
}
|
||
return id
|
||
}
|