2026-03-14 20:34:37 +01:00
// OwnCord chat server — self-hosted, Windows-native.
// Build: go build -o chatserver.exe -ldflags "-s -w -X main.version=1.0.0" .
package main
import (
2026-08-15 20:50:47 +02:00
"bytes"
2026-03-14 20:34:37 +01:00
"context"
2026-08-15 20:50:47 +02:00
"crypto/tls"
"encoding/pem"
2026-03-14 20:34:37 +01:00
"errors"
"fmt"
2026-03-15 07:07:59 +01:00
"io"
stdlog "log"
2026-03-14 20:34:37 +01:00
"log/slog"
2026-03-15 07:07:59 +01:00
"net"
2026-03-14 20:34:37 +01:00
"net/http"
"os"
"os/signal"
2026-08-15 20:50:47 +02:00
"path/filepath"
2026-03-15 07:07:59 +01:00
"runtime"
2026-08-15 20:50:47 +02:00
"strconv"
2026-03-14 22:05:13 +01:00
"strings"
2026-03-14 20:34:37 +01:00
"syscall"
"time"
2026-08-15 20:50:47 +02:00
"gopkg.in/yaml.v3"
2026-03-19 05:32:40 +01:00
"github.com/owncord/server/admin"
2026-03-14 20:34:37 +01:00
"github.com/owncord/server/api"
"github.com/owncord/server/auth"
"github.com/owncord/server/config"
"github.com/owncord/server/db"
2026-08-15 20:50:47 +02:00
"github.com/owncord/server/diskutil"
2026-07-24 11:06:59 +02:00
"github.com/owncord/server/logctx"
2026-04-06 09:00:47 +00:00
"github.com/owncord/server/plugin"
2026-03-21 10:08:44 +01:00
"github.com/owncord/server/storage"
2026-04-06 09:00:47 +00:00
"github.com/owncord/server/telemetry"
"github.com/owncord/server/ws"
2026-03-14 20:34:37 +01:00
)
// version is overridden at build time via -ldflags "-X main.version=1.0.0".
var version = "dev"
func main () {
2026-08-15 20:50:47 +02:00
// `server healthcheck` probes the running instance's /health and exits
// 0/1. It exists for container healthchecks: the distroless image has no
// shell or curl, so the binary is its own probe.
if len ( os . Args ) > 1 && os . Args [ 1 ] == "healthcheck" {
os . Exit ( runHealthcheckCLI ())
}
2026-07-29 13:25:46 +02:00
// `server token ...` is a direct-to-DB CLI (mint/list/revoke API tokens) —
// handled before any server/logging setup so it stays quiet and standalone.
if len ( os . Args ) > 1 && os . Args [ 1 ] == "token" {
os . Exit ( runTokenCLI ( os . Args [ 2 :]))
}
2026-03-19 05:32:40 +01:00
// Create ring buffer for admin log viewer, then build a multi-handler
2026-07-31 15:41:57 +02:00
// that tees log records to both stdout and the ring buffer.
2026-03-19 05:32:40 +01:00
logBuf := admin . NewRingBuffer ( 2000 )
2026-07-31 15:41:57 +02:00
// levelVar controls both handlers' thresholds. It starts at INFO (the
2026-07-24 11:06:59 +02:00
// zero value) so early-startup logs are captured, then run() raises/lowers
2026-07-31 15:41:57 +02:00
// it once config.yaml / OWNCORD_LOGGING_LEVEL is loaded. The ring buffer
// shares it rather than hard-wiring DEBUG: with both sinks gated, Enabled
// returns false for suppressed levels and every gated Debug call across
// the server becomes a no-op instead of formatting a ring entry.
2026-07-24 11:06:59 +02:00
levelVar := new ( slog . LevelVar )
stdoutHandler := slog . NewTextHandler ( os . Stdout , & slog . HandlerOptions { Level : levelVar })
2026-07-31 15:41:57 +02:00
multiHandler := admin . NewMultiHandler ( stdoutHandler , logBuf , levelVar )
2026-07-24 11:06:59 +02:00
// logctx enriches records logged with a request/trace context (the
// ...Context slog variants) with req_id and, under -tags otel, trace_id.
log := slog . New ( logctx . New ( multiHandler ))
2026-03-19 05:32:40 +01:00
slog . SetDefault ( log )
2026-03-14 20:34:37 +01:00
2026-07-24 11:06:59 +02:00
if err := run ( log , logBuf , levelVar ); err != nil {
2026-03-17 04:11:04 +01:00
_ , _ = fmt . Fprintf ( os . Stderr , "\n [ERROR] %v\n\n" , err )
2026-03-14 20:34:37 +01:00
log . Error ( "server exited with error" , "error" , err )
os . Exit ( 1 )
}
}
// run is the real entrypoint — separated for testability.
2026-07-24 11:06:59 +02:00
func run ( log * slog . Logger , logBuf * admin . RingBuffer , levelVar * slog . LevelVar ) error {
2026-04-07 05:28:53 +00:00
// bgCtx is a cancellable context shared by all background goroutines
2026-07-23 17:03:52 +02:00
// (event persister, event pruner, plugin loader, maintenance loop).
2026-08-15 20:50:47 +02:00
//
// This first deferred bgCancel is only the LIFO backstop — because it is
// registered before `defer database.Close()`, it would otherwise run
// AFTER the database is closed, leaving background goroutines running
// through teardown. The persistence and maintenance blocks below register
// their own later (= earlier-running) defers that cancel bgCtx and JOIN
// their goroutines before the database closes.
2026-04-07 05:28:53 +00:00
bgCtx , bgCancel := context . WithCancel ( context . Background ())
defer bgCancel ()
2026-03-14 22:05:13 +01:00
// Clean up old binary from a previous update.
2026-03-21 11:59:14 +01:00
exePath , exeErr := os . Executable ()
if exeErr != nil {
log . Warn ( "failed to determine executable path" , "error" , exeErr )
} else {
2026-03-14 22:05:13 +01:00
oldPath := exePath + ".old"
if _ , statErr := os . Stat ( oldPath ); statErr == nil {
if rmErr := os . Remove ( oldPath ); rmErr != nil {
log . Warn ( "failed to remove old binary" , "path" , oldPath , "error" , rmErr )
} else {
log . Info ( "removed old binary from previous update" , "path" , oldPath )
}
}
}
2026-03-14 20:34:37 +01:00
// ── 1. Load configuration ──────────────────────────────────────────────
2026-07-31 15:41:57 +02:00
cfg , err := config . Load ( config . DefaultPath )
2026-03-14 20:34:37 +01:00
if err != nil {
return fmt . Errorf ( "loading config: %w" , err )
}
2026-07-31 15:41:57 +02:00
// Apply the configured log level. The admin panel's live log view (ring
// buffer) follows the same threshold — set logging.level to "debug" to
// capture debug records there.
2026-07-24 11:06:59 +02:00
if lvl , ok := config . ParseLevel ( cfg . Logging . Level ); ok {
levelVar . Set ( lvl )
} else {
log . Warn ( "unknown logging.level, keeping info" , "value" , cfg . Logging . Level )
}
2026-03-14 20:34:37 +01:00
// ── 2. Ensure data directory exists ────────────────────────────────────
2026-04-01 11:38:33 +02:00
if mkdirErr := os . MkdirAll ( cfg . Server . DataDir , 0 o750 ); mkdirErr != nil {
2026-03-14 20:34:37 +01:00
return fmt . Errorf ( "creating data dir %s: %w" , cfg . Server . DataDir , mkdirErr )
}
2026-08-15 20:50:47 +02:00
// Disk-space awareness: the database (WAL growth included), uploads,
// certs, and by default backups all live on this volume, and running it
// dry breaks several of them at once. Probe errors are ignored — unknown
// is not "full". /health repeats this check continuously at 256 MiB.
warnLowDisk ( log , "data dir" , cfg . Server . DataDir )
if cfg . Backup . Dir != "" && cfg . Backup . Dir != filepath . Join ( cfg . Server . DataDir , "backups" ) {
warnLowDisk ( log , "backup dir" , cfg . Backup . Dir )
}
2026-03-19 05:32:40 +01:00
// ── 3. TLS ────────────────────────────────────────────────────────────
tlsResult , err := auth . LoadOrGenerate ( cfg . TLS )
if err != nil {
return fmt . Errorf ( "configuring TLS: %w" , err )
}
tlsCfg := tlsResult . TLSConfig
// Print startup banner first so it appears above all init logs.
printBanner ( cfg , version , tlsCfg != nil )
// ── 4. Open database + run migrations ─────────────────────────────────
2026-07-19 08:15:58 +02:00
// SQLite is the only supported backend; the unfinished Postgres
// scaffolding (stubbed query layer, never wired into the runtime) was
// removed rather than completed.
if t := cfg . Database . Type ; t != "" && t != "sqlite" {
return fmt . Errorf ( "database.type=%q is not supported; set \"sqlite\" or omit it" , t )
2026-04-06 07:43:08 +00:00
}
2026-08-15 20:50:47 +02:00
database , err := db . OpenWithMaxReaders ( cfg . Database . Path , cfg . Database . MaxReaders )
2026-03-14 20:34:37 +01:00
if err != nil {
return fmt . Errorf ( "opening database: %w" , err )
}
2026-03-17 08:09:52 +01:00
defer database . Close () //nolint:errcheck
2026-03-14 20:34:37 +01:00
2026-08-14 14:49:27 +02:00
// The admin "Restore backup" handler needs the real database file path:
// without this, it falls back to a hardcoded "data/chatserver.db" and
// silently no-ops on any server with a configured database.path.
admin . SetDatabasePath ( cfg . Database . Path )
2026-08-15 20:50:47 +02:00
// Backup handlers and the scheduled-backup maintenance write to the
// configured backup directory (defaults to data/backups).
admin . SetBackupDir ( cfg . Backup . Dir )
2026-08-14 14:49:27 +02:00
2026-03-14 20:34:37 +01:00
if err := db . Migrate ( database ); err != nil {
return fmt . Errorf ( "running migrations: %w" , err )
}
2026-03-15 11:42:25 +01:00
2026-07-23 17:03:52 +02:00
// Clear stale state from a previous run or crash. Startup work — nothing
// to inherit a context from yet.
if err := database . ResetAllUserStatuses ( context . Background ()); err != nil {
2026-03-15 12:21:10 +01:00
log . Warn ( "failed to reset stale user statuses" , "error" , err )
} else {
log . Info ( "reset all user statuses to offline" )
}
2026-07-23 17:03:52 +02:00
if err := database . ClearAllVoiceStates ( context . Background ()); err != nil {
2026-03-15 11:42:25 +01:00
log . Warn ( "failed to clear stale voice states" , "error" , err )
} else {
log . Info ( "cleared stale voice states" )
}
2026-04-06 09:00:47 +00:00
// ── 4b. Telemetry (Phase B Step 8) ─────────────────────────────────────
2026-04-06 13:48:41 +00:00
// Init can return (nil, err) when the otel build-tag skeleton hasn't been
// finished wiring to the upstream SDK. Normalise to a no-op shutdown so
// the deferred closure never calls a nil function.
2026-04-06 09:00:47 +00:00
telemetryShutdown , telErr := telemetry . Init ( context . Background (), cfg . Telemetry )
if telErr != nil {
log . Warn ( "telemetry init failed; continuing without OpenTelemetry" , "error" , telErr )
}
2026-04-06 13:48:41 +00:00
if telemetryShutdown == nil {
telemetryShutdown = func ( context . Context ) error { return nil }
}
2026-04-06 09:00:47 +00:00
defer func () {
shutdownCtx , cancel := context . WithTimeout ( context . Background (), 5 * time . Second )
defer cancel ()
if err := telemetryShutdown ( shutdownCtx ); err != nil {
log . Warn ( "telemetry shutdown returned error" , "error" , err )
}
}()
2026-04-06 09:29:29 +00:00
// ── 5a. Construct plugin runtime BEFORE the router so the router can
// wire the live registry into the plugin admin handler. ────────────────
var pluginRegistry * plugin . Registry
if cfg . Plugins . Enabled {
registry , plugErr := plugin . NewRegistry ( plugin . Config {
Directory : cfg . Plugins . Directory ,
MaxMemoryMB : cfg . Plugins . MaxMemoryMB ,
CPUBudgetMs : cfg . Plugins . CPUBudgetMs ,
HTTPAllowlist : cfg . Plugins . HTTPAllowlist ,
2026-07-19 16:33:58 +00:00
Store : database ,
2026-04-06 09:29:29 +00:00
})
if plugErr != nil {
log . Warn ( "plugin runtime init failed; continuing without plugins" , "error" , plugErr )
} else {
pluginRegistry = registry
2026-04-07 05:28:53 +00:00
if err := registry . LoadAll ( bgCtx ); err != nil {
2026-04-06 09:29:29 +00:00
log . Warn ( "plugin loader: failed to scan directory" , "error" , err )
}
defer func () {
closeCtx , cancel := context . WithTimeout ( context . Background (), 5 * time . Second )
defer cancel ()
_ = registry . Close ( closeCtx )
}()
}
}
// ── 5b. Build HTTP router ──────────────────────────────────────────────
router , hub , routerCleanup := api . NewRouter ( cfg , database , version , logBuf , pluginRegistry )
2026-03-29 19:39:46 +02:00
defer routerCleanup ()
2026-08-14 10:05:40 +02:00
// Backstop for every early return below (serve error, ACME shutdown
// failure, etc.): hub.GracefulStop is the only caller of
// LiveKitProcess.Stop(), so skipping it orphans the companion
// livekit-server process and leaves the hub's dispatch goroutine
// running. gracefulOnce makes it idempotent alongside the explicit call
// on the normal shutdown path below.
defer hub . GracefulStop ()
2026-03-14 20:34:37 +01:00
2026-04-06 09:29:29 +00:00
// ── 5c. Wire event persistence (Phase B Step 7) ────────────────────────
2026-04-06 09:00:47 +00:00
if cfg . EventPersistence . Enabled && hub != nil {
2026-08-15 16:30:05 +02:00
seedHubReplayState ( bgCtx , hub , database , log )
2026-04-06 09:29:29 +00:00
2026-04-06 09:00:47 +00:00
persister := ws . NewEventPersister (
2026-07-19 16:33:58 +00:00
database ,
2026-04-06 09:00:47 +00:00
4096 ,
cfg . EventPersistence . BatchSize ,
time . Duration ( cfg . EventPersistence . BatchFlushMs ) * time . Millisecond ,
)
2026-04-07 05:28:53 +00:00
persister . Start ( bgCtx )
2026-04-06 09:00:47 +00:00
hub . SetEventPersister ( persister )
2026-07-19 16:33:58 +00:00
hub . SetEventStore ( database )
2026-04-06 09:00:47 +00:00
retention := time . Duration ( cfg . EventPersistence . RetentionHours ) * time . Hour
prunerInterval := time . Duration ( cfg . EventPersistence . PrunerIntervalMinutes ) * time . Minute
2026-08-15 20:50:47 +02:00
prunerDone := ws . StartEventPruner ( bgCtx , database , retention , prunerInterval )
2026-04-06 09:00:47 +00:00
defer func () {
stopCtx , stopCancel := context . WithTimeout ( context . Background (), 5 * time . Second )
defer stopCancel ()
persister . Stop ( stopCtx )
2026-08-15 20:50:47 +02:00
// Cancel the shared background context and JOIN the pruner before
// the (LIFO-later) database.Close defer runs, so no prune is still
// mid-query against a closing pool. Bounded: a stuck prune delays
// shutdown by at most the timeout, then Close proceeds anyway.
bgCancel ()
select {
case <- prunerDone :
case <- stopCtx . Done ():
log . Warn ( "event pruner did not exit before shutdown timeout" )
}
2026-04-06 09:00:47 +00:00
}()
}
2026-07-31 15:41:57 +02:00
// ── 5d. Async audit writer ─────────────────────────────────────────────
// Moves audit-log INSERTs off the request path: once the writer is
// installed, WriteAudit enqueues here and a background goroutine batches
// the writes (same shape as the event persister above). Paths that never
// install a writer — the token CLI, tests — keep the synchronous
// behavior. This defer is registered after `defer database.Close()` so
// LIFO ordering drains the queue before the database is torn down.
auditWriter := db . NewAuditWriter ( database , 1024 , 50 , 100 * time . Millisecond )
auditWriter . Start ( bgCtx )
database . SetAuditWriter ( auditWriter )
defer func () {
stopCtx , stopCancel := context . WithTimeout ( context . Background (), 5 * time . Second )
defer stopCancel ()
auditWriter . Stop ( stopCtx )
}()
2026-03-14 20:34:37 +01:00
// ── 6. Start server ────────────────────────────────────────────────────
addr := fmt . Sprintf ( ":%d" , cfg . Server . Port )
srv := & http . Server {
Addr : addr ,
Handler : router ,
TLSConfig : tlsCfg ,
ReadTimeout : 30 * time . Second ,
WriteTimeout : 30 * time . Second ,
IdleTimeout : 120 * time . Second ,
2026-03-15 07:07:59 +01:00
ErrorLog : stdlog . New ( io . Discard , "" , 0 ), // suppress TLS handshake noise
2026-03-14 20:34:37 +01:00
}
2026-03-15 07:07:59 +01:00
// ── 6b. ACME HTTP challenge server on :80 ─────────────────────────────
// When using Let's Encrypt (tls.mode: acme), an HTTP server on port 80
// is needed for HTTP-01 challenge validation and HTTP→HTTPS redirect.
var acmeSrv * http . Server
if tlsResult . HTTPHandler != nil {
acmeSrv = & http . Server {
Addr : ":80" ,
Handler : tlsResult . HTTPHandler ,
ReadTimeout : 10 * time . Second ,
WriteTimeout : 10 * time . Second ,
}
go func () {
log . Info ( "ACME HTTP challenge server starting on :80" )
if err := acmeSrv . ListenAndServe (); err != nil && ! errors . Is ( err , http . ErrServerClosed ) {
log . Error ( "ACME HTTP server error" , "error" , err )
}
}()
}
// ── 7. Background maintenance ────────────────────────────────────────
2026-03-21 10:08:44 +01:00
// Periodically purge expired sessions and orphaned attachments.
fileStorage , fileStorageErr := storage . New ( cfg . Upload . StorageDir , cfg . Upload . MaxSizeMB )
if fileStorageErr != nil {
log . Warn ( "failed to create file storage for maintenance; orphan file cleanup disabled" , "error" , fileStorageErr )
}
2026-03-15 07:07:59 +01:00
stopMaintenance := make ( chan struct {})
2026-08-15 20:50:47 +02:00
maintenanceDone := make ( chan struct {})
defer func () {
// Backstop for early returns below (see hub.GracefulStop defer above),
// and a bounded join so an in-flight tick (which can hold the writer —
// scheduled backups run VACUUM INTO) isn't still using the database
// while the LIFO-later Close defer tears it down.
close ( stopMaintenance )
select {
case <- maintenanceDone :
case <- time . After ( 5 * time . Second ):
log . Warn ( "maintenance loop did not exit before shutdown timeout" )
}
}()
2026-03-15 07:07:59 +01:00
go func () {
2026-08-15 20:50:47 +02:00
defer close ( maintenanceDone )
2026-03-15 07:07:59 +01:00
ticker := time . NewTicker ( 15 * time . Minute )
defer ticker . Stop ()
2026-03-21 11:59:14 +01:00
consecutiveFailures := 0
const maxConsecutiveFailures = 5
2026-03-15 07:07:59 +01:00
for {
select {
case <- ticker . C :
2026-03-21 11:59:14 +01:00
if consecutiveFailures >= maxConsecutiveFailures {
log . Error ( "maintenance loop: circuit breaker open, skipping tick" ,
"consecutive_failures" , consecutiveFailures )
// Reset after one skip to allow retry next tick.
consecutiveFailures = maxConsecutiveFailures - 1
continue
}
tickFailed := false
2026-07-23 17:03:52 +02:00
if err := database . DeleteExpiredSessions ( bgCtx ); err != nil {
2026-03-15 07:07:59 +01:00
log . Warn ( "failed to delete expired sessions" , "error" , err )
2026-03-21 11:59:14 +01:00
tickFailed = true
2026-03-15 07:07:59 +01:00
}
2026-03-21 10:08:44 +01:00
2026-08-15 20:50:47 +02:00
// Scheduled backups + retention pruning, driven by the
// backup_schedule / backup_retention admin settings.
if err := admin . MaintainBackups ( bgCtx , database ); err != nil {
log . Warn ( "backup maintenance failed" , "error" , err )
tickFailed = true
}
2026-03-21 10:08:44 +01:00
// Clean up orphaned attachments (uploaded but never linked to a message).
2026-08-07 21:20:48 +02:00
//
// Skipped entirely with no file storage configured: the delete is
// atomic (row goes the instant it's selected, by design — see
// db/attachment_queries.go), so with fileStorage nil the returned
// stored_as names — the only remaining handle on those blobs —
// would just be discarded and the files stranded on disk with no
// query left able to name them. Leaving the rows in place keeps
// them reclaimable once storage is available again.
if fileStorage != nil {
cutoff := time . Now (). Add ( - 1 * time . Hour )
orphanFiles , orphanErr := database . DeleteOrphanedAttachments ( bgCtx , cutoff )
if orphanErr != nil {
log . Warn ( "failed to delete orphaned attachments" , "error" , orphanErr )
tickFailed = true
} else if len ( orphanFiles ) > 0 {
// Best-effort file cleanup.
2026-03-21 10:08:44 +01:00
for _ , filename := range orphanFiles {
if delErr := fileStorage . Delete ( filename ); delErr != nil {
log . Warn ( "failed to delete orphan file" , "file" , filename , "error" , delErr )
}
}
2026-08-07 21:20:48 +02:00
log . Info ( "cleaned up orphaned attachments" , "count" , len ( orphanFiles ))
2026-03-21 10:08:44 +01:00
}
}
2026-03-21 11:59:14 +01:00
if tickFailed {
consecutiveFailures ++
} else {
consecutiveFailures = 0
}
2026-03-15 07:07:59 +01:00
case <- stopMaintenance :
return
}
}
}()
2026-03-14 20:34:37 +01:00
// Listen for OS signals for graceful shutdown.
ctx , stop := signal . NotifyContext ( context . Background (), os . Interrupt , syscall . SIGTERM )
defer stop ()
// Start serving in a goroutine.
serveErr := make ( chan error , 1 )
go func () {
log . Info ( "server starting" , "addr" , addr , "tls" , tlsCfg != nil , "version" , version )
2026-07-29 13:25:46 +02:00
for attempt := range 20 {
2026-03-19 03:53:40 +01:00
var listenErr error
2026-03-14 22:05:13 +01:00
if tlsCfg != nil {
listenErr = srv . ListenAndServeTLS ( "" , "" )
} else {
listenErr = srv . ListenAndServe ()
}
if listenErr != nil && ! errors . Is ( listenErr , http . ErrServerClosed ) {
// Check if it's an "address already in use" error (port not released yet from old process)
if attempt < 19 && isAddrInUse ( listenErr ) {
log . Warn ( "port in use, retrying..." , "attempt" , attempt + 1 , "error" , listenErr )
time . Sleep ( 500 * time . Millisecond )
continue
}
serveErr <- listenErr
}
break
2026-03-14 20:34:37 +01:00
}
close ( serveErr )
}()
// Wait for shutdown signal or server error.
select {
case err := <- serveErr :
if err != nil {
return fmt . Errorf ( "server error: %w" , err )
}
case <- ctx . Done ():
log . Info ( "shutdown signal received, draining connections (30s timeout)" )
}
// Graceful shutdown.
shutdownCtx , cancel := context . WithTimeout ( context . Background (), 30 * time . Second )
defer cancel ()
2026-03-15 07:07:59 +01:00
if acmeSrv != nil {
if err := acmeSrv . Shutdown ( shutdownCtx ); err != nil {
log . Warn ( "ACME HTTP server shutdown error" , "error" , err )
}
}
2026-08-15 20:50:47 +02:00
// Drain in-flight HTTP handlers FIRST: their broadcasts must still reach
// a live hub (and the event persister) or the frames vanish from the
// replay/event store across the restart. Shutdown does not wait on
// hijacked WebSocket connections, so the hub's own stop below is not
// delayed by connected clients — they get the restart notice right after
// the drain instead of right before it.
shutdownErr := srv . Shutdown ( shutdownCtx )
2026-03-19 03:53:40 +01:00
2026-08-15 20:50:47 +02:00
// Stop the WebSocket hub: notify clients, stop LiveKit, close all client
// connections. Threaded with the same 30s budget the operator was told
// about — the notice sleep and LiveKit stop count against it rather than
// extending it.
hub . GracefulStopContext ( shutdownCtx )
if shutdownErr != nil {
return fmt . Errorf ( "graceful shutdown: %w" , shutdownErr )
2026-03-14 20:34:37 +01:00
}
log . Info ( "server stopped cleanly" )
return nil
}
2026-08-15 20:50:47 +02:00
// runHealthcheckCLI probes the local server's /health endpoint and returns a
// process exit code: 0 healthy, 1 degraded or unreachable. /health answers
// 503 with a subsystem reason when the hub, database, or disk is unhealthy,
// so a container orchestrator's healthcheck surfaces those too.
func runHealthcheckCLI () int {
// Deliberately NOT config.Load: that writes a default config.yaml when
// none exists, and a probe must have no side effects. Peek at the file
// (and the env overrides) for just the values that shape the URL and the
// certificate pin.
port := 8443
scheme := "https"
certFile := "data/cert.pem"
tlsMode := ""
acmeDomain := ""
if raw , err := os . ReadFile ( config . DefaultPath ); err == nil {
var partial struct {
Server struct {
Port int `yaml:"port"`
} `yaml:"server"`
TLS struct {
Mode string `yaml:"mode"`
CertFile string `yaml:"cert_file"`
Domain string `yaml:"domain"`
} `yaml:"tls"`
}
if yaml . Unmarshal ( raw , & partial ) == nil {
if partial . Server . Port > 0 {
port = partial . Server . Port
}
tlsMode = partial . TLS . Mode
if partial . TLS . Mode == "off" {
scheme = "http"
}
if partial . TLS . CertFile != "" {
certFile = partial . TLS . CertFile
}
acmeDomain = partial . TLS . Domain
}
}
if env := os . Getenv ( "OWNCORD_SERVER_PORT" ); env != "" {
if p , err := strconv . Atoi ( env ); err == nil && p > 0 {
port = p
}
}
if env := os . Getenv ( "OWNCORD_TLS_MODE" ); env != "" {
tlsMode = env
if env == "off" {
scheme = "http"
}
}
if env := os . Getenv ( "OWNCORD_TLS_DOMAIN" ); env != "" {
acmeDomain = env
}
client := & http . Client {
Timeout : 5 * time . Second ,
Transport : & http . Transport {
TLSClientConfig : healthcheckTLSConfig ( tlsMode , certFile , acmeDomain ),
},
}
if port < 1 || port > 65535 {
port = 8443
}
resp , err := client . Get ( fmt . Sprintf ( "%s://127.0.0.1:%d/health" , scheme , port )) //nolint:gosec // G704: host is hardcoded loopback; only the port comes from the operator's own config
if err != nil {
fmt . Fprintln ( os . Stderr , "healthcheck: unreachable:" , err )
return 1
}
defer resp . Body . Close () //nolint:errcheck
if resp . StatusCode != http . StatusOK {
body , _ := io . ReadAll ( io . LimitReader ( resp . Body , 512 ))
fmt . Fprintf ( os . Stderr , "healthcheck: status %d: %s\n" , resp . StatusCode , body )
return 1
}
return 0
}
// healthcheckTLSConfig builds the probe's TLS config, per TLS mode:
//
// - acme: the served cert is CA-issued for the configured domain, so
// standard WebPKI verification works — but the probe dials 127.0.0.1, so
// ServerName must be overridden to the domain or hostname verification
// fails unconditionally and the probe reports a healthy server as down.
// A stale pre-ACME data/cert.pem must NOT be pinned in this mode either;
// the pin would mismatch the served ACME leaf forever.
// - self_signed / manual: the cert can never pass WebPKI (the generated one
// has no SANs and IsCA=false), so hostname/chain checks are replaced (not
// skipped) by pinning: the presented leaf must be byte-identical to the
// local cert file.
// - anything else with no readable local cert: plain WebPKI.
func healthcheckTLSConfig ( tlsMode , certFile , acmeDomain string ) * tls . Config {
if tlsMode == "acme" && acmeDomain != "" {
return & tls . Config { MinVersion : tls . VersionTLS12 , ServerName : acmeDomain }
}
pinned := loadPinnedCert ( certFile )
if pinned == nil {
return & tls . Config { MinVersion : tls . VersionTLS12 }
}
return & tls . Config {
MinVersion : tls . VersionTLS12 ,
// Chain/hostname verification is replaced by the exact-match pin
// below, which is strictly stronger for a cert we hold on disk.
// VerifyConnection (not VerifyPeerCertificate) so the pin also runs
// on resumed sessions (gosec G123).
InsecureSkipVerify : true , //nolint:gosec // G402: VerifyConnection below pins the exact local certificate
VerifyConnection : func ( cs tls . ConnectionState ) error {
if len ( cs . PeerCertificates ) == 0 {
return errors . New ( "healthcheck: server presented no certificate" )
}
if ! bytes . Equal ( cs . PeerCertificates [ 0 ]. Raw , pinned ) {
return errors . New ( "healthcheck: server certificate does not match " + certFile )
}
return nil
},
}
}
// loadPinnedCert reads the first PEM certificate block from path, returning
// its DER bytes, or nil when unavailable.
func loadPinnedCert ( path string ) [] byte {
raw , err := os . ReadFile ( path ) //nolint:gosec // G304: path is the operator's own configured cert file
if err != nil {
return nil
}
block , _ := pem . Decode ( raw )
if block == nil || block . Type != "CERTIFICATE" {
return nil
}
return block . Bytes
}
2026-08-15 16:30:05 +02:00
// seedHubReplayState restores the hub's monotonic seq counter from the
// persisted MAX(events.seq) so wrapped-payload seqs stay monotonic across
// restarts. Without this, the events table accumulates rows whose payload
// seqs reset to 1 after every restart, breaking the reconnect "events since
// last_seq" contract.
//
// It also forces every client resuming from at or before that restored seq
// onto the full-ready path for this boot. h.seq is persisted and restored
// here, but the paired watermark that tells a resuming client whether a
// channel-visibility change happened since its last_seq
// (visibilityChangeSeq) is in-memory only and always starts at 0 on a fresh
// process — see ws/hub_events.go's mustFullResync. Channel-visibility
// changes made to an offline client (RefreshChannelVisibility,
// revokeUnreadableChannels) are sent as targeted, unsequenced messages that
// are never written to the events table, so replay can never recover them.
// Without the MarkVisibilityChanged call below, a client resuming with
// last_seq at or before the pre-restart max sails straight through
// mustFullResync's zeroed watermark and can silently miss a visibility
// change it should have converged on.
func seedHubReplayState ( ctx context . Context , hub * ws . Hub , database * db . DB , log * slog . Logger ) {
maxSeq , seedErr := database . GetMaxEventSeq ( ctx )
if seedErr != nil {
log . Warn ( "event persistence: failed to read MAX(events.seq); starting hub seq from 0" , "error" , seedErr )
return
}
if maxSeq <= 0 {
return
}
hub . SeedSeq ( uint64 ( maxSeq ))
log . Info ( "event persistence: seeded hub seq from persisted events" , "seq" , maxSeq )
hub . MarkVisibilityChanged ()
}
2026-03-14 22:05:13 +01:00
// isAddrInUse checks if an error is an "address already in use" error.
func isAddrInUse ( err error ) bool {
return err != nil && ( strings . Contains ( err . Error (), "address already in use" ) || strings . Contains ( err . Error (), "Only one usage of each socket address" ))
}
2026-03-15 07:07:59 +01:00
// printBanner writes the startup banner to stderr (so it doesn't mix with
2026-07-24 11:06:59 +02:00
// the structured log output on stdout).
2026-03-15 07:07:59 +01:00
func printBanner ( cfg * config . Config , ver string , tls bool ) {
scheme := "http"
if tls {
scheme = "https"
}
localIP := getOutboundIP ()
port := cfg . Server . Port
baseURL := fmt . Sprintf ( "%s://%s:%d" , scheme , localIP , port )
adminURL := baseURL + "/admin"
tlsStatus := "disabled"
if tls {
tlsStatus = "enabled"
}
banner := fmt . Sprintf ( `
___ ____ _
/ _ \__ ___ __ / ___|___ _ __ __| |
| | | \ \ /\ / / '_ \| | / _ \| '__/ _` + "`" + ` |
| |_| |\ V V /| | | | |__| (_) | | | (_| |
\___/ \_/\_/ |_| |_|\____\___/|_| \__,_|
─────────────────────────────────────────────
Server %s
Version %s
TLS %s
Platform %s/%s
─────────────────────────────────────────────
API %s/api/v1/info
WebSocket %s/api/v1/ws
Admin %s
Health %s/health
─────────────────────────────────────────────
Press Ctrl+C to stop the server.
` , cfg . Server . Name , ver , tlsStatus , runtime . GOOS , runtime . GOARCH ,
baseURL , wsURL ( scheme , localIP , port ), adminURL , baseURL )
2026-03-17 04:11:04 +01:00
_ , _ = fmt . Fprint ( os . Stderr , banner )
2026-03-15 07:07:59 +01:00
}
// wsURL builds the WebSocket URL with the correct scheme.
func wsURL ( httpScheme , ip string , port int ) string {
ws := "ws"
if httpScheme == "https" {
ws = "wss"
}
return fmt . Sprintf ( "%s://%s:%d" , ws , ip , port )
}
2026-08-15 20:50:47 +02:00
// Free-space thresholds for the boot-time disk warning. /health uses its own
// (lower) continuous threshold; these only shape startup log noise.
const (
diskWarnBytes = 1 << 30 // 1 GiB — warn
diskCriticalBytes = 256 << 20 // 256 MiB — error
)
// warnLowDisk logs when the volume holding path is low on space. Probe
// failures (unsupported platform, missing dir) are silent — unknown ≠ full.
func warnLowDisk ( log * slog . Logger , label , path string ) {
free , err := diskutil . FreeBytes ( path )
if err != nil {
return
}
switch {
case free < diskCriticalBytes :
log . Error ( "disk space critically low — writes will start failing soon" ,
"volume" , label , "path" , path , "free_mb" , free >> 20 )
case free < diskWarnBytes :
log . Warn ( "disk space low" , "volume" , label , "path" , path , "free_mb" , free >> 20 )
}
}
2026-03-15 07:07:59 +01:00
// getOutboundIP returns the preferred outbound IP of this machine by dialing
// a known external address (no actual connection is made with UDP).
func getOutboundIP () string {
conn , err := net . Dial ( "udp" , "8.8.8.8:80" )
if err != nil {
return "localhost"
}
2026-03-17 08:09:52 +01:00
defer conn . Close () //nolint:errcheck
2026-04-03 23:18:06 +02:00
addr , ok := conn . LocalAddr ().( * net . UDPAddr )
if ! ok {
slog . Warn ( "getOutboundIP: unexpected LocalAddr type, falling back to localhost" ,
"type" , fmt . Sprintf ( "%T" , conn . LocalAddr ()))
return "localhost"
}
2026-03-15 07:07:59 +01:00
return addr . IP . String ()
}