chore(yuctl): restructure and make pretty (#485)

This commit is contained in:
Antoine Lecompte
2026-08-18 09:53:59 -04:00
committed by GitHub
parent 9091c1fc9d
commit b7bbe2e80e
88 changed files with 2458 additions and 2316 deletions
+1 -1
View File
@@ -14,7 +14,7 @@ node_modules
mise.local.toml
dist/
packages/yuctl/internal/bench/bench-agent-linux-amd64.gz
packages/yuctl/resticbench/bench-agent-linux-amd64.gz
packages/michael/michael
# k3d/Tilt dev stack — Helm builds subchart snapshots on the fly; the .dev/
+1 -1
View File
@@ -9,6 +9,6 @@ cd packages/yuctl
# `yuctl tools bench` is self-contained. Plain `go build` still works without
# the tag — the embedded agent is then absent and --agent-bin is required.
CGO_ENABLED=0 GOOS=linux GOARCH=amd64 go build -o ../../dist/bench-agent-linux-amd64 ./cmd/bench-agent
gzip -9 -n -c ../../dist/bench-agent-linux-amd64 > internal/bench/bench-agent-linux-amd64.gz
gzip -9 -n -c ../../dist/bench-agent-linux-amd64 > resticbench/bench-agent-linux-amd64.gz
go build -tags embedagent -o ../../dist/yuctl .
+89 -63
View File
@@ -12,12 +12,19 @@ references, never values).
## Conventions
Built to match `packages/michael`: `module yuctl`, Go 1.25, `main.go` +
`internal/<pkg>`, `aws-sdk-go-v2` for S3, `rs/zerolog` for logging. The one
documented divergence is **`spf13/cobra`** for the nested subcommand tree
(michael is a single-purpose HTTP server and stays stdlib-only; yuctl is a
multi-verb CLI). There is no Dockerfile — yuctl is an operator CLI, not a
deployed service.
`module yuctl`, Go 1.25, `aws-sdk-go-v2` for S3, `rs/zerolog` for logging —
matching `packages/michael` — with two documented divergences: **`spf13/cobra`**
for the nested subcommand tree (michael is a single-purpose HTTP server and
stays stdlib-only; yuctl is a multi-verb CLI), and a **flat package layout**
with no `internal/` (yuctl is an unpublishable standalone module nothing else
imports, so the boundary bought nothing but a path segment). There is no
Dockerfile — yuctl is an operator CLI, not a deployed service.
The command layer follows the gh/kubectl shape: `cli/` mirrors the command
tree one package per topic (`yuctl ceph …` → `cli/ceph`, `yuctl tools warp …`
→ `cli/tools/warp`), and every command receives a `cmdutil.Factory` carrying
the lazily-resolved shared dependencies (IO streams, selected context,
memoized topology, admin-api login) instead of re-deriving them per command.
## Build
@@ -32,16 +39,29 @@ Or directly: `cd packages/yuctl && go build -o ../../dist/yuctl .`
```
packages/yuctl/
main.go # entrypoint → cli.NewRootCmd().ExecuteContext
internal/
cli/ # cobra command tree (root, select, ceph, infra, users)
discovery/ # S3 state reader + stack enumeration + topology queries
state/ # discovery output contract structs + tfstate parsing
op/ # `op read` / ReadToTempFile (0600) wrapper
context/ # ~/.config/yuctl/context.json {partition,region,ceph_cluster}
k8s/ # talosctl upgrade wrapper
ceph/ # RGW/dashboard health probe
adminapi/ # CLI loopback login + Bearer admin-api client
main.go # yuctl entrypoint → cli.NewRootCmd().ExecuteContext
cmd/bench-agent/ # second binary: the remote bench agent
cli/ # cobra wiring, one package per topic, mirrors the command tree
root.go select.go login.go
ceph/ infra/ config/ features/
users/{allowlist,features,connections}/
tools/{bench,fleetbench,warp}/
cmdutil/ # Factory (IO, context, topology, admin login) + Confirm/OpenBrowser
ui/ # IOStreams, lipgloss theme, meter/sparkline widgets
fleet/ # shared fleet-tool engine: watch loop, history, parallel fan-out
warp/ # K8s-pod transport: warp runner fleet vs RGW
fleetbench/ # cloud-VM transport: restic client fleet vs michael
sshx/ # the one ssh/scp layer (multiplexing, retry, stdin secrets)
resticbench/ # bench agent + orchestrator + loadgen + restic runner
discovery/ # S3 state reader + stack enumeration + topology queries
state/ # discovery output contract structs + tfstate parsing
op/ # `op read` / ReadToTempFile (0600) wrapper
ctxstore/ # ~/.config/yuctl/context.json {partition,region,ceph_cluster}
talos/ # talosctl upgrade wrapper
cephhealth/ # RGW/dashboard health probe
adminapi/ # CLI loopback login + Bearer admin-api client
provider/ # cloud-VM providers (DO, Hetzner) for fleet-bench
do/ netdev/ # DigitalOcean plumbing; /proc/net/dev parsing
```
## Command tree
@@ -69,14 +89,14 @@ yuctl
├── bench restic e2e benchmark against michael, run from a mgmt host
│ ├── compare <a> <b> render before/after deltas from two results files
│ └── cleanup forget+prune every bench snapshot (timed)
├── bench-do restic client fleet on DigitalOcean droplets vs michael
│ ├── deploy create/converge the droplet fleet (project yucca-bench)
├── fleet-bench restic client fleet on cloud VMs (--provider do|hetzner) vs michael
│ ├── deploy create/converge the host fleet (project yucca-bench)
│ ├── start launch the per-client backup loops (graceful restart)
│ ├── status one-shot dashboard (throughput, transfer budget, clients)
│ ├── watch live dashboard, continuously sampled
│ ├── stop kill the load, collect + save the results JSON
│ ├── cleanup forget+prune every bench-do repo (from the droplets)
│ └── undeploy destroy the droplets and the ephemeral ssh key
│ ├── cleanup forget+prune every fleet-bench repo (from the hosts)
│ └── undeploy destroy the hosts and the ephemeral ssh key
└── warp S3 load test fleet against the region's RGW gateways
├── deploy create/converge hostNetwork runner pods on the workers
├── start launch the load (graceful restart; non-stop by default)
@@ -100,7 +120,7 @@ Global flags: `--log-level` (trace|debug|info|warn|error), `--log-format`
- **bucket fallback**: `ListObjectsV2` on `yucca-tf-state` under prefix
`yucca/`, keeping `*/terraform.tfstate` keys.
2. **Resolve live values** — `GetObject` each `terraform.tfstate` and parse
`.outputs.discovery.value` into `internal/state.Discovery`. Stacks with no
`.outputs.discovery.value` into `state.Discovery`. Stacks with no
`discovery` output (pre-contract) or no applied state are skipped, not fatal.
3. **Query** the merged `Topology` (`HasRegion`, `Kubernetes`, `CephClusters`,
`PrimaryRegion`, `RegionMeta`).
@@ -240,58 +260,62 @@ The ssh session stays open for the whole run (keepalives set); run multi-hour
benchmarks inside tmux. Pair the client numbers with the michael dashboard
(TTFB, connection churn, S3 client metrics) for the server-side view.
## `tools bench-do` — DigitalOcean restic client fleet
## `tools fleet-bench` — cloud-VM restic client fleet
Drives michael from the outside: N DigitalOcean droplets each running real
restic clients over the public internet — the actual external-user path (DNS,
edge, michael, RGW), unlike `bench` (mgmt host on the fabric) and `warp`
(in-cluster, straight at RGW). Fleet lifecycle mirrors `warp`.
Drives michael from the outside: N cloud VMs (`--provider do|hetzner`) each
running real restic clients over the public internet — the actual
external-user path (DNS, edge, michael, RGW), unlike `bench` (mgmt host on
the fabric) and `warp` (in-cluster, straight at RGW). Fleet lifecycle mirrors
`warp`; fleets are per provider × partition, so several providers can load
michael at once.
```bash
yuctl select prod@htz-fsn1
yuctl login
yuctl tools bench-do start --droplets 6 --clients-per-droplet 2 \
yuctl tools fleet-bench start --hosts 6 --clients-per-host 2 \
--obj-size 64MiB --duration 2h --label big-packs # auto-deploys (confirms cost first)
yuctl tools bench-do watch # live dashboard: Gbps, transfer budget bars, client loops
yuctl tools bench-do stop # kill the load, save bench-do-<label>-<ts>.json
yuctl tools bench-do cleanup # forget+prune the bench repos (while droplets exist)
yuctl tools bench-do undeploy # destroy droplets + the ephemeral ssh key
yuctl tools fleet-bench watch # live dashboard: Gbps, transfer budget bars, client loops
yuctl tools fleet-bench stop # kill the load, save fleet-bench-<label>-<ts>.json
yuctl tools fleet-bench cleanup # forget+prune the bench repos (while the hosts exist)
yuctl tools fleet-bench undeploy # destroy the hosts + the ephemeral ssh key
```
How it works:
- **Fleet**: droplets (`--droplets`, default 3 × `s-2vcpu-4gb`) are created via
the DO API (token from `op://yucca/do_api_token/password`, or
`$DIGITALOCEAN_TOKEN`), round-robined across `--do-region`
(default `fra1,ams3,lon1,nyc3`), tagged `yuctl-bench-do-<partition>`, and
filed under the **`yucca-bench`** project (created if missing). Deploy
prints the hourly cost and transfer pool and asks before creating anything
(`--yes` skips); it converges — rerunning reconciles the fleet to the
requested size and re-pushes binaries.
- **Fleet**: hosts (`--hosts`, default 3 of the provider's default size) are
created via the provider API (DO token from
`op://yucca/do_api_token/password` / `$DIGITALOCEAN_TOKEN`; Hetzner from
`op://yucca_tf_prod/HCLOUD_API_TOKEN/password` / `$HCLOUD_TOKEN`),
round-robined across `--region`, tagged `yuctl-bench-<provider>-<partition>`,
and filed under the **`yucca-bench`** project where the provider has the
concept. Deploy prints the hourly cost and transfer pool and asks before
creating anything (`--yes` skips); it converges — rerunning reconciles the
fleet to the requested size and re-pushes binaries.
- **SSH**: an **ephemeral ed25519 keypair per fleet** — generated on deploy,
registered via the API, private key + per-fleet known_hosts under
`~/.config/yuctl/bench-do/`, deleted on undeploy. No personal keys involved.
- **Clients**: one admin-api repository per client (`--clients-per-droplet`),
named `yucca-benchdo-…`, created on first start and reused across restarts;
restic URLs are re-minted on every start and travel to the droplet over ssh
stdin (never argv). Repo passwords live in the 0600 fleet state file — they
are the only way back into the repos, so `cleanup` before `undeploy`.
`~/.config/yuctl/bench-wide/` (legacy dir name), deleted on undeploy. No
personal keys involved.
- **Clients**: one admin-api repository per client (`--clients-per-host`),
named `yucca-benchdo-…` (legacy prefix), created on first start and reused
across restarts; restic URLs are re-minted on every start and travel to the
host over ssh stdin (never argv). Repo passwords live in the 0600 fleet
state file — they are the only way back into the repos, so `cleanup` before
`undeploy`.
- **Load**: the bench agent's **loadgen mode** runs detached (nohup) on each
droplet, looping seeded generate→backup cycles per client — fresh seed every
host, looping seeded generate→backup cycles per client — fresh seed every
cycle so nothing dedups — with `--obj-size` as the restic pack size
(4–128 MiB, what michael sees as object size), `--size` per-cycle dataset,
`--connections` rest.connections. `--duration` bounds the run (`0` =
non-stop until `stop`). Progress goes to a droplet-local status file that
`status`/`watch` sample over ssh alongside `/proc/net/dev`.
- **Transfer cap (important)**: DO droplets have a monthly outbound transfer
allowance (pooled; overage is billed per GiB) and a sustained restic load
can burn through it in hours. The agent tracks wire TX and **hard-stops the
droplet's load at the allowance** (counted conservatively as decimal TB);
`--max-transfer` overrides. The dashboard shows a per-droplet budget bar and
the fleet pool. If the fleet state is lost the cap is re-derived from the
droplet size — the load never runs uncapped.
- **Results**: `stop` collects each droplet's final status and writes a local
JSON (per-client cycles, post-dedup uploaded bytes, errors; per-droplet wire
(4–128 MiB, what michael sees as object size), `--cycle-size` per-cycle
dataset, `--connections` rest.connections. `--duration` bounds the run
(`0` = non-stop until `stop`). Progress goes to a host-local status file
that `status`/`watch` sample over ssh alongside `/proc/net/dev`.
- **Transfer cap (important)**: cloud VMs have a monthly outbound transfer
allowance (overage is billed per GiB) and a sustained restic load can burn
through it in hours. The agent tracks wire TX and **hard-stops the host's
load at the allowance**; `--max-transfer` overrides. The dashboard shows a
per-host budget bar and the fleet pool. If the fleet state is lost the cap
is re-derived from the host size — the load never runs uncapped.
- **Results**: `stop` collects each host's final status and writes a local
JSON (per-client cycles, post-dedup uploaded bytes, errors; per-host wire
TX) plus a rendered summary. Pair with the michael dashboards for the
server-side view.
@@ -329,7 +353,7 @@ How it works:
- **Runners**: `minio/warp` pods (2 per worker by default) on **hostNetwork**,
so the load rides the workers' bonded NICs with no CNI hop. CPU request is
derived from node allocatable. Manifests are `go:embed`ded templates
(`internal/warp/manifests/`), server-side-applied via client-go — no
(`fleet/warp/manifests/`), server-side-applied via client-go — no
kubectl dependency. Credentials are copied from the `yucca-michael` secret
(the same RGW svc user as the real data path).
- **Gateway roster**: the ceph cluster's `rgw_s3_endpoint` is resolved to its
@@ -362,12 +386,14 @@ about the fleet size, gateway count, or endpoints is hardcoded.
| `OP_BIN` | 1Password CLI binary | `op` |
| `YUCTL_ADMIN_API_URL` | admin-api base URL (`login`, `users`) | derived from discovery `api_endpoint` |
| `YUCTL_GRAFANA_URL` | grafana base (`users view-dashboard`) | `https://grafana.futostatus.com` |
| `DIGITALOCEAN_TOKEN` | DO API token (`tools bench-do`) | resolved via op |
| `DIGITALOCEAN_TOKEN` | DO API token (`tools fleet-bench`) | resolved via op |
| `YUCTL_DO_TOKEN_REF` | op ref for the DO token | `op://yucca/do_api_token/password` |
| `HCLOUD_TOKEN` | Hetzner API token (`tools fleet-bench`)| resolved via op |
| `YUCTL_HCLOUD_TOKEN_REF` | op ref for the Hetzner token | `op://yucca_tf_prod/HCLOUD_API_TOKEN/password` |
## Tests
`go test ./...` covers the load-bearing offline logic: discovery contract
parsing (`internal/state`) and stack-key/topology queries
(`internal/discovery`). The network/`op`/`talosctl`/admin-api paths are not unit
parsing (`state`) and stack-key/topology queries
(`discovery`). The network/`op`/`talosctl`/admin-api paths are not unit
tested (they need live infra).
@@ -6,14 +6,14 @@ import (
"os"
"path/filepath"
yctx "yuctl/internal/context"
"yuctl/ctxstore"
)
// tokenCachePath returns the per-partition token cache file inside the yuctl
// config dir. Partition-scoped because each partition has its own primary
// region / admin-api.
func tokenCachePath(partition string) (string, error) {
dir, err := yctx.Dir()
dir, err := ctxstore.Dir()
if err != nil {
return "", err
}
@@ -43,7 +43,7 @@ func LoadToken(partition string) (Token, error) {
// SaveToken persists a CLI session token at 0600.
func SaveToken(partition string, t Token) error {
dir, err := yctx.Dir()
dir, err := ctxstore.Dir()
if err != nil {
return err
}
@@ -95,6 +95,21 @@ func (c *Client) ListUsers(ctx context.Context, limit int) ([]User, error) {
return all, nil
}
// ResolveUserID turns an email into a user id. The admin-api has no lookup
// endpoint, so this lists every user and matches case-insensitively.
func (c *Client) ResolveUserID(ctx context.Context, email string) (string, error) {
users, err := c.ListUsers(ctx, 0)
if err != nil {
return "", err
}
for _, u := range users {
if strings.EqualFold(u.Email, email) {
return u.ID, nil
}
}
return "", fmt.Errorf("no user with email %q", email)
}
func (c *Client) listUserPage(ctx context.Context, cursor string, limit int) (*userPage, error) {
q := url.Values{}
if cursor != "" {
@@ -1,7 +1,7 @@
// Package ceph implements health checks against a region's Ceph cluster RGW /
// Package cephhealth implements health checks against a region's Ceph cluster RGW /
// dashboard endpoint, using the `rgw_s3_endpoint` + `health_cred_ref` fields of
// the ceph discovery payload (credential resolved via 1Password).
package ceph
package cephhealth
import (
"context"
@@ -12,8 +12,8 @@ import (
"strings"
"time"
"yuctl/internal/op"
"yuctl/internal/state"
"yuctl/op"
"yuctl/state"
)
// HealthResult summarizes a probe.
@@ -1,92 +1,90 @@
package cli
package ceph
import (
"fmt"
"sort"
"strings"
"github.com/spf13/cobra"
"yuctl/internal/ceph"
yctx "yuctl/internal/context"
"yuctl/cephhealth"
"yuctl/cmdutil"
"yuctl/ctxstore"
"yuctl/ui"
)
func newCephCmd() *cobra.Command {
func New(f *cmdutil.Factory) *cobra.Command {
cmd := &cobra.Command{
Use: "ceph",
Short: "Ceph cluster operations within the selected region",
}
cmd.AddCommand(newCephSelectCmd(), newCephGetCmd())
cmd.AddCommand(newSelectCmd(f), newGetCmd(f))
return cmd
}
func newCephSelectCmd() *cobra.Command {
func newSelectCmd(f *cmdutil.Factory) *cobra.Command {
return &cobra.Command{
Use: "select <name>",
Short: "Select a Ceph cluster within the active region",
Args: cobra.ExactArgs(1),
RunE: func(cmd *cobra.Command, args []string) error {
ctx := cmd.Context()
name := args[0]
c, err := requireContext()
c, err := f.Context()
if err != nil {
return err
}
topo, err := resolveTopology(ctx)
topo, err := f.Topology(cmd.Context())
if err != nil {
return err
}
clusters := topo.CephClusters(c.Partition, c.Region)
if _, ok := clusters[name]; !ok {
if _, ok := topo.CephClusters(c.Partition, c.Region)[name]; !ok {
return fmt.Errorf("unknown ceph cluster %q in %s@%s; known: %s",
name, c.Partition, c.Region, strings.Join(cephClusterNames(clusters), ", "))
name, c.Partition, c.Region, strings.Join(topo.CephClusterNames(c.Partition, c.Region), ", "))
}
c.CephCluster = name
if err := yctx.Save(c); err != nil {
if err := ctxstore.Save(c); err != nil {
return err
}
fmt.Fprintf(cmd.OutOrStdout(), "selected ceph cluster %q in %s@%s\n", name, c.Partition, c.Region)
fmt.Fprintf(f.IO.Out, "selected ceph cluster %q in %s@%s\n", name, c.Partition, c.Region)
return nil
},
}
}
func newCephGetCmd() *cobra.Command {
func newGetCmd(f *cmdutil.Factory) *cobra.Command {
cmd := &cobra.Command{
Use: "get",
Short: "Read Ceph cluster state",
}
cmd.AddCommand(newCephGetHealthCmd())
cmd.AddCommand(newGetHealthCmd(f))
return cmd
}
func newCephGetHealthCmd() *cobra.Command {
func newGetHealthCmd(f *cmdutil.Factory) *cobra.Command {
var insecure bool
c := &cobra.Command{
Use: "health",
Short: "Probe the selected Ceph cluster's RGW/dashboard health",
RunE: func(cmd *cobra.Command, args []string) error {
ctx := cmd.Context()
cc, err := requireContext()
cc, err := f.Context()
if err != nil {
return err
}
if cc.CephCluster == "" {
return fmt.Errorf("no ceph cluster selected; run `yuctl ceph select <name>` first")
}
topo, err := resolveTopology(ctx)
topo, err := f.Topology(ctx)
if err != nil {
return err
}
clusters := topo.CephClusters(cc.Partition, cc.Region)
cluster, ok := clusters[cc.CephCluster]
cluster, ok := topo.CephClusters(cc.Partition, cc.Region)[cc.CephCluster]
if !ok {
return fmt.Errorf("ceph cluster %q no longer present in %s@%s", cc.CephCluster, cc.Partition, cc.Region)
}
res, err := ceph.CheckHealth(ctx, cluster, insecure)
res, err := cephhealth.CheckHealth(ctx, cluster, insecure)
if err != nil {
return err
}
@@ -94,12 +92,12 @@ func newCephGetHealthCmd() *cobra.Command {
if !res.Healthy {
status = "UNHEALTHY"
}
out := cmd.OutOrStdout()
out := f.IO.Out
fmt.Fprintf(out, "ceph cluster: %s (%s@%s)\n", cc.CephCluster, cc.Partition, cc.Region)
fmt.Fprintf(out, "endpoint: %s\n", res.Endpoint)
fmt.Fprintf(out, "status: %s (http %d)\n", status, res.StatusCode)
if res.Detail != "" {
fmt.Fprintf(out, "detail: %s\n", truncate(res.Detail, 200))
fmt.Fprintf(out, "detail: %s\n", ui.Truncate(res.Detail, 200))
}
if !res.Healthy {
return fmt.Errorf("ceph health check failed")
@@ -110,20 +108,3 @@ func newCephGetHealthCmd() *cobra.Command {
c.Flags().BoolVar(&insecure, "insecure-skip-tls-verify", false, "skip TLS verification (self-signed RGW/dashboard certs)")
return c
}
func truncate(s string, n int) string {
if len(s) <= n {
return s
}
return s[:n] + "…"
}
// sortedKeys of the ceph cluster map.
func cephClusterNames[T any](m map[string]T) []string {
out := make([]string, 0, len(m))
for k := range m {
out = append(out, k)
}
sort.Strings(out)
return out
}
@@ -1,6 +1,11 @@
package cli
// Package config implements `yuctl config`: scoped mutable configuration
// (global / per-site / per-cluster overrides of restic_pack_size_mib,
// connections_math, ...) stored by yucca-admin-api and served to clients via
// GET /api/meta.
package config
import (
"context"
"encoding/json"
"fmt"
"maps"
@@ -10,22 +15,16 @@ import (
"github.com/spf13/cobra"
"yuctl/internal/adminapi"
"yuctl/adminapi"
"yuctl/cmdutil"
)
// newConfigCmd builds the `config` subtree: scoped mutable configuration
// (global / per-site / per-cluster overrides of restic_pack_size_mib,
// connections_math, ...) stored by yucca-admin-api and served to clients via
// GET /api/meta.
func newConfigCmd() *cobra.Command {
func New(f *cmdutil.Factory) *cobra.Command {
cmd := &cobra.Command{
Use: "config",
Short: "Manage scoped config overrides (global / site / cluster) via yucca-admin-api",
}
cmd.AddCommand(newConfigListCmd())
cmd.AddCommand(newConfigGetCmd())
cmd.AddCommand(newConfigSetCmd())
cmd.AddCommand(newConfigUnsetCmd())
cmd.AddCommand(newListCmd(f), newGetCmd(f), newSetCmd(f), newUnsetCmd(f))
return cmd
}
@@ -52,14 +51,14 @@ func (s *scopeFlags) scope() string {
}
}
func newConfigListCmd() *cobra.Command {
flags := &adminFlags{}
func newListCmd(f *cmdutil.Factory) *cobra.Command {
admin := &cmdutil.AdminFlags{}
c := &cobra.Command{
Use: "list",
Short: "List every settings scope and its overrides",
Args: cobra.NoArgs,
RunE: func(cmd *cobra.Command, _ []string) error {
client, partition, err := flags.allowlistClient(cmd)
client, partition, err := admin.Client(cmd.Context(), f)
if err != nil {
return err
}
@@ -68,7 +67,7 @@ func newConfigListCmd() *cobra.Command {
return err
}
w := tabwriter.NewWriter(cmd.OutOrStdout(), 0, 2, 2, ' ', 0)
w := tabwriter.NewWriter(f.IO.Out, 0, 2, 2, ' ', 0)
fmt.Fprintln(w, "SCOPE\tVALUE\tUPDATED")
for _, e := range entries {
value, err := json.Marshal(e.Value)
@@ -78,52 +77,45 @@ func newConfigListCmd() *cobra.Command {
fmt.Fprintf(w, "%s\t%s\t%s\n", e.Scope, value, e.UpdatedAt)
}
w.Flush()
fmt.Fprintf(cmd.ErrOrStderr(), "\n%d scope(s) in partition %s\n", len(entries), partition)
fmt.Fprintf(f.IO.Err, "\n%d scope(s) in partition %s\n", len(entries), partition)
return nil
},
}
flags.register(c)
admin.Register(c)
return c
}
func newConfigGetCmd() *cobra.Command {
flags := &adminFlags{}
func newGetCmd(f *cmdutil.Factory) *cobra.Command {
admin := &cmdutil.AdminFlags{}
scope := &scopeFlags{}
c := &cobra.Command{
Use: "get",
Short: "Print the overrides stored for a scope",
Args: cobra.NoArgs,
RunE: func(cmd *cobra.Command, _ []string) error {
client, _, err := flags.allowlistClient(cmd)
client, _, err := admin.Client(cmd.Context(), f)
if err != nil {
return err
}
entries, err := client.ListSettings(cmd.Context())
value, err := currentValue(cmd.Context(), client, scope.scope())
if err != nil {
return err
}
value := map[string]any{}
for _, e := range entries {
if e.Scope == scope.scope() {
value = e.Value
break
}
}
out, err := json.MarshalIndent(value, "", " ")
if err != nil {
return err
}
fmt.Fprintln(cmd.OutOrStdout(), string(out))
fmt.Fprintln(f.IO.Out, string(out))
return nil
},
}
scope.register(c)
flags.register(c)
admin.Register(c)
return c
}
func newConfigSetCmd() *cobra.Command {
flags := &adminFlags{}
func newSetCmd(f *cmdutil.Factory) *cobra.Command {
admin := &cmdutil.AdminFlags{}
scope := &scopeFlags{}
c := &cobra.Command{
Use: "set key=value [key=value ...]",
@@ -144,13 +136,13 @@ func newConfigSetCmd() *cobra.Command {
updates[key] = parseValue(raw)
}
client, _, err := flags.allowlistClient(cmd)
ctx := cmd.Context()
client, _, err := admin.Client(ctx, f)
if err != nil {
return err
}
ctx := cmd.Context()
value, err := currentValue(cmd, client, scope.scope())
value, err := currentValue(ctx, client, scope.scope())
if err != nil {
return err
}
@@ -164,17 +156,17 @@ func newConfigSetCmd() *cobra.Command {
if err != nil {
return err
}
fmt.Fprintf(cmd.OutOrStdout(), "%s = %s\n", entry.Scope, out)
fmt.Fprintf(f.IO.Out, "%s = %s\n", entry.Scope, out)
return nil
},
}
scope.register(c)
flags.register(c)
admin.Register(c)
return c
}
func newConfigUnsetCmd() *cobra.Command {
flags := &adminFlags{}
func newUnsetCmd(f *cmdutil.Factory) *cobra.Command {
admin := &cmdutil.AdminFlags{}
scope := &scopeFlags{}
var all bool
c := &cobra.Command{
@@ -185,21 +177,21 @@ func newConfigUnsetCmd() *cobra.Command {
return fmt.Errorf("pass either keys to unset or --all")
}
client, _, err := flags.allowlistClient(cmd)
ctx := cmd.Context()
client, _, err := admin.Client(ctx, f)
if err != nil {
return err
}
ctx := cmd.Context()
if all {
if err := client.DeleteSettings(ctx, scope.scope()); err != nil {
return err
}
fmt.Fprintf(cmd.OutOrStdout(), "%s cleared\n", scope.scope())
fmt.Fprintf(f.IO.Out, "%s cleared\n", scope.scope())
return nil
}
value, err := currentValue(cmd, client, scope.scope())
value, err := currentValue(ctx, client, scope.scope())
if err != nil {
return err
}
@@ -211,7 +203,7 @@ func newConfigUnsetCmd() *cobra.Command {
if err := client.DeleteSettings(ctx, scope.scope()); err != nil {
return err
}
fmt.Fprintf(cmd.OutOrStdout(), "%s cleared\n", scope.scope())
fmt.Fprintf(f.IO.Out, "%s cleared\n", scope.scope())
return nil
}
@@ -223,18 +215,18 @@ func newConfigUnsetCmd() *cobra.Command {
if err != nil {
return err
}
fmt.Fprintf(cmd.OutOrStdout(), "%s = %s\n", entry.Scope, out)
fmt.Fprintf(f.IO.Out, "%s = %s\n", entry.Scope, out)
return nil
},
}
c.Flags().BoolVar(&all, "all", false, "delete the whole scope instead of individual keys")
scope.register(c)
flags.register(c)
admin.Register(c)
return c
}
func currentValue(cmd *cobra.Command, client *adminapi.Client, scope string) (map[string]any, error) {
entries, err := client.ListSettings(cmd.Context())
func currentValue(ctx context.Context, client *adminapi.Client, scope string) (map[string]any, error) {
entries, err := client.ListSettings(ctx)
if err != nil {
return nil, err
}
@@ -1,4 +1,6 @@
package cli
// Package features implements `yuctl features`: the feature-flag registry and
// fleet-wide operations (per-user set/clear lives under `users features`).
package features
import (
"fmt"
@@ -7,24 +9,21 @@ import (
"github.com/spf13/cobra"
"yuctl/internal/adminapi"
"yuctl/adminapi"
"yuctl/cmdutil"
)
// newFeaturesCmd builds the `features` subtree: the feature-flag registry and
// fleet-wide operations (per-user set/clear lives under `users features`).
func newFeaturesCmd() *cobra.Command {
func New(f *cmdutil.Factory) *cobra.Command {
cmd := &cobra.Command{
Use: "features",
Short: "Feature-flag registry and batch enrollment",
}
cmd.AddCommand(newFeaturesListCmd())
cmd.AddCommand(newFeaturesUsersCmd())
cmd.AddCommand(newFeaturesEnableBatchCmd())
cmd.AddCommand(newListCmd(f), newUsersCmd(f), newEnableBatchCmd(f))
return cmd
}
func printFeatureUsers(cmd *cobra.Command, items []adminapi.FeatureUser) {
w := tabwriter.NewWriter(cmd.OutOrStdout(), 0, 2, 2, ' ', 0)
func printFeatureUsers(f *cmdutil.Factory, items []adminapi.FeatureUser) {
w := tabwriter.NewWriter(f.IO.Out, 0, 2, 2, ' ', 0)
fmt.Fprintln(w, "EMAIL\tVALUE\tSET BY\tREASON\tUPDATED")
for _, u := range items {
reason := ""
@@ -36,14 +35,14 @@ func printFeatureUsers(cmd *cobra.Command, items []adminapi.FeatureUser) {
w.Flush()
}
func newFeaturesListCmd() *cobra.Command {
flags := &adminFlags{}
func newListCmd(f *cmdutil.Factory) *cobra.Command {
admin := &cmdutil.AdminFlags{}
c := &cobra.Command{
Use: "list",
Short: "List the feature-flag registry with override counts",
Args: cobra.NoArgs,
RunE: func(cmd *cobra.Command, _ []string) error {
client, _, err := flags.allowlistClient(cmd)
client, _, err := admin.Client(cmd.Context(), f)
if err != nil {
return err
}
@@ -52,27 +51,27 @@ func newFeaturesListCmd() *cobra.Command {
return err
}
w := tabwriter.NewWriter(cmd.OutOrStdout(), 0, 2, 2, ' ', 0)
w := tabwriter.NewWriter(f.IO.Out, 0, 2, 2, ' ', 0)
fmt.Fprintln(w, "FLAG\tDEFAULT\tSTAGE\tOVERRIDES\tSINCE\tDESCRIPTION")
for _, f := range featureFlags {
fmt.Fprintf(w, "%s\t%t\t%s\t%d\t%s\t%s\n", f.Key, f.Default, f.Stage, f.Overrides, f.Since, f.Description)
for _, ff := range featureFlags {
fmt.Fprintf(w, "%s\t%t\t%s\t%d\t%s\t%s\n", ff.Key, ff.Default, ff.Stage, ff.Overrides, ff.Since, ff.Description)
}
w.Flush()
return nil
},
}
flags.register(c)
admin.Register(c)
return c
}
func newFeaturesUsersCmd() *cobra.Command {
flags := &adminFlags{}
func newUsersCmd(f *cmdutil.Factory) *cobra.Command {
admin := &cmdutil.AdminFlags{}
c := &cobra.Command{
Use: "users <flag>",
Short: "List users holding an override for a flag",
Args: cobra.ExactArgs(1),
RunE: func(cmd *cobra.Command, args []string) error {
client, _, err := flags.allowlistClient(cmd)
client, _, err := admin.Client(cmd.Context(), f)
if err != nil {
return err
}
@@ -80,17 +79,17 @@ func newFeaturesUsersCmd() *cobra.Command {
if err != nil {
return err
}
printFeatureUsers(cmd, items)
fmt.Fprintf(cmd.ErrOrStderr(), "\n%d override(s) for %s\n", len(items), args[0])
printFeatureUsers(f, items)
fmt.Fprintf(f.IO.Err, "\n%d override(s) for %s\n", len(items), args[0])
return nil
},
}
flags.register(c)
admin.Register(c)
return c
}
func newFeaturesEnableBatchCmd() *cobra.Command {
flags := &adminFlags{}
func newEnableBatchCmd(f *cmdutil.Factory) *cobra.Command {
admin := &cmdutil.AdminFlags{}
c := &cobra.Command{
Use: "enable-batch <flag> <count>",
Short: "Enable a flag for the oldest <count> users without an override",
@@ -100,7 +99,7 @@ func newFeaturesEnableBatchCmd() *cobra.Command {
if err != nil || count < 1 {
return fmt.Errorf("count must be a positive integer")
}
client, _, err := flags.allowlistClient(cmd)
client, _, err := admin.Client(cmd.Context(), f)
if err != nil {
return err
}
@@ -108,11 +107,11 @@ func newFeaturesEnableBatchCmd() *cobra.Command {
if err != nil {
return err
}
printFeatureUsers(cmd, enabled)
fmt.Fprintf(cmd.ErrOrStderr(), "\nenabled %s for %d user(s)\n", args[0], len(enabled))
printFeatureUsers(f, enabled)
fmt.Fprintf(f.IO.Err, "\nenabled %s for %d user(s)\n", args[0], len(enabled))
return nil
},
}
flags.register(c)
admin.Register(c)
return c
}
@@ -1,36 +1,35 @@
package cli
package infra
import (
"bufio"
"fmt"
"os"
"strings"
"github.com/rs/zerolog/log"
"github.com/spf13/cobra"
"yuctl/internal/k8s"
"yuctl/cmdutil"
"yuctl/talos"
)
func newInfraCmd() *cobra.Command {
func New(f *cmdutil.Factory) *cobra.Command {
cmd := &cobra.Command{
Use: "infra",
Short: "Infrastructure / node operations for the active region",
}
cmd.AddCommand(newInfraTalosCmd())
cmd.AddCommand(newTalosCmd(f))
return cmd
}
func newInfraTalosCmd() *cobra.Command {
func newTalosCmd(f *cmdutil.Factory) *cobra.Command {
cmd := &cobra.Command{
Use: "talos",
Short: "Talos node operations",
}
cmd.AddCommand(newInfraTalosUpgradeCmd())
cmd.AddCommand(newTalosUpgradeCmd(f))
return cmd
}
func newInfraTalosUpgradeCmd() *cobra.Command {
func newTalosUpgradeCmd(f *cmdutil.Factory) *cobra.Command {
var (
image string
dryRun bool
@@ -44,11 +43,11 @@ func newInfraTalosUpgradeCmd() *cobra.Command {
"prompt unless --yes; use --dry-run to print the commands without running.",
RunE: func(cmd *cobra.Command, args []string) error {
ctx := cmd.Context()
cc, err := requireContext()
cc, err := f.Context()
if err != nil {
return err
}
topo, err := resolveTopology(ctx)
topo, err := f.Topology(ctx)
if err != nil {
return err
}
@@ -57,7 +56,7 @@ func newInfraTalosUpgradeCmd() *cobra.Command {
return fmt.Errorf("no kubernetes/talos discovery payload for %s@%s", cc.Partition, cc.Region)
}
out := cmd.OutOrStdout()
out := f.IO.Out
fmt.Fprintf(out, "Talos upgrade target: %s@%s (cluster %s)\n", cc.Partition, cc.Region, kube.ClusterName)
fmt.Fprintf(out, "control-plane nodes: %s\n", strings.Join(kube.CPNodeIPs, ", "))
if image != "" {
@@ -65,13 +64,13 @@ func newInfraTalosUpgradeCmd() *cobra.Command {
}
if !dryRun && !yes {
if !confirm(fmt.Sprintf("Upgrade Talos on %d node(s) in %s@%s?", len(kube.CPNodeIPs), cc.Partition, cc.Region)) {
if !cmdutil.Confirm(f.IO, fmt.Sprintf("Upgrade Talos on %d node(s) in %s@%s?", len(kube.CPNodeIPs), cc.Partition, cc.Region)) {
fmt.Fprintln(out, "aborted")
return nil
}
}
return k8s.TalosUpgrade(ctx, *kube, k8s.UpgradeOptions{
return talos.Upgrade(ctx, *kube, talos.UpgradeOptions{
Image: image,
DryRun: dryRun,
}, log.Logger)
@@ -82,15 +81,3 @@ func newInfraTalosUpgradeCmd() *cobra.Command {
c.Flags().BoolVar(&yes, "yes", false, "skip the confirmation prompt")
return c
}
// confirm prompts the operator for a yes/no answer on stdin, defaulting to no.
func confirm(prompt string) bool {
fmt.Fprintf(os.Stderr, "%s [y/N]: ", prompt)
reader := bufio.NewReader(os.Stdin)
line, err := reader.ReadString('\n')
if err != nil {
return false
}
answer := strings.ToLower(strings.TrimSpace(line))
return answer == "y" || answer == "yes"
}
+51
View File
@@ -0,0 +1,51 @@
package cli
import (
"fmt"
"time"
"github.com/spf13/cobra"
"yuctl/cmdutil"
)
func newLoginCmd(f *cmdutil.Factory) *cobra.Command {
admin := &cmdutil.AdminFlags{}
c := &cobra.Command{
Use: "login",
Short: "Log in to the partition's yucca-admin-api via the browser",
Long: "Authenticate against the selected partition's admin-api (primary region):\n" +
"opens the admin-api CLI login in your browser, receives a one-time code on a\n" +
"127.0.0.1 listener, exchanges it for a 24h session JWT, and caches it at\n" +
"${XDG_CONFIG_HOME:-~/.config}/yuctl/admin-token-<partition>.json.",
Args: cobra.NoArgs,
RunE: func(cmd *cobra.Command, _ []string) error {
ctx := cmd.Context()
cc, err := f.Context()
if err != nil {
return err
}
topo, err := f.Topology(ctx)
if err != nil {
return err
}
client, token, err := admin.Login(ctx, f, cc, topo)
if err != nil {
return err
}
// Round-trip the session so "login" only succeeds when the API
// actually accepts the token.
sub, err := client.GetAuth(ctx)
if err != nil {
return err
}
fmt.Fprintf(f.IO.Out, "logged in to %s as %s (expires %s)\n",
cc.Partition, sub, token.Expiry.Local().Format(time.RFC3339))
return nil
},
}
admin.Register(c)
return c
}
@@ -1,7 +1,6 @@
// Package cli wires the yuctl command tree (spf13/cobra). cobra is a documented
// divergence from michael's stdlib-only style, justified by yuctl's nested
// subcommand surface (select / ceph / infra / users). Logging matches michael
// (rs/zerolog).
// subcommand surface. Logging matches michael (rs/zerolog).
package cli
import (
@@ -14,7 +13,16 @@ import (
"github.com/rs/zerolog/log"
"github.com/spf13/cobra"
"yuctl/internal/discovery"
cephcmd "yuctl/cli/ceph"
configcmd "yuctl/cli/config"
featurescmd "yuctl/cli/features"
infracmd "yuctl/cli/infra"
toolscmd "yuctl/cli/tools"
userscmd "yuctl/cli/users"
"yuctl/cmdutil"
"yuctl/ctxstore"
"yuctl/discovery"
"yuctl/ui"
)
var (
@@ -23,8 +31,9 @@ var (
flagRefreshDiscovery bool
)
// NewRootCmd builds the root command and registers every subcommand.
func NewRootCmd() *cobra.Command {
f := newFactory()
root := &cobra.Command{
Use: "yuctl",
Short: "yucca operations CLI",
@@ -44,18 +53,46 @@ func NewRootCmd() *cobra.Command {
"bypass the cached topology and re-read Terraform state from S3")
root.AddCommand(
newSelectCmd(),
newLoginCmd(),
newCephCmd(),
newInfraCmd(),
newUsersCmd(),
newConfigCmd(),
newFeaturesCmd(),
newToolsCmd(),
newSelectCmd(f),
newLoginCmd(f),
cephcmd.New(f),
infracmd.New(f),
userscmd.New(f),
configcmd.New(f),
featurescmd.New(f),
toolscmd.New(f),
)
return root
}
// Topology is memoized: several layers of one command may need it (host
// resolution, admin URL derivation) but state is read at most once.
func newFactory() *cmdutil.Factory {
f := &cmdutil.Factory{IO: ui.System()}
f.Context = func() (*ctxstore.Context, error) {
c, err := ctxstore.Load()
if err != nil {
return nil, err
}
if !c.Selected() {
return nil, fmt.Errorf("no context selected; run `yuctl select <partition>@<region>` first")
}
return c, nil
}
var topo *discovery.Topology
f.Topology = func(ctx context.Context) (*discovery.Topology, error) {
if topo != nil {
return topo, nil
}
var err error
topo, err = resolveTopology(ctx)
return topo, err
}
return f
}
func setupLogging() error {
zerolog.TimeFieldFormat = time.RFC3339
level, err := zerolog.ParseLevel(flagLogLevel)
@@ -79,8 +116,7 @@ func setupLogging() error {
// resolveTopology returns the cached topology when fresh (see
// discovery.LoadCachedTopology; TTL 1h, YUCTL_DISCOVERY_TTL to override,
// --refresh-discovery to bypass), otherwise builds the discovery client,
// resolves live from S3 state, and refreshes the cache. Shared by every command
// that needs to read state.
// resolves live from S3 state, and refreshes the cache.
func resolveTopology(ctx context.Context) (*discovery.Topology, error) {
if !flagRefreshDiscovery {
if topo, ok := discovery.LoadCachedTopology(log.Logger); ok {
@@ -6,7 +6,8 @@ import (
"github.com/spf13/cobra"
yctx "yuctl/internal/context"
"yuctl/cmdutil"
"yuctl/ctxstore"
)
// parseTarget splits the human form `partition@region`.
@@ -18,7 +19,7 @@ func parseTarget(s string) (partition, region string, err error) {
return p, r, nil
}
func newSelectCmd() *cobra.Command {
func newSelectCmd(f *cmdutil.Factory) *cobra.Command {
return &cobra.Command{
Use: "select <partition>@<region>",
Short: "Select the active partition/region context",
@@ -27,13 +28,12 @@ func newSelectCmd() *cobra.Command {
"Ceph cluster.",
Args: cobra.ExactArgs(1),
RunE: func(cmd *cobra.Command, args []string) error {
ctx := cmd.Context()
partition, region, err := parseTarget(args[0])
if err != nil {
return err
}
topo, err := resolveTopology(ctx)
topo, err := f.Topology(cmd.Context())
if err != nil {
return err
}
@@ -42,24 +42,12 @@ func newSelectCmd() *cobra.Command {
partition, region, strings.Join(topo.Regions(), ", "))
}
newCtx := &yctx.Context{Partition: partition, Region: region}
if err := yctx.Save(newCtx); err != nil {
newCtx := &ctxstore.Context{Partition: partition, Region: region}
if err := ctxstore.Save(newCtx); err != nil {
return err
}
fmt.Fprintf(cmd.OutOrStdout(), "selected %s@%s\n", partition, region)
fmt.Fprintf(f.IO.Out, "selected %s@%s\n", partition, region)
return nil
},
}
}
// requireContext loads the persisted context, erroring if nothing is selected.
func requireContext() (*yctx.Context, error) {
c, err := yctx.Load()
if err != nil {
return nil, err
}
if !c.Selected() {
return nil, fmt.Errorf("no context selected; run `yuctl select <partition>@<region>` first")
}
return c, nil
}
@@ -1,6 +1,7 @@
package cli
package bench
import (
"context"
"crypto/rand"
"encoding/binary"
"encoding/hex"
@@ -13,23 +14,79 @@ import (
"github.com/rs/zerolog/log"
"github.com/spf13/cobra"
"yuctl/internal/adminapi"
"yuctl/internal/bench"
yctx "yuctl/internal/context"
"yuctl/internal/discovery"
"yuctl/adminapi"
"yuctl/cmdutil"
"yuctl/resticbench"
)
func newToolsCmd() *cobra.Command {
tools := &cobra.Command{
Use: "tools",
Short: "Operational tooling for the selected context",
func New(f *cmdutil.Factory) *cobra.Command {
o := &options{}
cmd := &cobra.Command{
Use: "bench",
Short: "Benchmark michael end-to-end with restic from a management host",
Long: "Pushes a bench agent + a pinned restic to a management host of the selected\n" +
"region and runs write/incremental/check/restore phases against michael,\n" +
"streaming progress back and saving a results JSON locally. Without --repo,\n" +
"a fresh benchmark repository is created via the admin-api (yuctl login).",
RunE: func(cmd *cobra.Command, _ []string) error {
return o.run(cmd.Context(), f, resticbench.DefaultPhases)
},
}
tools.AddCommand(newBenchCmd(), newBenchWideCmd(), newWarpCmd())
return tools
o.registerCommon(cmd)
cmd.Flags().StringVar(&o.size, "size", "32GiB", "dataset size per cell (e.g. 1TiB)")
cmd.Flags().StringVar(&o.fileSize, "file-size", "64MiB", "size of each generated file")
cmd.Flags().StringVar(&o.connections, "connections", "5", "rest.connections sweep, comma-separated (e.g. 5,16,32,64)")
cmd.Flags().IntVar(&o.readConc, "read-concurrency", 4, "restic backup read concurrency")
cmd.Flags().IntVar(&o.packSizeMiB, "pack-size", 16, "restic pack size in MiB (16 = restic default, max 128)")
cmd.Flags().StringVar(&o.compression, "compression", "off", "restic compression (off keeps client CPU out of the way; bench data is incompressible)")
cmd.Flags().IntVar(&o.incrementals, "incrementals", 2, "incremental backup rounds per cell")
cmd.Flags().Float64Var(&o.mutatePercent, "mutate-percent", 2, "percent of files rewritten before each incremental")
cmd.Flags().StringVar(&o.phases, "phases", strings.Join(resticbench.DefaultPhases, ","), "phases to run")
cmd.Flags().Uint64Var(&o.seed, "seed", 0, "dataset seed (0 = random; reuse for identical content across runs)")
cmd.Flags().StringVar(&o.label, "label", "run", "label stored in the results (e.g. before, after)")
cmd.Flags().StringVar(&o.out, "out", "", "local results file (default bench-<label>-<timestamp>.json)")
cmd.Flags().BoolVar(&o.keepData, "keep-data", false, "keep dataset and restore target on disk (doubles space needs)")
cmd.Flags().BoolVar(&o.cleanup, "cleanup", false, "forget+prune this tool's snapshots after the run (also timed)")
cmd.AddCommand(newCompareCmd(f), newCleanupCmd(f))
return cmd
}
type benchFlags struct {
admin adminFlags
func newCompareCmd(f *cmdutil.Factory) *cobra.Command {
return &cobra.Command{
Use: "compare <before.json> <after.json>",
Short: "Render before/after deltas from two results files",
Args: cobra.ExactArgs(2),
RunE: func(_ *cobra.Command, args []string) error {
before, err := resticbench.LoadResult(args[0])
if err != nil {
return err
}
after, err := resticbench.LoadResult(args[1])
if err != nil {
return err
}
resticbench.RenderCompare(f.IO.Out, before, after)
return nil
},
}
}
func newCleanupCmd(f *cmdutil.Factory) *cobra.Command {
o := &options{}
cmd := &cobra.Command{
Use: "cleanup",
Short: "Forget and prune every snapshot created by the bench (timed)",
RunE: func(cmd *cobra.Command, _ []string) error {
return o.run(cmd.Context(), f, []string{resticbench.PhaseCleanup})
},
}
o.registerCommon(cmd)
return cmd
}
type options struct {
admin cmdutil.AdminFlags
host string
fromHere bool
@@ -56,116 +113,40 @@ type benchFlags struct {
cleanup bool
}
func (f *benchFlags) registerCommon(c *cobra.Command) {
f.admin.register(c)
c.Flags().StringVar(&f.host, "host", "", "ssh destination of the management host (default: the region's first mgmt host from discovery)")
c.Flags().BoolVar(&f.fromHere, "from-here", false, "run the benchmark on this machine (no ssh; agent runs in-process)")
c.Flags().StringVar(&f.sshIdentity, "ssh-identity", "", "ssh private key for the management host (default: ssh agent/config)")
c.Flags().StringVar(&f.sshUser, "ssh-user", "", "ssh username for the discovery-resolved mgmt host (default: your local username; identity-registry accounts differ)")
c.Flags().StringVar(&f.agentBin, "agent-bin", "", "local linux/amd64 bench-agent binary (default: the embedded one)")
c.Flags().StringVar(&f.repo, "repo", "", "restic repository URL; skips admin-api provisioning (default $RESTIC_REPOSITORY, else a repo is created via admin-api)")
c.Flags().StringVar(&f.repoID, "repo-id", "", "existing repository id; a fresh URL is minted via admin-api")
c.Flags().StringVar(&f.passwordFile, "password-file", "", "local file with the restic password (default $RESTIC_PASSWORD; auto-generated for auto-created repos)")
c.Flags().StringVar(&f.workdir, "workdir", "/var/tmp/yucca-bench", "remote scratch directory (must fit --size)")
func (o *options) registerCommon(c *cobra.Command) {
o.admin.Register(c)
c.Flags().StringVar(&o.host, "host", "", "ssh destination of the management host (default: the region's first mgmt host from discovery)")
c.Flags().BoolVar(&o.fromHere, "from-here", false, "run the benchmark on this machine (no ssh; agent runs in-process)")
c.Flags().StringVar(&o.sshIdentity, "ssh-identity", "", "ssh private key for the management host (default: ssh agent/config)")
c.Flags().StringVar(&o.sshUser, "ssh-user", "", "ssh username for the discovery-resolved mgmt host (default: your local username; identity-registry accounts differ)")
c.Flags().StringVar(&o.agentBin, "agent-bin", "", "local linux/amd64 bench-agent binary (default: the embedded one)")
c.Flags().StringVar(&o.repo, "repo", "", "restic repository URL; skips admin-api provisioning (default $RESTIC_REPOSITORY, else a repo is created via admin-api)")
c.Flags().StringVar(&o.repoID, "repo-id", "", "existing repository id; a fresh URL is minted via admin-api")
c.Flags().StringVar(&o.passwordFile, "password-file", "", "local file with the restic password (default $RESTIC_PASSWORD; auto-generated for auto-created repos)")
c.Flags().StringVar(&o.workdir, "workdir", "/var/tmp/yucca-bench", "remote scratch directory (must fit --size)")
}
func newBenchCmd() *cobra.Command {
f := &benchFlags{}
cmd := &cobra.Command{
Use: "bench",
Short: "Benchmark michael end-to-end with restic from a management host",
Long: "Pushes a bench agent + a pinned restic to a management host of the selected\n" +
"region and runs write/incremental/check/restore phases against michael,\n" +
"streaming progress back and saving a results JSON locally. Without --repo,\n" +
"a fresh benchmark repository is created via the admin-api (yuctl login).",
RunE: func(cmd *cobra.Command, _ []string) error {
return f.runBench(cmd, bench.DefaultPhases)
},
}
f.registerCommon(cmd)
cmd.Flags().StringVar(&f.size, "size", "32GiB", "dataset size per cell (e.g. 1TiB)")
cmd.Flags().StringVar(&f.fileSize, "file-size", "64MiB", "size of each generated file")
cmd.Flags().StringVar(&f.connections, "connections", "5", "rest.connections sweep, comma-separated (e.g. 5,16,32,64)")
cmd.Flags().IntVar(&f.readConc, "read-concurrency", 4, "restic backup read concurrency")
cmd.Flags().IntVar(&f.packSizeMiB, "pack-size", 16, "restic pack size in MiB (16 = restic default, max 128)")
cmd.Flags().StringVar(&f.compression, "compression", "off", "restic compression (off keeps client CPU out of the way; bench data is incompressible)")
cmd.Flags().IntVar(&f.incrementals, "incrementals", 2, "incremental backup rounds per cell")
cmd.Flags().Float64Var(&f.mutatePercent, "mutate-percent", 2, "percent of files rewritten before each incremental")
cmd.Flags().StringVar(&f.phases, "phases", strings.Join(bench.DefaultPhases, ","), "phases to run")
cmd.Flags().Uint64Var(&f.seed, "seed", 0, "dataset seed (0 = random; reuse for identical content across runs)")
cmd.Flags().StringVar(&f.label, "label", "run", "label stored in the results (e.g. before, after)")
cmd.Flags().StringVar(&f.out, "out", "", "local results file (default bench-<label>-<timestamp>.json)")
cmd.Flags().BoolVar(&f.keepData, "keep-data", false, "keep dataset and restore target on disk (doubles space needs)")
cmd.Flags().BoolVar(&f.cleanup, "cleanup", false, "forget+prune this tool's snapshots after the run (also timed)")
cmd.AddCommand(newBenchCompareCmd(), newBenchCleanupCmd())
return cmd
}
func newBenchCompareCmd() *cobra.Command {
return &cobra.Command{
Use: "compare <before.json> <after.json>",
Short: "Render before/after deltas from two results files",
Args: cobra.ExactArgs(2),
RunE: func(_ *cobra.Command, args []string) error {
before, err := bench.LoadResult(args[0])
if err != nil {
return err
}
after, err := bench.LoadResult(args[1])
if err != nil {
return err
}
bench.RenderCompare(os.Stdout, before, after)
return nil
},
}
}
func newBenchCleanupCmd() *cobra.Command {
f := &benchFlags{}
cmd := &cobra.Command{
Use: "cleanup",
Short: "Forget and prune every snapshot created by the bench (timed)",
RunE: func(cmd *cobra.Command, _ []string) error {
return f.runBench(cmd, []string{bench.PhaseCleanup})
},
}
f.registerCommon(cmd)
return cmd
}
// runBench resolves the context (host from discovery, repository via admin-api
// unless supplied) and drives the run. Context/topology are resolved lazily so
// a --from-here run with an explicit --repo needs neither state creds nor a
// selected context (e.g. yuctl invoked directly on a mgmt host).
func (f *benchFlags) runBench(cmd *cobra.Command, defaultPhases []string) error {
ctx := cmd.Context()
var cc *yctx.Context
var topo *discovery.Topology
loadTopo := func() error {
if topo != nil {
return nil
}
var err error
if cc, err = requireContext(); err != nil {
return err
}
topo, err = resolveTopology(ctx)
return err
}
// run resolves the target (host from discovery, repository via admin-api
// unless supplied) and drives the benchmark. Context/topology are resolved
// lazily through the factory so a --from-here run with an explicit --repo
// needs neither state creds nor a selected context (e.g. yuctl invoked
// directly on a mgmt host).
func (o *options) run(ctx context.Context, f *cmdutil.Factory, defaultPhases []string) error {
var host string
switch {
case f.fromHere && f.host != "":
case o.fromHere && o.host != "":
return fmt.Errorf("--from-here and --host are mutually exclusive")
case f.fromHere:
case o.fromHere:
// no remote host: the agent runs in-process on this machine
case f.host != "":
host = f.host
case o.host != "":
host = o.host
default:
if err := loadTopo(); err != nil {
cc, err := f.Context()
if err != nil {
return err
}
topo, err := f.Topology(ctx)
if err != nil {
return err
}
hosts := topo.MgmtHosts(cc.Partition, cc.Region)
@@ -173,35 +154,31 @@ func (f *benchFlags) runBench(cmd *cobra.Command, defaultPhases []string) error
return fmt.Errorf("discovery has no mgmt hosts for %s@%s; pass --host (or --from-here)", cc.Partition, cc.Region)
}
host = hosts[0].PublicIP
if f.sshUser != "" {
host = f.sshUser + "@" + host
if o.sshUser != "" {
host = o.sshUser + "@" + host
}
log.Info().Str("mgmt", hosts[0].Name).Str("host", host).Msg("using mgmt host from discovery")
}
cfg, err := f.buildConfig(defaultPhases)
cfg, err := o.buildConfig(defaultPhases)
if err != nil {
return err
}
if cfg.Repo == "" {
// Topology is only needed to derive the admin URL; an explicit
// --admin-url / $YUCTL_ADMIN_API_URL skips state access entirely
// (the context file alone names the token-cache partition).
if f.admin.adminURL == "" && os.Getenv("YUCTL_ADMIN_API_URL") == "" {
if err := loadTopo(); err != nil {
return err
}
} else if cc == nil {
if cc, err = requireContext(); err != nil {
return err
}
}
client, _, err := f.admin.adminLogin(ctx, cmd, cc, topo)
cc, err := f.Context()
if err != nil {
return err
}
repoID := f.repoID
topo, err := o.admin.OptionalTopology(ctx, f)
if err != nil {
return err
}
client, _, err := o.admin.Login(ctx, f, cc, topo)
if err != nil {
return err
}
repoID := o.repoID
if repoID == "" {
name := fmt.Sprintf("yucca-bench-%s-%s", cfg.Label, time.Now().Format("20060102-150405"))
repo, err := client.CreateRepository(ctx, name, false, adminapi.CreateRepositoryOptions{})
@@ -228,47 +205,47 @@ func (f *benchFlags) runBench(cmd *cobra.Command, defaultPhases []string) error
return fmt.Errorf("no password: set --password-file or RESTIC_PASSWORD")
}
if f.out == "" {
f.out = fmt.Sprintf("bench-%s-%s.json", cfg.Label, time.Now().Format("20060102-150405"))
if o.out == "" {
o.out = fmt.Sprintf("bench-%s-%s.json", cfg.Label, time.Now().Format("20060102-150405"))
}
outPath := f.out
if len(cfg.Phases) == 1 && cfg.Phases[0] == bench.PhaseCleanup {
outPath := o.out
if len(cfg.Phases) == 1 && cfg.Phases[0] == resticbench.PhaseCleanup {
outPath = ""
}
opts := bench.RunOpts{Host: host, SSHIdentity: f.sshIdentity, AgentBin: f.agentBin, Config: cfg, Out: outPath}
if f.fromHere {
_, err = bench.RunHere(ctx, opts)
opts := resticbench.RunOpts{Host: host, SSHIdentity: o.sshIdentity, AgentBin: o.agentBin, Config: cfg, Out: outPath, Summary: f.IO.Out}
if o.fromHere {
_, err = resticbench.RunHere(ctx, opts)
} else {
_, err = bench.Run(ctx, opts)
_, err = resticbench.Run(ctx, opts)
}
return err
}
func (f *benchFlags) buildConfig(defaultPhases []string) (bench.Config, error) {
cfg := bench.Config{
Workdir: f.workdir,
ReadConcurrency: f.readConc,
PackSizeMiB: f.packSizeMiB,
Compression: f.compression,
Incrementals: f.incrementals,
MutatePercent: f.mutatePercent,
Label: f.label,
func (o *options) buildConfig(defaultPhases []string) (resticbench.Config, error) {
cfg := resticbench.Config{
Workdir: o.workdir,
ReadConcurrency: o.readConc,
PackSizeMiB: o.packSizeMiB,
Compression: o.compression,
Incrementals: o.incrementals,
MutatePercent: o.mutatePercent,
Label: o.label,
Tag: "yucca-bench",
KeepData: f.keepData,
Seed: f.seed,
KeepData: o.keepData,
Seed: o.seed,
Phases: defaultPhases,
}
if cfg.Label == "" {
cfg.Label = "run"
}
cfg.Repo = f.repo
if cfg.Repo == "" && f.repoID == "" {
cfg.Repo = o.repo
if cfg.Repo == "" && o.repoID == "" {
cfg.Repo = os.Getenv("RESTIC_REPOSITORY")
}
if f.passwordFile != "" {
b, err := os.ReadFile(f.passwordFile)
if o.passwordFile != "" {
b, err := os.ReadFile(o.passwordFile)
if err != nil {
return cfg, err
}
@@ -277,21 +254,21 @@ func (f *benchFlags) buildConfig(defaultPhases []string) (bench.Config, error) {
cfg.Password = os.Getenv("RESTIC_PASSWORD")
}
if f.size != "" {
size, err := bench.ParseSize(f.size)
if o.size != "" {
size, err := resticbench.ParseSize(o.size)
if err != nil {
return cfg, err
}
cfg.Size = size
fileSize, err := bench.ParseSize(f.fileSize)
fileSize, err := resticbench.ParseSize(o.fileSize)
if err != nil {
return cfg, err
}
cfg.FileSize = fileSize
}
if f.connections != "" {
for _, s := range strings.Split(f.connections, ",") {
if o.connections != "" {
for _, s := range strings.Split(o.connections, ",") {
n, err := strconv.Atoi(strings.TrimSpace(s))
if err != nil || n < 1 {
return cfg, fmt.Errorf("invalid --connections value %q", s)
@@ -300,11 +277,11 @@ func (f *benchFlags) buildConfig(defaultPhases []string) (bench.Config, error) {
}
}
if f.phases != "" {
cfg.Phases = strings.Split(f.phases, ",")
if o.phases != "" {
cfg.Phases = strings.Split(o.phases, ",")
}
if f.cleanup && !cfg.HasPhase(bench.PhaseCleanup) {
cfg.Phases = append(cfg.Phases, bench.PhaseCleanup)
if o.cleanup && !cfg.HasPhase(resticbench.PhaseCleanup) {
cfg.Phases = append(cfg.Phases, resticbench.PhaseCleanup)
}
if cfg.Seed == 0 {
@@ -1,43 +1,41 @@
package cli
package fleetbench
import (
"bufio"
"context"
"fmt"
"os"
"os/signal"
"strings"
"syscall"
"time"
"github.com/rs/zerolog/log"
"github.com/spf13/cobra"
"yuctl/internal/adminapi"
"yuctl/internal/bench"
"yuctl/internal/benchwide"
"yuctl/internal/provider"
"yuctl/cmdutil"
"yuctl/fleet"
fbfleet "yuctl/fleet/fleetbench"
"yuctl/provider"
"yuctl/resticbench"
)
// benchDoFlags are shared by every bench-wide subcommand.
type benchDoFlags struct {
// flags are shared by every fleet-bench subcommand.
type flags struct {
yes bool
provider string
}
func (f *benchDoFlags) register(c *cobra.Command) {
c.PersistentFlags().BoolVar(&f.yes, "yes", false, "skip the host-creation confirmation prompt")
c.PersistentFlags().StringVar(&f.provider, "provider", "do", "cloud provider for this fleet ("+strings.Join(provider.Names(), " | ")+")")
func (o *flags) register(c *cobra.Command) {
c.PersistentFlags().BoolVar(&o.yes, "yes", false, "skip the host-creation confirmation prompt")
c.PersistentFlags().StringVar(&o.provider, "provider", "do", "cloud provider for this fleet ("+strings.Join(provider.Names(), " | ")+")")
}
// session opens the fleet session for the selected partition + provider. The
// label describes the target for display ("prod · yuctl-bench-do-prod").
func (f *benchDoFlags) session(ctx context.Context) (*benchwide.Session, string, error) {
cc, err := requireContext()
func (o *flags) session(ctx context.Context, f *cmdutil.Factory) (*fbfleet.Session, string, error) {
cc, err := f.Context()
if err != nil {
return nil, "", err
}
s, err := benchwide.NewSession(ctx, cc.Partition, f.provider)
s, err := fbfleet.NewSession(ctx, cc.Partition, o.provider)
if err != nil {
return nil, "", err
}
@@ -45,43 +43,32 @@ func (f *benchDoFlags) session(ctx context.Context) (*benchwide.Session, string,
}
// minter runs the admin-api login flow and returns the repo-minting client.
func (f *benchDoFlags) minter(ctx context.Context, cmd *cobra.Command, admin *adminFlags) (benchwide.RepoMinter, error) {
cc, err := requireContext()
func (o *flags) minter(ctx context.Context, f *cmdutil.Factory, admin *cmdutil.AdminFlags) (fbfleet.RepoMinter, error) {
cc, err := f.Context()
if err != nil {
return nil, err
}
// Topology is only needed to derive the admin URL; an explicit --admin-url
// or $YUCTL_ADMIN_API_URL skips state access entirely.
var client *adminapi.Client
if admin.adminURL == "" && os.Getenv("YUCTL_ADMIN_API_URL") == "" {
topo, err := resolveTopology(ctx)
if err != nil {
return nil, err
}
client, _, err = admin.adminLogin(ctx, cmd, cc, topo)
if err != nil {
return nil, err
}
} else {
client, _, err = admin.adminLogin(ctx, cmd, cc, nil)
if err != nil {
return nil, err
}
topo, err := admin.OptionalTopology(ctx, f)
if err != nil {
return nil, err
}
client, _, err := admin.Login(ctx, f, cc, topo)
if err != nil {
return nil, err
}
return client, nil
}
// confirm returns the deploy confirmation callback: nil with --yes, otherwise
// an interactive y/N prompt on the terminal.
func (f *benchDoFlags) confirm(cmd *cobra.Command) func(string) bool {
if f.yes {
func (o *flags) confirm(f *cmdutil.Factory) func(string) bool {
if o.yes {
return nil
}
return func(plan string) bool {
out := cmd.ErrOrStderr()
fmt.Fprintln(out, "\nbench-wide will "+plan)
fmt.Fprint(out, "\nProceed? [y/N] ")
sc := bufio.NewScanner(cmd.InOrStdin())
fmt.Fprintln(f.IO.Err, "\nfleet-bench will "+plan)
fmt.Fprint(f.IO.Err, "\nProceed? [y/N] ")
sc := bufio.NewScanner(f.IO.In)
if !sc.Scan() {
return false
}
@@ -90,10 +77,10 @@ func (f *benchDoFlags) confirm(cmd *cobra.Command) func(string) bool {
}
}
func newBenchWideCmd() *cobra.Command {
f := &benchDoFlags{}
func New(f *cmdutil.Factory) *cobra.Command {
o := &flags{}
cmd := &cobra.Command{
Use: "bench-wide",
Use: "fleet-bench",
Short: "Restic client fleet across cloud providers writing against michael",
Long: "Deploys a fleet of cloud VMs on a chosen --provider (DigitalOcean, Hetzner;\n" +
"OVH later) with a per-fleet ephemeral ssh key and runs real restic clients\n" +
@@ -104,48 +91,48 @@ func newBenchWideCmd() *cobra.Command {
"overage. Fleets are per provider × partition, so several providers can load\n" +
"michael at once (run deploy/start per --provider).",
}
f.register(cmd)
o.register(cmd)
cmd.AddCommand(
newBenchDoDeployCmd(f),
newBenchDoStartCmd(f),
newBenchDoStatusCmd(f),
newBenchDoWatchCmd(f),
newBenchDoStopCmd(f),
newBenchDoCleanupCmd(f),
newBenchDoUndeployCmd(f),
newDeployCmd(f, o),
newStartCmd(f, o),
newStatusCmd(f, o),
newWatchCmd(f, o),
newStopCmd(f, o),
newCleanupCmd(f, o),
newUndeployCmd(f, o),
)
return cmd
}
func benchDoDeployFlags(c *cobra.Command, o *benchwide.DeployOptions) {
c.Flags().IntVar(&o.Droplets, "droplets", 3, "fleet size")
func deployFlags(c *cobra.Command, o *fbfleet.DeployOptions) {
c.Flags().IntVar(&o.Hosts, "hosts", 3, "fleet size")
c.Flags().StringSliceVar(&o.Regions, "region", nil, "regions, round-robined across the fleet (default: the provider's)")
c.Flags().StringVar(&o.Size, "size", "", "instance size slug (default: the provider's)")
c.Flags().StringVar(&o.Image, "image", "", "image slug (default: the provider's)")
c.Flags().StringVar(&o.AgentBin, "agent-bin", "", "local linux/amd64 bench-agent binary (default: the embedded one)")
}
func newBenchDoDeployCmd(f *benchDoFlags) *cobra.Command {
o := benchwide.DeployOptions{}
func newDeployCmd(f *cmdutil.Factory, fl *flags) *cobra.Command {
o := fbfleet.DeployOptions{}
cmd := &cobra.Command{
Use: "deploy",
Short: "Create (or converge) the droplet fleet and push the agent + restic",
Short: "Create (or converge) the host fleet and push the agent + restic",
RunE: func(cmd *cobra.Command, _ []string) error {
s, _, err := f.session(cmd.Context())
s, _, err := fl.session(cmd.Context(), f)
if err != nil {
return err
}
o.Confirm = f.confirm(cmd)
o.Confirm = fl.confirm(f)
return s.Deploy(cmd.Context(), o)
},
}
benchDoDeployFlags(cmd, &o)
deployFlags(cmd, &o)
return cmd
}
func newBenchDoStartCmd(f *benchDoFlags) *cobra.Command {
admin := &adminFlags{}
d := benchwide.DeployOptions{}
func newStartCmd(f *cmdutil.Factory, fl *flags) *cobra.Command {
admin := &cmdutil.AdminFlags{}
d := fbfleet.DeployOptions{}
var (
objSize string
cycleSize string
@@ -154,25 +141,25 @@ func newBenchDoStartCmd(f *benchDoFlags) *cobra.Command {
maxTransfer string
autoDeploy bool
)
o := benchwide.StartOptions{}
o := fbfleet.StartOptions{}
cmd := &cobra.Command{
Use: "start",
Short: "Start (or gracefully restart) the restic load on the fleet",
Long: "Ensures one repository per client (admin-api; `yuctl login` session), mints\n" +
"fresh restic URLs, and launches the detached load supervisor on every\n" +
"droplet. A second start kills the previous load first and relaunches with\n" +
"the new parameters. The load ends when --duration elapses, the per-droplet\n" +
"transfer cap is hit, or `bench-wide stop`.",
"host. A second start kills the previous load first and relaunches with\n" +
"the new parameters. The load ends when --duration elapses, the per-host\n" +
"transfer cap is hit, or `fleet-bench stop`.",
RunE: func(cmd *cobra.Command, _ []string) error {
ctx := cmd.Context()
var err error
if o.PackSizeMiB, err = parseMiB(objSize, "--obj-size"); err != nil {
return err
}
if o.CycleSize, err = bench.ParseSize(cycleSize); err != nil {
if o.CycleSize, err = resticbench.ParseSize(cycleSize); err != nil {
return err
}
if o.FileSize, err = bench.ParseSize(fileSize); err != nil {
if o.FileSize, err = resticbench.ParseSize(fileSize); err != nil {
return err
}
if duration != "" && duration != "0" {
@@ -183,49 +170,49 @@ func newBenchDoStartCmd(f *benchDoFlags) *cobra.Command {
o.Duration = dur
}
if maxTransfer != "" {
if o.MaxTransfer, err = bench.ParseSize(maxTransfer); err != nil {
if o.MaxTransfer, err = resticbench.ParseSize(maxTransfer); err != nil {
return err
}
}
s, _, err := f.session(ctx)
s, _, err := fl.session(ctx, f)
if err != nil {
return err
}
if autoDeploy {
if droplets, err := s.Droplets(ctx); err == nil && len(droplets) == 0 {
d.Confirm = f.confirm(cmd)
if hosts, err := s.Hosts(ctx); err == nil && len(hosts) == 0 {
d.Confirm = fl.confirm(f)
if err := s.Deploy(ctx, d); err != nil {
return err
}
}
}
minter, err := f.minter(ctx, cmd, admin)
minter, err := fl.minter(ctx, f, admin)
if err != nil {
return err
}
return s.Start(ctx, minter, o)
},
}
cmd.Flags().IntVar(&o.ClientsPerDroplet, "clients-per-droplet", 1, "restic clients per droplet (each with its own repository)")
cmd.Flags().IntVar(&o.ClientsPerHost, "clients-per-host", 1, "restic clients per host (each with its own repository)")
cmd.Flags().StringVar(&objSize, "obj-size", "16MiB", "restic pack size — the object size michael sees (4..128 MiB)")
cmd.Flags().StringVar(&cycleSize, "cycle-size", "8GiB", "dataset per client per cycle (freshly seeded every cycle)")
cmd.Flags().StringVar(&fileSize, "file-size", "64MiB", "size of each generated file")
cmd.Flags().IntVar(&o.Connections, "connections", 5, "rest.connections per client")
cmd.Flags().IntVar(&o.ReadConcurrency, "read-concurrency", 4, "restic backup read concurrency")
cmd.Flags().StringVar(&o.Compression, "compression", "off", "restic compression (bench data is incompressible)")
cmd.Flags().StringVar(&duration, "duration", "1h", "run length (e.g. 2h; 0 = non-stop until bench-wide stop)")
cmd.Flags().StringVar(&maxTransfer, "max-transfer", "", "per-droplet wire-TX cap (default: the droplet size's transfer allowance)")
cmd.Flags().StringVar(&duration, "duration", "1h", "run length (e.g. 2h; 0 = non-stop until fleet-bench stop)")
cmd.Flags().StringVar(&maxTransfer, "max-transfer", "", "per-host wire-TX cap (default: the host size's transfer allowance)")
cmd.Flags().StringVar(&o.Label, "label", "run", "label stored in the results")
cmd.Flags().Uint64Var(&o.Seed, "seed", 0, "dataset seed (0 = random)")
cmd.Flags().BoolVar(&autoDeploy, "auto-deploy", true, "deploy the fleet first if it is missing (asks before creating droplets)")
benchDoDeployFlags(cmd, &d)
admin.register(cmd)
cmd.Flags().BoolVar(&autoDeploy, "auto-deploy", true, "deploy the fleet first if it is missing (asks before creating hosts)")
deployFlags(cmd, &d)
admin.Register(cmd)
return cmd
}
func parseMiB(v, flag string) (int, error) {
n, err := bench.ParseSize(v)
n, err := resticbench.ParseSize(v)
if err != nil {
return 0, fmt.Errorf("%s: %w", flag, err)
}
@@ -235,23 +222,23 @@ func parseMiB(v, flag string) (int, error) {
return int(n >> 20), nil
}
func newBenchDoStatusCmd(f *benchDoFlags) *cobra.Command {
func newStatusCmd(f *cmdutil.Factory, fl *flags) *cobra.Command {
var sample int
cmd := &cobra.Command{
Use: "status",
Short: "Show per-droplet load state, throughput, and transfer budget",
Short: "Show per-host load state, throughput, and transfer budget",
RunE: func(cmd *cobra.Command, _ []string) error {
s, label, err := f.session(cmd.Context())
s, label, err := fl.session(cmd.Context(), f)
if err != nil {
return err
}
log.Info().Int("window_s", sample).Msg("sampling droplets")
log.Info().Int("window_s", sample).Msg("sampling hosts")
report, err := s.Status(cmd.Context(), sample)
if err != nil {
return err
}
v := &benchDoView{label: label}
fmt.Fprintln(cmd.OutOrStdout(), v.render(report, time.Now(), sample, false))
v := &view{label: label}
fmt.Fprintln(f.IO.Out, v.render(report, time.Now(), sample, false))
return nil
},
}
@@ -259,61 +246,44 @@ func newBenchDoStatusCmd(f *benchDoFlags) *cobra.Command {
return cmd
}
func newBenchDoWatchCmd(f *benchDoFlags) *cobra.Command {
func newWatchCmd(f *cmdutil.Factory, fl *flags) *cobra.Command {
var sample int
cmd := &cobra.Command{
Use: "watch",
Short: "Live dashboard: continuously sampled fleet state and throughput",
RunE: func(cmd *cobra.Command, _ []string) error {
ctx, stop := signal.NotifyContext(cmd.Context(), os.Interrupt, syscall.SIGTERM)
defer stop()
s, label, err := f.session(ctx)
s, label, err := fl.session(cmd.Context(), f)
if err != nil {
return err
}
out := cmd.OutOrStdout()
fmt.Fprint(out, "\x1b[?1049h\x1b[?25l") // alt screen, hide cursor
defer fmt.Fprint(out, "\x1b[?25h\x1b[?1049l")
fmt.Fprintf(out, "%s connecting · sampling %ds window…\n", label, sample)
v := &benchDoView{label: label}
for {
v := &view{label: label}
return fleet.Watch(cmd.Context(), f.IO.Out, label, sample, func(ctx context.Context) (string, error) {
report, err := s.Status(ctx, sample)
if ctx.Err() != nil {
return nil
}
fmt.Fprint(out, "\x1b[H\x1b[2J")
if err != nil {
fmt.Fprintf(out, "status error (retrying): %v\n", err)
select {
case <-ctx.Done():
return nil
case <-time.After(3 * time.Second):
}
continue
return "", err
}
var combined float64
for _, d := range report.Droplets {
for _, d := range report.Hosts {
combined += d.TxBps
}
if len(report.Droplets) > 0 {
v.push(combined)
if len(report.Hosts) > 0 {
v.history.Push(combined)
}
fmt.Fprintln(out, v.render(report, time.Now(), sample, true))
}
return v.render(report, time.Now(), sample, true), nil
})
},
}
cmd.Flags().IntVar(&sample, "sample", 5, "NIC sampling window per refresh in seconds")
return cmd
}
func newBenchDoStopCmd(f *benchDoFlags) *cobra.Command {
func newStopCmd(f *cmdutil.Factory, fl *flags) *cobra.Command {
var out string
cmd := &cobra.Command{
Use: "stop",
Short: "Stop the load everywhere and save the results JSON (droplets stay)",
Short: "Stop the load everywhere and save the results JSON (hosts stay)",
RunE: func(cmd *cobra.Command, _ []string) error {
s, _, err := f.session(cmd.Context())
s, _, err := fl.session(cmd.Context(), f)
if err != nil {
return err
}
@@ -322,36 +292,36 @@ func newBenchDoStopCmd(f *benchDoFlags) *cobra.Command {
return err
}
if out == "" {
out = fmt.Sprintf("bench-wide-%s-%s.json", res.Label, time.Now().Format("20060102-150405"))
out = fmt.Sprintf("fleet-bench-%s-%s.json", res.Label, time.Now().Format("20060102-150405"))
}
if err := saveBenchDoResult(out, res); err != nil {
if err := fbfleet.SaveResult(out, res); err != nil {
return err
}
log.Info().Str("path", out).Msg("results saved")
renderBenchDoResult(cmd.OutOrStdout(), res)
fbfleet.RenderResult(f.IO.Out, res)
return nil
},
}
cmd.Flags().StringVar(&out, "out", "", "local results file (default bench-wide-<label>-<timestamp>.json)")
cmd.Flags().StringVar(&out, "out", "", "local results file (default fleet-bench-<label>-<timestamp>.json)")
return cmd
}
func newBenchDoCleanupCmd(f *benchDoFlags) *cobra.Command {
admin := &adminFlags{}
func newCleanupCmd(f *cmdutil.Factory, fl *flags) *cobra.Command {
admin := &cmdutil.AdminFlags{}
var force bool
cmd := &cobra.Command{
Use: "cleanup",
Short: "Forget and prune every bench-wide snapshot (run before undeploy)",
Long: "Runs restic forget+prune for every client repository, from its droplet —\n" +
Short: "Forget and prune every fleet-bench snapshot (run before undeploy)",
Long: "Runs restic forget+prune for every client repository, from its host —\n" +
"the repos themselves persist (admin-api deletion is unimplemented) but end\n" +
"~empty. Needs the fleet still deployed and a `yuctl login` session.",
RunE: func(cmd *cobra.Command, _ []string) error {
ctx := cmd.Context()
s, _, err := f.session(ctx)
s, _, err := fl.session(ctx, f)
if err != nil {
return err
}
minter, err := f.minter(ctx, cmd, admin)
minter, err := fl.minter(ctx, f, admin)
if err != nil {
return err
}
@@ -359,17 +329,17 @@ func newBenchDoCleanupCmd(f *benchDoFlags) *cobra.Command {
},
}
cmd.Flags().BoolVar(&force, "force", false, "clean even while load is running")
admin.register(cmd)
admin.Register(cmd)
return cmd
}
func newBenchDoUndeployCmd(f *benchDoFlags) *cobra.Command {
func newUndeployCmd(f *cmdutil.Factory, fl *flags) *cobra.Command {
var force bool
cmd := &cobra.Command{
Use: "undeploy",
Short: "Destroy every fleet droplet and the ephemeral ssh key",
Short: "Destroy every fleet host and the ephemeral ssh key",
RunE: func(cmd *cobra.Command, _ []string) error {
s, _, err := f.session(cmd.Context())
s, _, err := fl.session(cmd.Context(), f)
if err != nil {
return err
}
+163
View File
@@ -0,0 +1,163 @@
package fleetbench
import (
"fmt"
"strings"
"time"
"yuctl/fleet"
fbfleet "yuctl/fleet/fleetbench"
"yuctl/resticbench"
"yuctl/ui"
)
// view renders StatusReports as a styled dashboard; watch mode feeds it
// successive samples and it keeps a TX history for the sparkline.
type view struct {
label string
history fleet.History
}
func (v *view) render(r *fbfleet.StatusReport, sampledAt time.Time, sampleSec int, watching bool) string {
var b strings.Builder
b.WriteString(ui.Badge.Render("FLEET-BENCH") + " " + ui.Title.Render(v.label) + "\n")
if len(r.Hosts) == 0 {
b.WriteString(ui.Warn.Render("no hosts deployed") +
ui.Muted.Render(" — yuctl tools fleet-bench deploy") + "\n")
return ui.Frame.Render(strings.TrimRight(b.String(), "\n"))
}
if r.Run != nil {
since := time.Since(r.Run.StartedAt).Round(time.Second).String() + " ago"
p := r.Run.Params
b.WriteString(fmt.Sprintf("%s %s · %s objects · %s/cycle · %s clients/host · %s\n",
ui.OK.Render("● "+r.Run.Label),
ui.Muted.Render("started "+since),
p["obj_size_mib"]+"MiB", p["cycle_size"], p["clients_per_host"], p["duration"]))
b.WriteString(ui.Muted.Render("cap "+p["cap_per_host"]+" per host") + "\n")
} else {
b.WriteString(ui.Warn.Render("○ no active run recorded") +
ui.Muted.Render(" — load stopped or never started") + "\n")
}
b.WriteString("\n")
// Hosts: TX bar scaled to the busiest host, transfer budget bar scaled to
// the cap.
var maxBps float64
nameW := 7
for _, d := range r.Hosts {
maxBps = max(maxBps, d.TxBps)
nameW = max(nameW, len(d.Name))
}
var wire, capTotal int64
var txBps float64
for _, d := range r.Hosts {
capBytes := r.TransferCap
var used int64
state := "-"
if d.Status != nil {
used = d.Status.WireTxBytes
if d.Status.CapBytes > 0 {
capBytes = d.Status.CapBytes
}
state = d.Status.State
}
budget := " "
if capBytes > 0 {
pct := float64(used) / float64(capBytes) * 100
st := ui.OK
switch {
case pct >= 90:
st = ui.Bad
case pct >= 60:
st = ui.Warn
}
budget = fmt.Sprintf("%s %s", st.Render(ui.Meter(float64(used), float64(capBytes), 10)), st.Render(fmt.Sprintf("%3.0f%%", pct)))
}
line := fmt.Sprintf("%-*s %s %s %s %s %s %s",
nameW, d.Name,
ui.Muted.Render(fmt.Sprintf("%-5s", d.Region)),
ui.TX.Render(ui.Meter(d.TxBps, maxBps, 14)),
ui.PadGbps(d.TxBps),
budget,
ui.Muted.Render(fmt.Sprintf("%9s", resticbench.FormatBytes(used))),
stateCell(d, state))
b.WriteString(line + "\n")
txBps += d.TxBps
wire += used
capTotal += capBytes
}
b.WriteString(fmt.Sprintf("%s %s %s · %s\n",
ui.Muted.Render(fmt.Sprintf("%-*s", nameW+6, "aggregate")),
ui.TX.Render("TX"), ui.PadGbps(txBps),
ui.Total.Render(fmt.Sprintf("pool %s / %s", resticbench.FormatBytes(wire), resticbench.FormatBytes(capTotal)))))
if len(v.history.Values()) > 1 {
b.WriteString(ui.Muted.Render("history ") + ui.Total.Render(ui.Sparkline(v.history.Values(), 40)) + "\n")
}
b.WriteString("\n")
// Clients: loop progress per restic identity.
cnameW := 6
for _, d := range r.Hosts {
if d.Status == nil {
continue
}
for _, c := range d.Status.Clients {
cnameW = max(cnameW, len(fbfleet.ShortName(c.Name)))
}
}
b.WriteString(ui.Muted.Render(fmt.Sprintf("%-*s %-9s %6s %10s %6s %s", cnameW, "CLIENT", "PHASE", "CYCLES", "UPLOADED", "ERR", "LAST ERROR")) + "\n")
for _, d := range r.Hosts {
if d.Status == nil {
if d.Err != "" {
b.WriteString(fmt.Sprintf("%-*s %s\n", cnameW, fbfleet.ShortName(d.Name), ui.Bad.Render(d.Err)))
}
continue
}
for _, c := range d.Status.Clients {
lastErr := ""
if c.LastError != "" {
lastErr = ui.Bad.Render(ui.Truncate(c.LastError, 48))
}
b.WriteString(fmt.Sprintf("%-*s %-9s %6d %10s %s %s\n",
cnameW, fbfleet.ShortName(c.Name), phaseCell(c.Phase), c.Cycles,
resticbench.FormatBytes(c.Uploaded+c.CurrentBytes), ui.ErrCell(c.Errors, 6), lastErr))
}
}
b.WriteString("\n" + ui.Muted.Render(fleet.Footer(sampleSec, sampledAt, watching)))
return ui.Frame.Render(b.String())
}
func phaseCell(phase string) string {
switch phase {
case "backup":
return ui.OK.Render(fmt.Sprintf("%-9s", phase))
case "generate":
return ui.TX.Render(fmt.Sprintf("%-9s", phase))
default:
return ui.Muted.Render(fmt.Sprintf("%-9s", phase))
}
}
// stateCell colors the host's agent state, flagging a dead agent while a run
// is recorded.
func stateCell(d fbfleet.HostStatus, state string) string {
switch {
case !d.Reachable:
return ui.Bad.Render("unreachable")
case state == "running" && d.AgentProcs > 0:
return ui.OK.Render("running")
case state == "capped":
return ui.Warn.Render("capped")
case state == "done":
return ui.Muted.Render("done")
case d.AgentProcs == 0 && state == "running":
return ui.Bad.Render("agent dead")
default:
return ui.Muted.Render(state)
}
}
+19
View File
@@ -0,0 +1,19 @@
package tools
import (
"github.com/spf13/cobra"
"yuctl/cli/tools/bench"
"yuctl/cli/tools/fleetbench"
"yuctl/cli/tools/warp"
"yuctl/cmdutil"
)
func New(f *cmdutil.Factory) *cobra.Command {
tools := &cobra.Command{
Use: "tools",
Short: "Operational tooling for the selected context",
}
tools.AddCommand(bench.New(f), fleetbench.New(f), warp.New(f))
return tools
}
+114
View File
@@ -0,0 +1,114 @@
package warp
import (
"fmt"
"strings"
"time"
"yuctl/fleet"
warpfleet "yuctl/fleet/warp"
"yuctl/ui"
)
// view renders StatusReports as a styled dashboard; watch mode feeds it
// successive samples and it keeps a throughput history for the sparkline.
type view struct {
label string
history fleet.History
}
func (v *view) render(r *warpfleet.StatusReport, sampledAt time.Time, sampleSec int, watching bool) string {
var b strings.Builder
b.WriteString(ui.Badge.Render("WARP") + " " + ui.Title.Render(v.label) + "\n")
if len(r.Pods) == 0 {
b.WriteString(ui.Warn.Render("no runner pods deployed") +
ui.Muted.Render(" — yuctl tools warp deploy") + "\n")
return ui.Frame.Render(strings.TrimRight(b.String(), "\n"))
}
if r.Config != nil {
since := r.Config["started_at"]
if t, err := time.Parse(time.RFC3339, since); err == nil {
since = time.Since(t).Round(time.Second).String() + " ago"
}
b.WriteString(fmt.Sprintf("%s %s · %s PUT / %s GET · %s/%s objects · %s RGWs\n",
ui.OK.Render("● "+r.Config["mode"]),
ui.Muted.Render("started "+since),
r.Config["put_streams"], r.Config["get_streams"],
r.Config["put_obj_size"], r.Config["get_obj_size"],
r.Config["rgw_endpoints"]))
b.WriteString(ui.Muted.Render("endpoint "+r.Config["endpoint"]) + "\n")
} else {
b.WriteString(ui.Warn.Render("○ no active run recorded") +
ui.Muted.Render(" — load stopped or never started") + "\n")
}
b.WriteString("\n")
// Nodes: TX/RX bars scaled to the busiest direction in this frame.
var maxBps float64
for _, n := range r.Nodes {
maxBps = max(maxBps, max(n.TxBps, n.RxBps))
}
nodeW, ifaceW := 4, 5
for _, n := range r.Nodes {
nodeW = max(nodeW, len(n.Node))
ifaceW = max(ifaceW, len(n.Iface))
}
var tx, rx float64
for _, n := range r.Nodes {
b.WriteString(fmt.Sprintf("%-*s %s %s %s %s %s\n",
nodeW, n.Node,
ui.Muted.Render(fmt.Sprintf("%-*s", ifaceW, n.Iface)),
ui.TX.Render(ui.Meter(n.TxBps, maxBps, 14)),
ui.PadGbps(n.TxBps),
ui.RX.Render(ui.Meter(n.RxBps, maxBps, 14)),
ui.PadGbps(n.RxBps)))
tx += n.TxBps
rx += n.RxBps
}
if len(r.Nodes) > 0 {
b.WriteString(fmt.Sprintf("%s %s %s · %s %s · %s\n",
ui.Muted.Render(fmt.Sprintf("%-*s", nodeW+ifaceW-2, "aggregate")),
ui.TX.Render("TX"), ui.PadGbps(tx),
ui.RX.Render("RX"), ui.PadGbps(rx),
ui.Total.Render("combined "+ui.FmtGbps(tx+rx))))
if len(v.history.Values()) > 1 {
b.WriteString(ui.Muted.Render("history ") + ui.Total.Render(ui.Sparkline(v.history.Values(), 40)) + "\n")
}
b.WriteString("\n")
}
// Pods: process liveness and log error counts.
podW := 3
for _, p := range r.Pods {
podW = max(podW, len(p.Name))
}
b.WriteString(ui.Muted.Render(fmt.Sprintf("%-*s %-*s %6s %6s %10s %10s", podW, "POD", nodeW, "NODE", "PUT", "GET", "ERR(PUT)", "ERR(GET)")) + "\n")
for _, p := range r.Pods {
b.WriteString(fmt.Sprintf("%-*s %-*s %s %s %s %s\n",
podW, p.Name, nodeW, p.Node,
procCell(p.PutProcs, r.Config != nil),
procCell(p.GetProcs, r.Config != nil),
ui.ErrCell(p.PutErrors, 10), ui.ErrCell(p.GetErrors, 10)))
}
b.WriteString("\n" + ui.Muted.Render(fleet.Footer(sampleSec, sampledAt, watching)))
return ui.Frame.Render(b.String())
}
// procCell colors a warp process count: green when alive, red when a run is
// recorded but nothing is running, dim otherwise.
func procCell(n int, runRecorded bool) string {
s := fmt.Sprintf("%6d", n)
switch {
case n > 0:
return ui.OK.Render(s)
case runRecorded:
return ui.Bad.Render(s)
default:
return ui.Muted.Render(s)
}
}
@@ -1,33 +1,33 @@
package cli
package warp
import (
"strings"
"testing"
"time"
"yuctl/internal/warp"
warpfleet "yuctl/fleet/warp"
)
func TestWarpViewRender(t *testing.T) {
v := &warpView{label: "prod@htz-fsn1 → spice · father"}
r := &warp.StatusReport{
func TestViewRender(t *testing.T) {
v := &view{label: "prod@htz-fsn1 → spice · father"}
r := &warpfleet.StatusReport{
Config: map[string]string{
"mode": "nonstop", "started_at": time.Now().Add(-2 * time.Hour).UTC().Format(time.RFC3339),
"put_streams": "1002", "get_streams": "102",
"put_obj_size": "16MiB", "get_obj_size": "16MiB",
"rgw_endpoints": "47", "endpoint": "https://s3.example",
},
Pods: []warp.PodStatus{
Pods: []warpfleet.PodStatus{
{Name: "warp-runner-a", Node: "jeanne", PutProcs: 1, GetProcs: 1},
{Name: "warp-runner-b", Node: "sheron", PutProcs: 0, GetProcs: 1, PutErrors: 3},
},
Nodes: []warp.NodeThroughput{
Nodes: []warpfleet.NodeThroughput{
{Node: "jeanne", Iface: "bond0", TxBps: 49e9, RxBps: 31e9},
{Node: "sheron", Iface: "bond0", TxBps: 50e9, RxBps: 35e9},
},
}
v.push(160e9)
v.push(165e9)
v.history.Push(160e9)
v.history.Push(165e9)
out := v.render(r, time.Now(), 5, true)
for _, want := range []string{"WARP", "nonstop", "jeanne", "bond0", "combined", "ctrl-c"} {
if !strings.Contains(out, want) {
@@ -35,20 +35,8 @@ func TestWarpViewRender(t *testing.T) {
}
}
empty := (&warpView{label: "x"}).render(&warp.StatusReport{}, time.Now(), 5, false)
empty := (&view{label: "x"}).render(&warpfleet.StatusReport{}, time.Now(), 5, false)
if !strings.Contains(empty, "no runner pods deployed") {
t.Errorf("empty render: %s", empty)
}
}
func TestSparklineAndMeter(t *testing.T) {
if got := sparkline([]float64{0, 1, 2, 4}, 10); len([]rune(got)) != 4 {
t.Errorf("sparkline length: %q", got)
}
if got := meter(50, 100, 10); !strings.HasPrefix(got, "█████░") {
t.Errorf("meter: %q", got)
}
if got := meter(0, 0, 4); got != "░░░░" {
t.Errorf("empty meter: %q", got)
}
}
@@ -1,23 +1,22 @@
package cli
package warp
import (
"context"
"fmt"
"os"
"os/signal"
"strings"
"syscall"
"time"
"github.com/rs/zerolog/log"
"github.com/spf13/cobra"
"yuctl/internal/warp"
"yuctl/cmdutil"
"yuctl/fleet"
warpfleet "yuctl/fleet/warp"
)
// warpFlags are shared by every warp subcommand: how to reach the cluster and
// flags are shared by every warp subcommand: how to reach the cluster and
// which ceph cluster's RGW fleet to target. Everything else is derived.
type warpFlags struct {
type flags struct {
namespace string
kubeconfig string
cephCluster string
@@ -25,23 +24,23 @@ type warpFlags struct {
insecure bool
}
func (f *warpFlags) register(c *cobra.Command) {
c.PersistentFlags().StringVar(&f.namespace, "namespace", "loadtest", "namespace for the runner fleet")
c.PersistentFlags().StringVar(&f.kubeconfig, "kubeconfig", "", "kubeconfig path (default: materialized from discovery via 1Password)")
c.PersistentFlags().StringVar(&f.cephCluster, "ceph-cluster", "", "target ceph cluster (default: the selected one, or the region's only one)")
c.PersistentFlags().StringVar(&f.s3Endpoint, "s3-endpoint", "", "override the RGW S3 endpoint from discovery")
c.PersistentFlags().BoolVar(&f.insecure, "insecure", true, "skip RGW TLS verification (self-signed certs)")
func (o *flags) register(c *cobra.Command) {
c.PersistentFlags().StringVar(&o.namespace, "namespace", "loadtest", "namespace for the runner fleet")
c.PersistentFlags().StringVar(&o.kubeconfig, "kubeconfig", "", "kubeconfig path (default: materialized from discovery via 1Password)")
c.PersistentFlags().StringVar(&o.cephCluster, "ceph-cluster", "", "target ceph cluster (default: the selected one, or the region's only one)")
c.PersistentFlags().StringVar(&o.s3Endpoint, "s3-endpoint", "", "override the RGW S3 endpoint from discovery")
c.PersistentFlags().BoolVar(&o.insecure, "insecure", true, "skip RGW TLS verification (self-signed certs)")
}
// session resolves context + topology and opens a warp session against the
// region's K8s cluster and chosen ceph cluster. The label describes the target
// for display ("prod@htz-fsn1 → spice · father").
func (f *warpFlags) session(ctx context.Context) (*warp.Session, string, error) {
cc, err := requireContext()
func (o *flags) session(ctx context.Context, f *cmdutil.Factory) (*warpfleet.Session, string, error) {
cc, err := f.Context()
if err != nil {
return nil, "", err
}
topo, err := resolveTopology(ctx)
topo, err := f.Topology(ctx)
if err != nil {
return nil, "", err
}
@@ -50,14 +49,14 @@ func (f *warpFlags) session(ctx context.Context) (*warp.Session, string, error)
return nil, "", fmt.Errorf("no kubernetes payload in discovery for %s@%s", cc.Partition, cc.Region)
}
clusters := topo.CephClusters(cc.Partition, cc.Region)
name := f.cephCluster
name := o.cephCluster
if name == "" {
name = cc.CephCluster
}
if name == "" {
if len(clusters) != 1 {
return nil, "", fmt.Errorf("region has %d ceph clusters (%s); `yuctl ceph select` or --ceph-cluster",
len(clusters), strings.Join(cephClusterNames(clusters), ", "))
len(clusters), strings.Join(topo.CephClusterNames(cc.Partition, cc.Region), ", "))
}
for n := range clusters {
name = n
@@ -66,15 +65,15 @@ func (f *warpFlags) session(ctx context.Context) (*warp.Session, string, error)
ceph, ok := clusters[name]
if !ok {
return nil, "", fmt.Errorf("unknown ceph cluster %q in %s@%s; known: %s",
name, cc.Partition, cc.Region, strings.Join(cephClusterNames(clusters), ", "))
name, cc.Partition, cc.Region, strings.Join(topo.CephClusterNames(cc.Partition, cc.Region), ", "))
}
label := fmt.Sprintf("%s@%s → %s · %s", cc.Partition, cc.Region, name, k8s.ClusterName)
s, err := warp.NewSession(ctx, *k8s, ceph, f.namespace, f.kubeconfig)
s, err := warpfleet.NewSession(ctx, *k8s, ceph, o.namespace, o.kubeconfig)
return s, label, err
}
func newWarpCmd() *cobra.Command {
f := &warpFlags{}
func New(f *cmdutil.Factory) *cobra.Command {
o := &flags{}
cmd := &cobra.Command{
Use: "warp",
Short: "S3 load testing against the region's RGW fleet (MinIO warp)",
@@ -84,20 +83,20 @@ func newWarpCmd() *cobra.Command {
"(workers, CPU, RGW IPs, replica count) is discovered at runtime; defaults\n" +
"reproduce the proven ~250Gbps father configuration per pod.",
}
f.register(cmd)
o.register(cmd)
cmd.AddCommand(
newWarpDeployCmd(f),
newWarpStartCmd(f),
newWarpStatusCmd(f),
newWarpWatchCmd(f),
newWarpStopCmd(f),
newWarpCleanupCmd(f),
newWarpUndeployCmd(f),
newDeployCmd(f, o),
newStartCmd(f, o),
newStatusCmd(f, o),
newWatchCmd(f, o),
newStopCmd(f, o),
newCleanupCmd(f, o),
newUndeployCmd(f, o),
)
return cmd
}
func warpDeployFlags(c *cobra.Command, o *warp.DeployOptions) {
func deployFlags(c *cobra.Command, o *warpfleet.DeployOptions) {
c.Flags().StringVar(&o.Image, "image", "minio/warp:latest", "warp container image")
c.Flags().IntVar(&o.PodsPerNode, "pods-per-node", 2, "runner pods per worker node")
c.Flags().IntVar(&o.WorkerCount, "workers", 0, "use only the first N workers (0 = all)")
@@ -106,26 +105,26 @@ func warpDeployFlags(c *cobra.Command, o *warp.DeployOptions) {
c.Flags().StringVar(&o.SourceSecret, "creds-secret", "yucca-michael", "secret holding the RGW access/secret keys")
}
func newWarpDeployCmd(f *warpFlags) *cobra.Command {
o := warp.DeployOptions{}
func newDeployCmd(f *cmdutil.Factory, fl *flags) *cobra.Command {
o := warpfleet.DeployOptions{}
cmd := &cobra.Command{
Use: "deploy",
Short: "Deploy (or converge) the runner fleet on the region's workers",
RunE: func(cmd *cobra.Command, _ []string) error {
s, _, err := f.session(cmd.Context())
s, _, err := fl.session(cmd.Context(), f)
if err != nil {
return err
}
return s.Deploy(cmd.Context(), o)
},
}
warpDeployFlags(cmd, &o)
deployFlags(cmd, &o)
return cmd
}
func newWarpStartCmd(f *warpFlags) *cobra.Command {
o := warp.StartOptions{}
d := warp.DeployOptions{}
func newStartCmd(f *cmdutil.Factory, fl *flags) *cobra.Command {
o := warpfleet.StartOptions{}
d := warpfleet.DeployOptions{}
var autoDeploy bool
cmd := &cobra.Command{
Use: "start",
@@ -138,12 +137,12 @@ func newWarpStartCmd(f *warpFlags) *cobra.Command {
"that is the proven 1002/102 ~250Gbps shape.",
RunE: func(cmd *cobra.Command, _ []string) error {
ctx := cmd.Context()
s, _, err := f.session(ctx)
s, _, err := fl.session(ctx, f)
if err != nil {
return err
}
o.Insecure = f.insecure
o.S3Override = f.s3Endpoint
o.Insecure = fl.insecure
o.S3Override = fl.s3Endpoint
if autoDeploy {
if pods, err := s.RunnerPods(ctx); err == nil && len(pods) == 0 {
if err := s.Deploy(ctx, d); err != nil {
@@ -165,17 +164,17 @@ func newWarpStartCmd(f *warpFlags) *cobra.Command {
cmd.Flags().StringVar(&o.CycleDuration, "cycle", "6h", "warp run length inside the non-stop loop")
cmd.Flags().StringVar(&o.BucketPrefix, "bucket-prefix", "yuctl-warp-", "bucket name prefix (cleanup purges by this)")
cmd.Flags().BoolVar(&autoDeploy, "auto-deploy", true, "deploy the runner fleet first if it is missing")
warpDeployFlags(cmd, &d)
deployFlags(cmd, &d)
return cmd
}
func newWarpStatusCmd(f *warpFlags) *cobra.Command {
func newStatusCmd(f *cmdutil.Factory, fl *flags) *cobra.Command {
var sample int
cmd := &cobra.Command{
Use: "status",
Short: "Show per-pod load state and per-node NIC throughput",
RunE: func(cmd *cobra.Command, _ []string) error {
s, label, err := f.session(cmd.Context())
s, label, err := fl.session(cmd.Context(), f)
if err != nil {
return err
}
@@ -184,8 +183,8 @@ func newWarpStatusCmd(f *warpFlags) *cobra.Command {
if err != nil {
return err
}
v := &warpView{label: label}
fmt.Fprintln(cmd.OutOrStdout(), v.render(report, time.Now(), sample, false))
v := &view{label: label}
fmt.Fprintln(f.IO.Out, v.render(report, time.Now(), sample, false))
return nil
},
}
@@ -193,60 +192,43 @@ func newWarpStatusCmd(f *warpFlags) *cobra.Command {
return cmd
}
func newWarpWatchCmd(f *warpFlags) *cobra.Command {
func newWatchCmd(f *cmdutil.Factory, fl *flags) *cobra.Command {
var sample int
cmd := &cobra.Command{
Use: "watch",
Short: "Live dashboard: continuously sampled load state and throughput",
RunE: func(cmd *cobra.Command, _ []string) error {
ctx, stop := signal.NotifyContext(cmd.Context(), os.Interrupt, syscall.SIGTERM)
defer stop()
s, label, err := f.session(ctx)
s, label, err := fl.session(cmd.Context(), f)
if err != nil {
return err
}
out := cmd.OutOrStdout()
fmt.Fprint(out, "\x1b[?1049h\x1b[?25l") // alt screen, hide cursor
defer fmt.Fprint(out, "\x1b[?25h\x1b[?1049l")
fmt.Fprintf(out, "%s connecting · sampling %ds window…\n", label, sample)
v := &warpView{label: label}
for {
v := &view{label: label}
return fleet.Watch(cmd.Context(), f.IO.Out, label, sample, func(ctx context.Context) (string, error) {
report, err := s.Status(ctx, sample)
if ctx.Err() != nil {
return nil
}
fmt.Fprint(out, "\x1b[H\x1b[2J")
if err != nil {
fmt.Fprintf(out, "status error (retrying): %v\n", err)
select {
case <-ctx.Done():
return nil
case <-time.After(3 * time.Second):
}
continue
return "", err
}
var combined float64
for _, n := range report.Nodes {
combined += n.TxBps + n.RxBps
}
if len(report.Nodes) > 0 {
v.push(combined)
v.history.Push(combined)
}
fmt.Fprintln(out, v.render(report, time.Now(), sample, true))
}
return v.render(report, time.Now(), sample, true), nil
})
},
}
cmd.Flags().IntVar(&sample, "sample", 5, "NIC sampling window per refresh in seconds")
return cmd
}
func newWarpStopCmd(f *warpFlags) *cobra.Command {
func newStopCmd(f *cmdutil.Factory, fl *flags) *cobra.Command {
return &cobra.Command{
Use: "stop",
Short: "Stop the load everywhere (runners stay deployed for instant restart)",
RunE: func(cmd *cobra.Command, _ []string) error {
s, _, err := f.session(cmd.Context())
s, _, err := fl.session(cmd.Context(), f)
if err != nil {
return err
}
@@ -255,8 +237,8 @@ func newWarpStopCmd(f *warpFlags) *cobra.Command {
}
}
func newWarpCleanupCmd(f *warpFlags) *cobra.Command {
o := warp.CleanupOptions{}
func newCleanupCmd(f *cmdutil.Factory, fl *flags) *cobra.Command {
o := warpfleet.CleanupOptions{}
var timeout time.Duration
cmd := &cobra.Command{
Use: "cleanup",
@@ -266,13 +248,14 @@ func newWarpCleanupCmd(f *warpFlags) *cobra.Command {
"while load is active unless --force. Note: a mass delete is itself a load\n" +
"event — RGW garbage collection churns for a while afterwards.",
RunE: func(cmd *cobra.Command, _ []string) error {
s, _, err := f.session(cmd.Context())
s, _, err := fl.session(cmd.Context(), f)
if err != nil {
return err
}
o.Insecure = f.insecure
o.S3Override = f.s3Endpoint
o.Insecure = fl.insecure
o.S3Override = fl.s3Endpoint
o.Timeout = timeout
o.JobLogs = f.IO.Out
return s.Cleanup(cmd.Context(), o)
},
}
@@ -285,13 +268,13 @@ func newWarpCleanupCmd(f *warpFlags) *cobra.Command {
return cmd
}
func newWarpUndeployCmd(f *warpFlags) *cobra.Command {
func newUndeployCmd(f *cmdutil.Factory, fl *flags) *cobra.Command {
var force bool
cmd := &cobra.Command{
Use: "undeploy",
Short: "Delete the runner fleet and its namespace entirely",
RunE: func(cmd *cobra.Command, _ []string) error {
s, _, err := f.session(cmd.Context())
s, _, err := fl.session(cmd.Context(), f)
if err != nil {
return err
}
@@ -0,0 +1,176 @@
package allowlist
import (
"fmt"
"strconv"
"strings"
"text/tabwriter"
"github.com/spf13/cobra"
"yuctl/adminapi"
"yuctl/cmdutil"
)
func New(f *cmdutil.Factory) *cobra.Command {
cmd := &cobra.Command{
Use: "allowlist",
Short: "Manage the beta email allowlist and invites",
}
cmd.AddCommand(
newListCmd(f),
newAddCmd(f),
newRemoveCmd(f),
newInviteCmd(f),
newInviteBatchCmd(f),
)
return cmd
}
func printEntries(f *cmdutil.Factory, entries []adminapi.AllowlistEntry) {
w := tabwriter.NewWriter(f.IO.Out, 0, 2, 2, ' ', 0)
fmt.Fprintln(w, "EMAIL\tCODE\tINVITED\tUSED\tUSED AT\tCREATED")
for _, e := range entries {
usedAt := ""
if e.InviteUsedAt != nil {
usedAt = *e.InviteUsedAt
}
fmt.Fprintf(w, "%s\t%s\t%t\t%t\t%s\t%s\n", e.Email, e.InviteCode, e.Invited, e.InviteUsed, usedAt, e.CreatedAt)
}
w.Flush()
}
func newListCmd(f *cmdutil.Factory) *cobra.Command {
admin := &cmdutil.AdminFlags{}
var limitFlag string
c := &cobra.Command{
Use: "list",
Short: "List allowlist entries",
RunE: func(cmd *cobra.Command, args []string) error {
limit, err := adminapi.ParseLimit(limitFlag)
if err != nil {
return err
}
client, partition, err := admin.Client(cmd.Context(), f)
if err != nil {
return err
}
entries, err := client.ListAllowlist(cmd.Context(), limit)
if err != nil {
return err
}
printEntries(f, entries)
fmt.Fprintf(f.IO.Err, "\n%d entries in partition %s\n", len(entries), partition)
return nil
},
}
c.Flags().StringVar(&limitFlag, "limit", "", "page size for the admin-api (default: server default)")
admin.Register(c)
return c
}
func newAddCmd(f *cmdutil.Factory) *cobra.Command {
admin := &cmdutil.AdminFlags{}
var staged bool
c := &cobra.Command{
Use: "add <email>",
Short: "Allow an email to sign up (--staged to waitlist it instead)",
Args: cobra.ExactArgs(1),
RunE: func(cmd *cobra.Command, args []string) error {
client, _, err := admin.Client(cmd.Context(), f)
if err != nil {
return err
}
entry, err := client.AddAllowlistEntry(cmd.Context(), args[0], staged)
if err != nil {
return err
}
printEntries(f, []adminapi.AllowlistEntry{*entry})
return nil
},
}
c.Flags().BoolVar(&staged, "staged", false, "stage the email without allowing login yet")
admin.Register(c)
return c
}
func newRemoveCmd(f *cmdutil.Factory) *cobra.Command {
admin := &cmdutil.AdminFlags{}
c := &cobra.Command{
Use: "remove <email>",
Short: "Remove an email from the allowlist",
Args: cobra.ExactArgs(1),
RunE: func(cmd *cobra.Command, args []string) error {
client, _, err := admin.Client(cmd.Context(), f)
if err != nil {
return err
}
if err := client.RemoveAllowlistEntry(cmd.Context(), args[0]); err != nil {
return err
}
fmt.Fprintf(f.IO.Err, "removed %s\n", args[0])
return nil
},
}
admin.Register(c)
return c
}
func newInviteCmd(f *cmdutil.Factory) *cobra.Command {
admin := &cmdutil.AdminFlags{}
c := &cobra.Command{
Use: "invite <email>[,<email>...]",
Short: "Invite emails: allow them to sign up, creating entries as needed",
Args: cobra.MinimumNArgs(1),
RunE: func(cmd *cobra.Command, args []string) error {
var emails []string
for _, arg := range args {
for _, email := range strings.Split(arg, ",") {
if email = strings.TrimSpace(email); email != "" {
emails = append(emails, email)
}
}
}
client, _, err := admin.Client(cmd.Context(), f)
if err != nil {
return err
}
entries, err := client.InviteEmails(cmd.Context(), emails)
if err != nil {
return err
}
printEntries(f, entries)
return nil
},
}
admin.Register(c)
return c
}
func newInviteBatchCmd(f *cmdutil.Factory) *cobra.Command {
admin := &cmdutil.AdminFlags{}
c := &cobra.Command{
Use: "invite-batch <count>",
Short: "Invite the oldest <count> staged (waitlisted) emails",
Args: cobra.ExactArgs(1),
RunE: func(cmd *cobra.Command, args []string) error {
count, err := strconv.Atoi(args[0])
if err != nil || count < 1 {
return fmt.Errorf("count must be a positive integer")
}
client, _, err := admin.Client(cmd.Context(), f)
if err != nil {
return err
}
entries, err := client.InviteBatch(cmd.Context(), count)
if err != nil {
return err
}
printEntries(f, entries)
fmt.Fprintf(f.IO.Err, "\ninvited %d entries\n", len(entries))
return nil
},
}
admin.Register(c)
return c
}
@@ -0,0 +1,57 @@
package connections
import (
"fmt"
"text/tabwriter"
"github.com/spf13/cobra"
"yuctl/cmdutil"
)
func New(f *cmdutil.Factory) *cobra.Command {
cmd := &cobra.Command{
Use: "connections",
Short: "A user's connection instances (immich/restic)",
}
cmd.AddCommand(newListCmd(f))
return cmd
}
func newListCmd(f *cmdutil.Factory) *cobra.Command {
admin := &cmdutil.AdminFlags{}
c := &cobra.Command{
Use: "list <email>",
Short: "List a user's connections",
Args: cobra.ExactArgs(1),
RunE: func(cmd *cobra.Command, args []string) error {
ctx := cmd.Context()
client, _, err := admin.Client(ctx, f)
if err != nil {
return err
}
userID, err := client.ResolveUserID(ctx, args[0])
if err != nil {
return err
}
connections, err := client.GetUserConnections(ctx, userID)
if err != nil {
return err
}
w := tabwriter.NewWriter(f.IO.Out, 0, 2, 2, ' ', 0)
fmt.Fprintln(w, "ID\tTYPE\tNAME\tCREATED\tLAST SEEN")
for _, connection := range connections {
lastSeen := ""
if connection.LastSeenAt != nil {
lastSeen = *connection.LastSeenAt
}
fmt.Fprintf(w, "%s\t%s\t%s\t%s\t%s\n", connection.ID, connection.Type, connection.Name, connection.CreatedAt, lastSeen)
}
w.Flush()
return nil
},
}
admin.Register(c)
return c
}
@@ -0,0 +1,133 @@
package features
import (
"fmt"
"text/tabwriter"
"github.com/spf13/cobra"
"yuctl/adminapi"
"yuctl/cmdutil"
)
func New(f *cmdutil.Factory) *cobra.Command {
cmd := &cobra.Command{
Use: "features",
Short: "Per-user feature-flag overrides",
}
cmd.AddCommand(newListCmd(f), newSetCmd(f), newClearCmd(f))
return cmd
}
func newListCmd(f *cmdutil.Factory) *cobra.Command {
admin := &cmdutil.AdminFlags{}
c := &cobra.Command{
Use: "list <email>",
Short: "Show a user's resolved feature flags and overrides",
Args: cobra.ExactArgs(1),
RunE: func(cmd *cobra.Command, args []string) error {
ctx := cmd.Context()
client, _, err := admin.Client(ctx, f)
if err != nil {
return err
}
userID, err := client.ResolveUserID(ctx, args[0])
if err != nil {
return err
}
features, err := client.GetUserFeatures(ctx, userID)
if err != nil {
return err
}
overridden := map[string]adminapi.FeatureOverride{}
for _, o := range features.Overrides {
overridden[o.Flag] = o
}
w := tabwriter.NewWriter(f.IO.Out, 0, 2, 2, ' ', 0)
fmt.Fprintln(w, "FLAG\tVALUE\tSOURCE\tSET BY\tREASON")
for flag, value := range features.Features {
if o, ok := overridden[flag]; ok {
reason := ""
if o.Reason != nil {
reason = *o.Reason
}
fmt.Fprintf(w, "%s\t%t\toverride\t%s\t%s\n", flag, value, o.SetBy, reason)
} else {
fmt.Fprintf(w, "%s\t%t\tdefault\t\t\n", flag, value)
}
}
w.Flush()
return nil
},
}
admin.Register(c)
return c
}
func newSetCmd(f *cmdutil.Factory) *cobra.Command {
admin := &cmdutil.AdminFlags{}
var reason string
c := &cobra.Command{
Use: "set <email> <flag> on|off",
Short: "Set a per-user feature-flag override",
Args: cobra.ExactArgs(3),
RunE: func(cmd *cobra.Command, args []string) error {
var value bool
switch args[2] {
case "on", "true":
value = true
case "off", "false":
value = false
default:
return fmt.Errorf("value must be on|off, got %q", args[2])
}
ctx := cmd.Context()
client, _, err := admin.Client(ctx, f)
if err != nil {
return err
}
userID, err := client.ResolveUserID(ctx, args[0])
if err != nil {
return err
}
if _, err := client.SetUserFeature(ctx, userID, args[1], value, reason); err != nil {
return err
}
fmt.Fprintf(f.IO.Err, "%s set to %t for %s\n", args[1], value, args[0])
return nil
},
}
c.Flags().StringVar(&reason, "reason", "", "audit note stored on the override")
admin.Register(c)
return c
}
func newClearCmd(f *cmdutil.Factory) *cobra.Command {
admin := &cmdutil.AdminFlags{}
c := &cobra.Command{
Use: "clear <email> <flag>",
Short: "Clear an override (revert to the registry default)",
Args: cobra.ExactArgs(2),
RunE: func(cmd *cobra.Command, args []string) error {
ctx := cmd.Context()
client, _, err := admin.Client(ctx, f)
if err != nil {
return err
}
userID, err := client.ResolveUserID(ctx, args[0])
if err != nil {
return err
}
if err := client.ClearUserFeature(ctx, userID, args[1]); err != nil {
return err
}
fmt.Fprintf(f.IO.Err, "%s override cleared for %s\n", args[1], args[0])
return nil
},
}
admin.Register(c)
return c
}
+124
View File
@@ -0,0 +1,124 @@
package users
import (
"fmt"
"net/url"
"os"
"strings"
"text/tabwriter"
"github.com/spf13/cobra"
"yuctl/adminapi"
"yuctl/cli/users/allowlist"
"yuctl/cli/users/connections"
ufeatures "yuctl/cli/users/features"
"yuctl/cmdutil"
)
func New(f *cmdutil.Factory) *cobra.Command {
cmd := &cobra.Command{
Use: "users",
Short: "User administration via yucca-admin-api",
}
cmd.AddCommand(
newListCmd(f),
allowlist.New(f),
newViewDashboardCmd(f),
ufeatures.New(f),
connections.New(f),
)
return cmd
}
func newListCmd(f *cmdutil.Factory) *cobra.Command {
admin := &cmdutil.AdminFlags{}
var limitFlag string
c := &cobra.Command{
Use: "list",
Short: "List users in the partition's primary region",
RunE: func(cmd *cobra.Command, args []string) error {
ctx := cmd.Context()
limit, err := adminapi.ParseLimit(limitFlag)
if err != nil {
return err
}
client, partition, err := admin.Client(ctx, f)
if err != nil {
return err
}
users, err := client.ListUsers(ctx, limit)
if err != nil {
return err
}
w := tabwriter.NewWriter(f.IO.Out, 0, 2, 2, ' ', 0)
fmt.Fprintln(w, "ID\tSUB\tNAME\tEMAIL\tDISABLED")
for _, u := range users {
fmt.Fprintf(w, "%s\t%s\t%s\t%s\t%t\n", u.ID, u.Sub, u.Name, u.Email, u.Disabled)
}
w.Flush()
fmt.Fprintf(f.IO.Err, "\n%d user(s) in partition %s\n", len(users), partition)
return nil
},
}
c.Flags().StringVar(&limitFlag, "limit", "", "page size for the admin-api (default: server default)")
admin.Register(c)
return c
}
const defaultGrafanaURL = "https://grafana.futostatus.com"
func newViewDashboardCmd(f *cmdutil.Factory) *cobra.Command {
admin := &cmdutil.AdminFlags{}
var userID, email, grafanaURL string
var noOpen bool
c := &cobra.Command{
Use: "view-dashboard",
Short: "Open the per-user Grafana dashboard for a user",
Long: "Build the Grafana per-user drill-down URL (dashboard uid yucca-per-user) for\n" +
"a user and open it in the browser. --id builds the URL without contacting the\n" +
"admin-api; --email resolves the user via the partition's admin-api first.",
Args: cobra.NoArgs,
RunE: func(cmd *cobra.Command, _ []string) error {
id := userID
if email != "" {
client, partition, err := admin.Client(cmd.Context(), f)
if err != nil {
return err
}
id, err = client.ResolveUserID(cmd.Context(), email)
if err != nil {
return fmt.Errorf("%w in partition %s", err, partition)
}
}
base := grafanaURL
if base == "" {
base = os.Getenv("YUCTL_GRAFANA_URL")
}
if base == "" {
base = defaultGrafanaURL
}
dashboardURL := strings.TrimRight(base, "/") + "/d/yucca-per-user?var-user=" + url.QueryEscape(id)
fmt.Fprintln(f.IO.Out, dashboardURL)
if noOpen {
return nil
}
if err := cmdutil.OpenBrowser(dashboardURL); err != nil {
return fmt.Errorf("open browser: %w", err)
}
return nil
},
}
c.Flags().StringVar(&userID, "id", "", "user id (uuid)")
c.Flags().StringVar(&email, "email", "", "user email; resolved to an id via the admin-api")
c.Flags().StringVar(&grafanaURL, "grafana-url", "", "Grafana base URL (default: $YUCTL_GRAFANA_URL or "+defaultGrafanaURL+")")
c.Flags().BoolVar(&noOpen, "no-open", false, "print the dashboard URL instead of opening the browser")
c.MarkFlagsOneRequired("id", "email")
c.MarkFlagsMutuallyExclusive("id", "email")
admin.Register(c)
return c
}
+15 -15
View File
@@ -1,8 +1,8 @@
// Command bench-agent is the remote half of `yuctl tools bench` and
// `yuctl tools bench-do`. In bench mode (no args) it reads a bench.Config as
// `yuctl tools fleet-bench`. In bench mode (no args) it reads a resticbench.Config as
// JSON on stdin, executes the benchmark phases, and streams JSON events on
// stdout. In loadgen mode (--loadgen [config-path]) it reads a
// bench.LoadgenConfig — from the file, which it deletes after reading, or
// resticbench.LoadgenConfig — from the file, which it deletes after reading, or
// from stdin when no path is given — and either supervises the detached
// continuous load or runs the synchronous repo cleanup. The orchestrator
// pushes it over ssh; it is embedded into yuctl by the build task.
@@ -16,7 +16,7 @@ import (
"os/signal"
"syscall"
"yuctl/internal/bench"
"yuctl/resticbench"
)
func main() {
@@ -31,20 +31,20 @@ func main() {
return
}
emit := bench.EmitJSON(os.Stdout)
var cfg bench.Config
emit := resticbench.EmitJSON(os.Stdout)
var cfg resticbench.Config
if err := json.NewDecoder(os.Stdin).Decode(&cfg); err != nil {
emit(bench.Event{Type: "fatal", Message: "read config from stdin: " + err.Error()})
emit(resticbench.Event{Type: "fatal", Message: "read config from stdin: " + err.Error()})
os.Exit(1)
}
if err := bench.RunAgent(ctx, cfg, emit); err != nil {
emit(bench.Event{Type: "fatal", Message: err.Error()})
if err := resticbench.RunAgent(ctx, cfg, emit); err != nil {
emit(resticbench.Event{Type: "fatal", Message: err.Error()})
os.Exit(1)
}
}
func loadgen(ctx context.Context, args []string) error {
var cfg bench.LoadgenConfig
var cfg resticbench.LoadgenConfig
if len(args) > 0 {
// Config file: written 0600 by the launcher, consumed exactly once.
b, err := os.ReadFile(args[0])
@@ -59,14 +59,14 @@ func loadgen(ctx context.Context, args []string) error {
return fmt.Errorf("read config from stdin: %w", err)
}
if cfg.Op == bench.LoadgenOpCleanup {
emit := bench.EmitJSON(os.Stdout)
if err := bench.RunLoadgenCleanup(ctx, cfg, emit); err != nil {
emit(bench.Event{Type: "fatal", Message: err.Error()})
if cfg.Op == resticbench.LoadgenOpCleanup {
emit := resticbench.EmitJSON(os.Stdout)
if err := resticbench.RunLoadgenCleanup(ctx, cfg, emit); err != nil {
emit(resticbench.Event{Type: "fatal", Message: err.Error()})
os.Exit(1)
}
emit(bench.Event{Type: "result", Result: &bench.RunResult{}})
emit(resticbench.Event{Type: "result", Result: &resticbench.RunResult{}})
return nil
}
return bench.RunLoadgen(ctx, cfg)
return resticbench.RunLoadgen(ctx, cfg)
}
+147
View File
@@ -0,0 +1,147 @@
package cmdutil
import (
"context"
"crypto/tls"
"fmt"
"net/http"
"net/url"
"os"
"strings"
"time"
"github.com/spf13/cobra"
"yuctl/adminapi"
"yuctl/ctxstore"
"yuctl/discovery"
)
// AdminFlags is the shared flag set for commands that talk to the admin-api.
type AdminFlags struct {
URL string
Insecure bool
Reauth bool
NoBrowser bool
}
func (a *AdminFlags) Register(c *cobra.Command) {
c.Flags().StringVar(&a.URL, "admin-url", "", "admin-api base URL (default: derived from discovery, or $YUCTL_ADMIN_API_URL)")
c.Flags().BoolVar(&a.Insecure, "insecure-skip-tls-verify", false, "skip TLS verification")
c.Flags().BoolVar(&a.Reauth, "reauth", false, "force a fresh browser login")
c.Flags().BoolVar(&a.NoBrowser, "no-browser", false, "do not auto-open the browser; just print the login URL")
}
func (a *AdminFlags) httpClient() *http.Client {
hc := &http.Client{Timeout: 30 * time.Second}
if a.Insecure {
tr := http.DefaultTransport.(*http.Transport).Clone()
tr.TLSClientConfig = &tls.Config{InsecureSkipVerify: true}
hc.Transport = tr
}
return hc
}
// Client returns an authenticated admin-api client and the selected partition.
func (a *AdminFlags) Client(ctx context.Context, f *Factory) (*adminapi.Client, string, error) {
cc, err := f.Context()
if err != nil {
return nil, "", err
}
topo, err := f.Topology(ctx)
if err != nil {
return nil, "", err
}
client, _, err := a.Login(ctx, f, cc, topo)
if err != nil {
return nil, "", err
}
return client, cc.Partition, nil
}
// Login returns an authenticated admin-api client, reusing the cached
// per-partition session when valid and running the browser loopback flow
// otherwise. topo may be nil when the admin URL comes from --admin-url or
// $YUCTL_ADMIN_API_URL.
func (a *AdminFlags) Login(ctx context.Context, f *Factory, cc *ctxstore.Context, topo *discovery.Topology) (*adminapi.Client, *adminapi.Token, error) {
adminURL, err := a.resolveAdminURL(topo, cc)
if err != nil {
return nil, nil, err
}
hc := a.httpClient()
token, err := adminapi.LoadToken(cc.Partition)
if err != nil {
return nil, nil, err
}
if a.Reauth || !token.Valid() {
var openFn func(string) error
if !a.NoBrowser {
openFn = OpenBrowser
}
fresh, err := adminapi.BrowserLogin(ctx, hc, adminURL, openFn, func(loginURL string) {
fmt.Fprintln(f.IO.Err, "Complete the login in your browser:")
fmt.Fprintf(f.IO.Err, " %s\n", loginURL)
})
if err != nil {
return nil, nil, fmt.Errorf("browser login: %w", err)
}
token = *fresh
if err := adminapi.SaveToken(cc.Partition, token); err != nil {
return nil, nil, err
}
}
return adminapi.NewClient(adminURL, token, hc), &token, nil
}
// OptionalTopology resolves the topology only when Login will need it to
// derive the admin URL — an explicit --admin-url or $YUCTL_ADMIN_API_URL
// skips state access entirely (the context file alone names the token-cache
// partition).
func (a *AdminFlags) OptionalTopology(ctx context.Context, f *Factory) (*discovery.Topology, error) {
if a.URL != "" || os.Getenv("YUCTL_ADMIN_API_URL") != "" {
return nil, nil
}
return f.Topology(ctx)
}
// resolveAdminURL applies the override chain: --admin-url > $YUCTL_ADMIN_API_URL
// > derived from the primary region's discovery.
func (a *AdminFlags) resolveAdminURL(topo *discovery.Topology, cc *ctxstore.Context) (string, error) {
if a.URL != "" {
return a.URL, nil
}
if env := os.Getenv("YUCTL_ADMIN_API_URL"); env != "" {
return env, nil
}
if topo == nil {
return "", fmt.Errorf("no topology available to derive the admin-api URL; pass --admin-url or set YUCTL_ADMIN_API_URL")
}
primary := topo.PrimaryRegion(cc.Partition)
if primary == "" {
return "", fmt.Errorf("no primary region found for partition %q (no stack with discovery.role==\"primary\")", cc.Partition)
}
return deriveAdminURL(topo, cc.Partition, primary)
}
// deriveAdminURL builds the admin-api origin for the partition's primary
// region from its k8s discovery payload: the Talos API endpoint lives at
// kube.<cluster>.<region>.<provider>.yucca.futo.network, and the admin-api is
// published on the same NetBird overlay zone as admin.<...> (the
// YUCCA_ADMIN_HOST cluster-setting follows the same convention).
func deriveAdminURL(topo *discovery.Topology, partition, region string) (string, error) {
k8s := topo.Kubernetes(partition, region)
if k8s == nil || k8s.APIEndpoint == "" {
return "", fmt.Errorf("no kubernetes discovery payload for %s@%s; pass --admin-url or set YUCTL_ADMIN_API_URL", partition, region)
}
u, err := url.Parse(k8s.APIEndpoint)
if err != nil {
return "", fmt.Errorf("parse api_endpoint %q: %w", k8s.APIEndpoint, err)
}
host, ok := strings.CutPrefix(u.Hostname(), "kube.")
if !ok {
return "", fmt.Errorf("api_endpoint host %q does not start with kube.; pass --admin-url or set YUCTL_ADMIN_API_URL", u.Hostname())
}
return "https://admin." + host, nil
}
+50
View File
@@ -0,0 +1,50 @@
// Package cmdutil sits between cli and the domain packages: command packages
// receive a *Factory instead of re-deriving shared dependencies per command,
// and domain packages must never import it.
package cmdutil
import (
"bufio"
"context"
"fmt"
"os/exec"
"runtime"
"strings"
"yuctl/ctxstore"
"yuctl/discovery"
"yuctl/ui"
)
// Factory's closures are lazy — a command that never touches topology never
// reads Terraform state — and memoized per invocation where resolution is
// expensive.
type Factory struct {
IO *ui.IOStreams
// Context errors when nothing is selected (`yuctl select` first).
Context func() (*ctxstore.Context, error)
Topology func(ctx context.Context) (*discovery.Topology, error)
}
// Confirm prompts the operator for a yes/no answer, defaulting to no.
func Confirm(io *ui.IOStreams, prompt string) bool {
fmt.Fprintf(io.Err, "%s [y/N]: ", prompt)
line, err := bufio.NewReader(io.In).ReadString('\n')
if err != nil {
return false
}
answer := strings.ToLower(strings.TrimSpace(line))
return answer == "y" || answer == "yes"
}
// OpenBrowser opens url in the OS default browser, best-effort.
func OpenBrowser(url string) error {
switch runtime.GOOS {
case "darwin":
return exec.Command("open", url).Start()
default:
return exec.Command("xdg-open", url).Start()
}
}
@@ -1,8 +1,8 @@
// Package context persists the operator's selected topology to
// Package ctxstore persists the operator's selected topology to
// ${XDG_CONFIG_HOME:-~/.config}/yuctl/context.json. It is the small bit of
// sticky state that lets `yuctl select prod@htz-fsn1` be followed by
// `yuctl ceph get health` without re-specifying the target each time.
package context
package ctxstore
import (
"encoding/json"
@@ -8,7 +8,7 @@ import (
"github.com/rs/zerolog"
yctx "yuctl/internal/context"
"yuctl/ctxstore"
)
// The resolved topology is cached on disk so repeat invocations skip the
@@ -31,7 +31,7 @@ type cacheEnvelope struct {
}
func cachePath() (string, error) {
dir, err := yctx.Dir()
dir, err := ctxstore.Dir()
if err != nil {
return "", err
}
@@ -8,7 +8,7 @@ import (
"github.com/rs/zerolog"
"yuctl/internal/state"
"yuctl/state"
)
func testTopology() *Topology {
@@ -1,6 +1,6 @@
// Package discovery resolves the live yucca topology by reading Terraform state
// objects directly from the shared `yucca-tf-state` S3 bucket and parsing each
// stack's `discovery` output (see internal/state). It never shells out to
// stack's `discovery` output (see yuctl/state). It never shells out to
// `terragrunt output`, so it works without a checkout, provider plugins, or
// `terragrunt init`.
//
@@ -14,16 +14,18 @@ import (
"context"
"fmt"
"io"
"maps"
"os"
"path/filepath"
"slices"
"sort"
"strings"
"sync"
"github.com/rs/zerolog"
"yuctl/internal/op"
"yuctl/internal/state"
"yuctl/op"
"yuctl/state"
"github.com/aws/aws-sdk-go-v2/aws"
"github.com/aws/aws-sdk-go-v2/credentials"
@@ -386,6 +388,11 @@ func (t *Topology) CephClusters(partition, region string) map[string]state.CephC
return out
}
// CephClusterNames returns the region's ceph cluster names, sorted.
func (t *Topology) CephClusterNames(partition, region string) []string {
return slices.Sorted(maps.Keys(t.CephClusters(partition, region)))
}
// MgmtHosts returns the region's management-host roster (fabric payload), or
// nil for regions without a fabric stack (e.g. austin).
func (t *Topology) MgmtHosts(partition, region string) []state.MgmtHost {
@@ -4,7 +4,7 @@ import (
"reflect"
"testing"
"yuctl/internal/state"
"yuctl/state"
)
func TestStackFromKey(t *testing.T) {
@@ -1,4 +1,4 @@
// Package do wraps the DigitalOcean API (godo) for the bench-do droplet
// Package do wraps the DigitalOcean API (godo) for the fleet-bench droplet
// fleet: token resolution via 1Password, the yucca-bench project, ephemeral
// SSH keys, and tagged droplet lifecycle. Nothing here knows about restic or
// michael — it is pure fleet plumbing.
@@ -14,7 +14,7 @@ import (
"github.com/digitalocean/godo"
"github.com/rs/zerolog/log"
"yuctl/internal/op"
"yuctl/op"
)
const (
@@ -74,7 +74,7 @@ func (c *Client) EnsureProject(ctx context.Context) (string, error) {
p, _, err := c.do.Projects.Create(ctx, &godo.CreateProjectRequest{
Name: ProjectName,
Purpose: "Operational / Developer tooling",
Description: "yuctl tools bench-do restic load-test fleets",
Description: "yuctl tools fleet-bench restic load-test fleets",
Environment: "Development",
})
if err != nil {
@@ -185,7 +185,7 @@ func (c *Client) GetSize(ctx context.Context, slug string) (*Size, error) {
return nil, fmt.Errorf("unknown DO size slug %q", slug)
}
// Droplet is the subset of droplet state bench-do tracks.
// Droplet is the subset of droplet state fleet-bench tracks.
type Droplet struct {
ID int
Name string
+82
View File
@@ -0,0 +1,82 @@
// Package fleet holds what the load-test fleet tools (fleet/warp on K8s
// pods, fleet/fleetbench on cloud VMs) share; anything transport-specific
// stays in the subpackages.
package fleet
import (
"context"
"fmt"
"io"
"os"
"os/signal"
"sync"
"syscall"
"time"
)
type History struct {
vals []float64
}
func (h *History) Push(v float64) {
h.vals = append(h.vals, v)
if len(h.vals) > 60 {
h.vals = h.vals[len(h.vals)-60:]
}
}
func (h *History) Values() []float64 { return h.vals }
func Footer(sampleSec int, sampledAt time.Time, watching bool) string {
footer := fmt.Sprintf("sampled %ds window at %s", sampleSec, sampledAt.Format("15:04:05"))
if watching {
footer += " · ctrl-c to quit"
}
return footer
}
// Watch redraws frame's result on the alternate screen until interrupted; a
// frame error is shown and retried rather than ending the watch.
func Watch(ctx context.Context, out io.Writer, label string, sample int, frame func(context.Context) (string, error)) error {
ctx, stop := signal.NotifyContext(ctx, os.Interrupt, syscall.SIGTERM)
defer stop()
fmt.Fprint(out, "\x1b[?1049h\x1b[?25l")
defer fmt.Fprint(out, "\x1b[?25h\x1b[?1049l")
fmt.Fprintf(out, "%s connecting · sampling %ds window…\n", label, sample)
for {
s, err := frame(ctx)
if ctx.Err() != nil {
return nil
}
fmt.Fprint(out, "\x1b[H\x1b[2J")
if err != nil {
fmt.Fprintf(out, "status error (retrying): %v\n", err)
select {
case <-ctx.Done():
return nil
case <-time.After(3 * time.Second):
}
continue
}
fmt.Fprintln(out, s)
}
}
// Each runs fn for every index concurrently; every index completes even when
// one fails, and the first error is returned.
func Each(n int, fn func(int) error) error {
var wg sync.WaitGroup
errs := make([]error, n)
for i := range n {
wg.Go(func() { errs[i] = fn(i) })
}
wg.Wait()
for _, err := range errs {
if err != nil {
return err
}
}
return nil
}
@@ -1,4 +1,4 @@
package benchwide
package fleetbench
import (
"context"
@@ -7,7 +7,6 @@ import (
"encoding/pem"
"fmt"
"os"
"os/exec"
"sort"
"strings"
"sync/atomic"
@@ -16,14 +15,15 @@ import (
"github.com/rs/zerolog/log"
"golang.org/x/crypto/ssh"
"yuctl/internal/bench"
"yuctl/internal/provider"
"yuctl/provider"
"yuctl/resticbench"
"yuctl/sshx"
)
// DeployOptions shape the host fleet. Regions/Size/Image default to the
// provider's own defaults when left empty (resolved in Deploy).
type DeployOptions struct {
Droplets int // fleet size (default 3)
Hosts int // fleet size (default 3)
Regions []string // round-robined; provider default if empty
Size string // size slug; provider default if empty
Image string // image slug; provider default if empty
@@ -35,8 +35,8 @@ type DeployOptions struct {
}
func (o *DeployOptions) defaults(d provider.Defaults) {
if o.Droplets <= 0 {
o.Droplets = 3
if o.Hosts <= 0 {
o.Hosts = 3
}
if len(o.Regions) == 0 {
o.Regions = d.Regions
@@ -49,13 +49,13 @@ func (o *DeployOptions) defaults(d provider.Defaults) {
}
}
// dropletName is the fleet-stable name of host n (1-based) — reconciliation
// hostName is the fleet-stable name of host n (1-based) — reconciliation
// keys on these names, so deploy converges instead of duplicating.
func (s *Session) dropletName(n int) string {
func (s *Session) hostName(n int) string {
return fmt.Sprintf("yucca-bench-%s-%02d", s.slug(), n)
}
// Deploy converges the fleet: creates missing droplets (confirmed, with the
// Deploy converges the fleet: creates missing hosts (confirmed, with the
// hourly cost and transfer pool spelled out), trims surplus ones, files
// everything under the yucca-bench project, waits for ssh, and pushes the
// bench agent + pinned restic everywhere.
@@ -66,15 +66,15 @@ func (s *Session) Deploy(ctx context.Context, opts DeployOptions) error {
if err != nil {
return err
}
live, err := s.Droplets(ctx)
live, err := s.Hosts(ctx)
if err != nil {
return err
}
want := map[string]string{} // name -> region
var wantNames []string
for i := range opts.Droplets {
name := s.dropletName(i + 1)
for i := range opts.Hosts {
name := s.hostName(i + 1)
want[name] = opts.Regions[i%len(opts.Regions)]
wantNames = append(wantNames, name)
}
@@ -112,8 +112,8 @@ func (s *Session) Deploy(ctx context.Context, opts DeployOptions) error {
" transfer allowance: %s per host → %s pool (hard cap enforced by the agent)",
len(missing), size.Slug, opts.Image, s.Provider.Name(),
strings.Join(regions, ", "),
size.PriceHourly, size.PriceHourly*float64(opts.Droplets), size.PriceHourly*float64(opts.Droplets)*730, opts.Droplets,
bench.FormatBytes(size.TransferBytes), bench.FormatBytes(size.TransferBytes*int64(opts.Droplets)))
size.PriceHourly, size.PriceHourly*float64(opts.Hosts), size.PriceHourly*float64(opts.Hosts)*730, opts.Hosts,
resticbench.FormatBytes(size.TransferBytes), resticbench.FormatBytes(size.TransferBytes*int64(opts.Hosts)))
if opts.Confirm != nil && !opts.Confirm(plan) {
return fmt.Errorf("deploy aborted")
}
@@ -143,42 +143,43 @@ func (s *Session) Deploy(ctx context.Context, opts DeployOptions) error {
}
}
droplets, err := s.Provider.WaitActive(ctx, s.Tag(), opts.Droplets, 5*time.Minute)
hosts, err := s.Provider.WaitActive(ctx, s.Tag(), opts.Hosts, 5*time.Minute)
if err != nil {
return err
}
// Filing under the project is organizational only — the tag is what every
// fleet operation keys on — so a scoped token without project permissions
// must not strand freshly created (billing!) droplets mid-deploy. Assign
// the whole fleet on every converge; it is idempotent, so a run after a
// token fix picks the droplets up.
s.assignProject(ctx, droplets)
// must not strand freshly created (billing!) hosts mid-deploy. Assign the
// whole fleet on every converge; it is idempotent, so a run after a token
// fix picks the hosts up.
s.assignProject(ctx, hosts)
log.Info().Int("droplets", len(droplets)).Msg("waiting for ssh")
log.Info().Int("hosts", len(hosts)).Msg("waiting for ssh")
var sshReady atomic.Int32
if err := eachDroplet(droplets, func(d provider.Host) error {
if err := eachHost(hosts, func(d provider.Host) error {
if err := s.waitSSH(ctx, d, 4*time.Minute); err != nil {
return err
}
log.Info().Str("droplet", d.Name).Int32("ready", sshReady.Add(1)).Int("want", len(droplets)).Msg("ssh ready")
log.Info().Str("host", d.Name).Int32("ready", sshReady.Add(1)).Int("want", len(hosts)).Msg("ssh ready")
return nil
}); err != nil {
return err
}
agentBin, cleanup, err := bench.AgentBinary(opts.AgentBin)
agentBin, cleanup, err := resticbench.AgentBinary(opts.AgentBin)
if err != nil {
return err
}
defer cleanup()
resticBin, err := bench.EnsureResticLinux(ctx)
resticBin, err := resticbench.EnsureResticLinux(ctx)
if err != nil {
return err
}
log.Info().Int("droplets", len(droplets)).Str("restic", bench.ResticVersion).Msg("pushing bench agent + restic")
if err := eachDroplet(droplets, func(d provider.Host) error {
return s.pushBinaries(ctx, d.PublicIP, agentBin, resticBin)
log.Info().Int("hosts", len(hosts)).Str("restic", resticbench.ResticVersion).Msg("pushing bench agent + restic")
if err := eachHost(hosts, func(d provider.Host) error {
return s.ssh.Push(ctx, d.PublicIP, resticbench.RemoteBinDir,
map[string]string{agentBin: "bench-agent", resticBin: "restic"})
}); err != nil {
return err
}
@@ -192,7 +193,7 @@ func (s *Session) Deploy(ctx context.Context, opts DeployOptions) error {
if err := s.SaveState(); err != nil {
return err
}
log.Info().Int("droplets", len(droplets)).Msg("fleet deployed and ready")
log.Info().Int("hosts", len(hosts)).Msg("fleet deployed and ready")
return nil
}
@@ -218,7 +219,7 @@ func (s *Session) ensureKey(ctx context.Context) (string, error) {
if err != nil {
return "", err
}
block, err := ssh.MarshalPrivateKey(priv, "yuctl bench-wide "+s.slug())
block, err := ssh.MarshalPrivateKey(priv, "yuctl fleet-bench "+s.slug())
if err != nil {
return "", fmt.Errorf("marshal ssh private key: %w", err)
}
@@ -243,84 +244,58 @@ func (s *Session) ensureKey(ctx context.Context) (string, error) {
return keyID, s.SaveState()
}
// waitSSH retries a no-op ssh until the droplet accepts the fleet key. Fresh
// droplets take a little while between "active" and sshd+key readiness.
// waitSSH retries a no-op ssh until the host accepts the fleet key. Fresh
// hosts take a little while between "active" and sshd+key readiness.
// Single-shot probes (the loop is the retry) with the latest failure surfaced
// periodically, so a stuck droplet is visible instead of a silent hang.
// periodically, so a stuck host is visible instead of a silent hang.
func (s *Session) waitSSH(ctx context.Context, d provider.Host, timeout time.Duration) error {
start := time.Now()
deadline := start.Add(timeout)
lastLog := start
for {
_, err := s.sshRunOnce(ctx, d.PublicIP, "true", nil)
_, err := s.ssh.RunOnce(ctx, d.PublicIP, "true", nil)
if err == nil {
return nil
}
if time.Now().After(deadline) {
return fmt.Errorf("droplet %s (%s) not reachable over ssh after %s: %w", d.Name, d.PublicIP, timeout, err)
return fmt.Errorf("host %s (%s) not reachable over ssh after %s: %w", d.Name, d.PublicIP, timeout, err)
}
if time.Since(lastLog) > 30*time.Second {
lastLog = time.Now()
log.Warn().Str("droplet", d.Name).Str("ip", d.PublicIP).
log.Warn().Str("host", d.Name).Str("ip", d.PublicIP).
Str("waited", time.Since(start).Round(time.Second).String()).
Str("error", tailStr(err.Error(), 160)).Msg("still waiting for ssh")
Str("error", sshx.Tail(err.Error(), 160)).Msg("still waiting for ssh")
}
select {
case <-ctx.Done():
return ctx.Err()
// A gentle cadence: on paths that graylist port-22 SYN bursts,
// hammering a not-yet-reachable droplet only extends the block.
// hammering a not-yet-reachable host only extends the block.
case <-time.After(10 * time.Second):
}
}
}
// pushBinaries scps the agent and restic into RemoteBinDir on the droplet.
func (s *Session) pushBinaries(ctx context.Context, ip, agentBin, resticBin string) error {
if _, err := s.sshRun(ctx, ip, "mkdir -p "+bench.RemoteBinDir, nil); err != nil {
return err
}
dest := s.Provider.SSHUser() + "@" + ip + ":" + bench.RemoteBinDir + "/"
scp := func(local string) error {
cmd := exec.CommandContext(ctx, "scp", append([]string{"-q"}, s.sshArgs(local, dest)...)...)
if out, err := cmd.CombinedOutput(); err != nil {
return fmt.Errorf("scp %s to %s: %w: %s", local, ip, err, tailStr(string(out), 500))
}
return nil
}
// The local temp names are already bench-agent / restic_<ver>; normalize
// on the remote side so launch paths are stable.
if err := scp(agentBin); err != nil {
return err
}
if err := scp(resticBin); err != nil {
return err
}
script := fmt.Sprintf("cd %s && mv -f $(basename %q) bench-agent 2>/dev/null; mv -f $(basename %q) restic 2>/dev/null; chmod +x bench-agent restic",
bench.RemoteBinDir, agentBin, resticBin)
_, err := s.sshRun(ctx, ip, script, nil)
return err
}
// Undeploy destroys the whole fleet: droplets by tag, the DO ssh key, and the
// local key/known_hosts/state files. Repositories (and their data) persist —
// run `bench-do cleanup` first to prune them while the droplets still exist.
// Undeploy destroys the whole fleet: hosts by tag, the provider ssh key, and
// the local key/known_hosts/state files. Repositories (and their data)
// persist — run `fleet-bench cleanup` first to prune them while the hosts
// still exist.
func (s *Session) Undeploy(ctx context.Context, force bool) error {
droplets, err := s.Droplets(ctx)
hosts, err := s.Hosts(ctx)
if err != nil {
return err
}
if len(droplets) == 0 {
log.Info().Msg("no droplets tagged for this fleet")
if len(hosts) == 0 {
log.Info().Msg("no hosts tagged for this fleet")
} else {
if !force {
if running, err := s.anyLoadRunning(ctx, droplets); err != nil {
if running, err := s.anyLoadRunning(ctx, hosts); err != nil {
log.Warn().Err(err).Msg("could not check for running load")
} else if running {
return fmt.Errorf("load is still running; `yuctl tools bench-wide stop` first (or --force)")
return fmt.Errorf("load is still running; `yuctl tools fleet-bench stop` first (or --force)")
}
}
log.Info().Int("hosts", len(droplets)).Msg("destroying fleet hosts")
log.Info().Int("hosts", len(hosts)).Msg("destroying fleet hosts")
if err := s.Provider.DeleteByTag(ctx, s.Tag()); err != nil {
return err
}
@@ -343,10 +318,10 @@ func (s *Session) Undeploy(ctx context.Context, force bool) error {
// anyLoadRunning reports whether any host still has an agent or restic
// process alive.
func (s *Session) anyLoadRunning(ctx context.Context, droplets []provider.Host) (bool, error) {
func (s *Session) anyLoadRunning(ctx context.Context, hosts []provider.Host) (bool, error) {
running := false
err := eachDroplet(droplets, func(d provider.Host) error {
out, err := s.sshRun(ctx, d.PublicIP,
err := eachHost(hosts, func(d provider.Host) error {
out, err := s.ssh.Run(ctx, d.PublicIP,
`echo "$(pgrep -fc 'bench-agent --[l]oadgen') $(pgrep -xc restic)" 2>/dev/null || true`, nil)
if err != nil {
return err
@@ -1,4 +1,4 @@
package benchwide
package fleetbench
import (
"regexp"
@@ -6,10 +6,10 @@ import (
"testing"
)
func TestDropletName(t *testing.T) {
func TestHostName(t *testing.T) {
s := &Session{Partition: "staging", providerName: "do"}
if got := s.dropletName(7); got != "yucca-bench-do-staging-07" {
t.Fatalf("dropletName = %q", got)
if got := s.hostName(7); got != "yucca-bench-do-staging-07" {
t.Fatalf("hostName = %q", got)
}
if got := s.Tag(); got != "yuctl-bench-do-staging" {
t.Fatalf("Tag = %q", got)
@@ -44,7 +44,7 @@ Inter-| Receive | Transmit
---S3---
1 2
`
ds := DropletStatus{}
ds := HostStatus{}
parseSample(out, 5, &ds)
if ds.TxBps != float64(1900000-900000)*8/5 {
t.Fatalf("TxBps = %v", ds.TxBps)
@@ -65,7 +65,7 @@ Inter-| Receive | Transmit
func TestParseSampleMissingStatus(t *testing.T) {
out := "x---S1---y---S2---\n\n---S3---\n0 0\n"
ds := DropletStatus{}
ds := HostStatus{}
parseSample(out, 5, &ds)
if ds.Status != nil {
t.Fatalf("expected nil status, got %+v", ds.Status)
@@ -78,7 +78,7 @@ func TestParseSampleMissingStatus(t *testing.T) {
func TestStartOptionsDefaults(t *testing.T) {
o := StartOptions{}
o.defaults()
if o.ClientsPerDroplet != 1 || o.PackSizeMiB != 16 || o.CycleSize != 8<<30 || o.Seed == 0 {
if o.ClientsPerHost != 1 || o.PackSizeMiB != 16 || o.CycleSize != 8<<30 || o.Seed == 0 {
t.Fatalf("defaults = %+v", o)
}
if o.Label != "run" || o.Compression != "off" {
+54
View File
@@ -0,0 +1,54 @@
package fleetbench
import (
"encoding/json"
"fmt"
"io"
"os"
"strings"
"text/tabwriter"
"yuctl/resticbench"
)
// ShortName drops the shared fleet prefix so tables stay narrow.
func ShortName(name string) string {
return strings.TrimPrefix(name, "yucca-bench-")
}
func SaveResult(path string, r *Result) error {
b, err := json.MarshalIndent(r, "", " ")
if err != nil {
return err
}
return os.WriteFile(path, append(b, '\n'), 0o644)
}
func RenderResult(w io.Writer, r *Result) {
fmt.Fprintf(w, "\n%s partition=%s elapsed=%s\n", r.Label, r.Partition, resticbench.FormatDuration(r.ElapsedSeconds))
tw := tabwriter.NewWriter(w, 2, 4, 2, ' ', 0)
fmt.Fprintln(tw, "HOST\tREGION\tSTATE\tWIRE TX\tCLIENT\tCYCLES\tUPLOADED\tERRORS")
for _, d := range r.Hosts {
if len(d.Clients) == 0 {
fmt.Fprintf(tw, "%s\t%s\t%s\t%s\t-\t-\t-\t-\n",
ShortName(d.Name), d.Region, d.State, resticbench.FormatBytes(d.WireTxBytes))
continue
}
for i, c := range d.Clients {
name, region, state, wireTx := ShortName(d.Name), d.Region, d.State, resticbench.FormatBytes(d.WireTxBytes)
if i > 0 {
name, region, state, wireTx = "", "", "", ""
}
fmt.Fprintf(tw, "%s\t%s\t%s\t%s\t%s\t%d\t%s\t%d\n",
name, region, state, wireTx, "c"+strings.TrimPrefix(c.Name, d.Name+"-c"), c.Cycles,
resticbench.FormatBytes(c.Uploaded), c.Errors)
}
}
tw.Flush()
fmt.Fprintf(w, "totals: uploaded=%s wire-tx=%s cycles=%d errors=%d",
resticbench.FormatBytes(r.TotalUploaded), resticbench.FormatBytes(r.TotalWireTx), r.TotalCycles, r.TotalErrors)
if r.ElapsedSeconds > 0 && r.TotalUploaded > 0 {
fmt.Fprintf(w, " avg=%s", resticbench.FormatBPS(float64(r.TotalUploaded)/r.ElapsedSeconds))
}
fmt.Fprintln(w)
}
@@ -1,14 +1,12 @@
package benchwide
package fleetbench
import (
"bufio"
"context"
"crypto/rand"
"encoding/binary"
"encoding/json"
"fmt"
"os"
"os/exec"
"sort"
"strconv"
"strings"
@@ -17,10 +15,11 @@ import (
"github.com/rs/zerolog/log"
"yuctl/internal/adminapi"
"yuctl/internal/bench"
"yuctl/internal/netdev"
"yuctl/internal/provider"
"yuctl/adminapi"
"yuctl/netdev"
"yuctl/provider"
"yuctl/resticbench"
"yuctl/sshx"
)
// RepoMinter is the slice of the admin-api client Start needs: repository
@@ -31,24 +30,24 @@ type RepoMinter interface {
RepositoryURL(ctx context.Context, id string) (string, error)
}
// StartOptions shape the load each droplet runs.
// StartOptions shape the load each host runs.
type StartOptions struct {
ClientsPerDroplet int // default 1
CycleSize int64 // dataset per client per cycle (default 8GiB)
FileSize int64 // default 64MiB
PackSizeMiB int // restic pack ("object") size, 4..128 (default 16)
Connections int // rest.connections per client (default 5)
ReadConcurrency int // default 4
Compression string // default off
Duration time.Duration // 0 = until `stop`
MaxTransfer int64 // per-droplet wire cap; 0 = the size's allowance
Label string // results label
Seed uint64 // 0 = random
ClientsPerHost int // default 1
CycleSize int64 // dataset per client per cycle (default 8GiB)
FileSize int64 // default 64MiB
PackSizeMiB int // restic pack ("object") size, 4..128 (default 16)
Connections int // rest.connections per client (default 5)
ReadConcurrency int // default 4
Compression string // default off
Duration time.Duration // 0 = until `stop`
MaxTransfer int64 // per-host wire cap; 0 = the size's allowance
Label string // results label
Seed uint64 // 0 = random
}
func (o *StartOptions) defaults() {
if o.ClientsPerDroplet <= 0 {
o.ClientsPerDroplet = 1
if o.ClientsPerHost <= 0 {
o.ClientsPerHost = 1
}
if o.CycleSize <= 0 {
o.CycleSize = 8 << 30
@@ -83,7 +82,7 @@ func (o *StartOptions) defaults() {
// pattern from matching this very script's command line.
const killScript = `pkill -f 'bench-agent --[l]oadgen' 2>/dev/null; sleep 1; pkill -x restic 2>/dev/null; true`
// prepScript and launchScript are one droplet's start, as two SEPARATE ssh
// prepScript and launchScript are one host's start, as two SEPARATE ssh
// execs. They must not be merged: prep runs the kill patterns, and launch's
// command line spells out "bench-agent --loadgen" verbatim — in one script
// the pkill would match its own shell's command line and SIGTERM it mid-run
@@ -98,11 +97,11 @@ func prepScript() string {
func launchScript() string {
return fmt.Sprintf(
"setsid nohup $HOME/%s/bench-agent --loadgen %s </dev/null >> %s 2>&1 & echo launched",
bench.RemoteBinDir, configPath, agentLog)
resticbench.RemoteBinDir, configPath, agentLog)
}
// Start (re)launches the load on every droplet: ensures one repository per
// client via the admin-api, re-mints restic URLs, and hands each droplet a
// Start (re)launches the load on every host: ensures one repository per
// client via the admin-api, re-mints restic URLs, and hands each host a
// LoadgenConfig over ssh stdin (secrets never in argv). A second start is a
// graceful restart with new parameters.
func (s *Session) Start(ctx context.Context, minter RepoMinter, opts StartOptions) error {
@@ -110,15 +109,15 @@ func (s *Session) Start(ctx context.Context, minter RepoMinter, opts StartOption
if opts.PackSizeMiB < 4 || opts.PackSizeMiB > 128 {
return fmt.Errorf("--obj-size %dMiB out of restic's pack-size range (4..128 MiB)", opts.PackSizeMiB)
}
droplets, err := s.Droplets(ctx)
hosts, err := s.Hosts(ctx)
if err != nil {
return err
}
if len(droplets) == 0 {
return fmt.Errorf("no fleet droplets; run `yuctl tools bench-wide deploy` first")
if len(hosts) == 0 {
return fmt.Errorf("no fleet hosts; run `yuctl tools fleet-bench deploy` first")
}
clients, err := s.ensureClients(ctx, minter, droplets, opts.ClientsPerDroplet)
clients, err := s.ensureClients(ctx, minter, hosts, opts.ClientsPerHost)
if err != nil {
return err
}
@@ -136,9 +135,9 @@ func (s *Session) Start(ctx context.Context, minter RepoMinter, opts StartOption
capBytes = s.State.TransferBytes
}
if capBytes <= 0 {
// State was lost but droplets exist: re-derive the allowance rather
// State was lost but hosts exist: re-derive the allowance rather
// than ever running uncapped into paid overage.
size, err := s.Provider.ResolveSize(ctx, droplets[0].SizeSlug)
size, err := s.Provider.ResolveSize(ctx, hosts[0].SizeSlug)
if err != nil {
return fmt.Errorf("no recorded transfer allowance and size lookup failed: %w", err)
}
@@ -146,23 +145,23 @@ func (s *Session) Start(ctx context.Context, minter RepoMinter, opts StartOption
s.State.TransferBytes = capBytes
s.State.SizeSlug = size.Slug
}
log.Info().Int("droplets", len(droplets)).Int("clients", len(clients)).
log.Info().Int("hosts", len(hosts)).Int("clients", len(clients)).
Str("obj_size", fmt.Sprintf("%dMiB", opts.PackSizeMiB)).
Str("cycle_size", bench.FormatBytes(opts.CycleSize)).
Str("cycle_size", resticbench.FormatBytes(opts.CycleSize)).
Str("duration", durationLabel(opts.Duration)).
Str("cap_per_droplet", bench.FormatBytes(capBytes)).
Str("cap_pool", bench.FormatBytes(capBytes*int64(len(droplets)))).
Str("cap_per_host", resticbench.FormatBytes(capBytes)).
Str("cap_pool", resticbench.FormatBytes(capBytes*int64(len(hosts)))).
Msg("launching load (kill previous, then detach the agent)")
err = eachDroplet(droplets, func(d provider.Host) error {
var mine []bench.LoadgenClient
err = eachHost(hosts, func(d provider.Host) error {
var mine []resticbench.LoadgenClient
for _, cl := range clients {
if cl.Droplet == d.Name {
mine = append(mine, bench.LoadgenClient{Name: cl.Name, Repo: urls[cl.Name], Password: cl.Password})
if cl.Host == d.Name {
mine = append(mine, resticbench.LoadgenClient{Name: cl.Name, Repo: urls[cl.Name], Password: cl.Password})
}
}
cfg := bench.LoadgenConfig{
Op: bench.LoadgenOpLoad,
cfg := resticbench.LoadgenConfig{
Op: resticbench.LoadgenOpLoad,
Clients: mine,
Workdir: Workdir,
CycleSize: opts.CycleSize,
@@ -180,17 +179,17 @@ func (s *Session) Start(ctx context.Context, minter RepoMinter, opts StartOption
if err != nil {
return err
}
if _, err := s.sshRun(ctx, d.PublicIP, prepScript(), raw); err != nil {
if _, err := s.ssh.Run(ctx, d.PublicIP, prepScript(), raw); err != nil {
return fmt.Errorf("prepare %s: %w", d.Name, err)
}
out, err := s.sshRun(ctx, d.PublicIP, launchScript(), nil)
out, err := s.ssh.Run(ctx, d.PublicIP, launchScript(), nil)
if err != nil {
return fmt.Errorf("launch on %s: %w", d.Name, err)
}
if !strings.Contains(out, "launched") {
return fmt.Errorf("launch on %s: unexpected output %q", d.Name, tailStr(out, 200))
return fmt.Errorf("launch on %s: unexpected output %q", d.Name, sshx.Tail(out, 200))
}
log.Info().Str("droplet", d.Name).Str("region", d.Region).Int("clients", len(mine)).Msg("load launched")
log.Info().Str("host", d.Name).Str("region", d.Region).Int("clients", len(mine)).Msg("load launched")
return nil
})
if err != nil {
@@ -201,20 +200,20 @@ func (s *Session) Start(ctx context.Context, minter RepoMinter, opts StartOption
StartedAt: time.Now().UTC(),
Label: opts.Label,
Params: map[string]string{
"droplets": strconv.Itoa(len(droplets)),
"clients_per_droplet": strconv.Itoa(opts.ClientsPerDroplet),
"obj_size_mib": strconv.Itoa(opts.PackSizeMiB),
"cycle_size": bench.FormatBytes(opts.CycleSize),
"connections": strconv.Itoa(opts.Connections),
"duration": durationLabel(opts.Duration),
"cap_per_droplet": bench.FormatBytes(capBytes),
"seed": strconv.FormatUint(opts.Seed, 10),
"hosts": strconv.Itoa(len(hosts)),
"clients_per_host": strconv.Itoa(opts.ClientsPerHost),
"obj_size_mib": strconv.Itoa(opts.PackSizeMiB),
"cycle_size": resticbench.FormatBytes(opts.CycleSize),
"connections": strconv.Itoa(opts.Connections),
"duration": durationLabel(opts.Duration),
"cap_per_host": resticbench.FormatBytes(capBytes),
"seed": strconv.FormatUint(opts.Seed, 10),
},
}
if err := s.SaveState(); err != nil {
return err
}
log.Info().Msg("all droplets launched; `yuctl tools bench-wide watch` for the live dashboard")
log.Info().Msg("all hosts launched; `yuctl tools fleet-bench watch` for the live dashboard")
return nil
}
@@ -226,17 +225,17 @@ func durationLabel(d time.Duration) string {
}
// ensureClients reconciles the persisted client list against the live fleet:
// one repository per (droplet, slot), created on first use and reused across
// one repository per (host, slot), created on first use and reused across
// restarts (fresh seeds keep reuse dedup-proof).
func (s *Session) ensureClients(ctx context.Context, minter RepoMinter, droplets []provider.Host, perDroplet int) ([]Client, error) {
func (s *Session) ensureClients(ctx context.Context, minter RepoMinter, hosts []provider.Host, perHost int) ([]Client, error) {
existing := map[string]Client{}
for _, cl := range s.State.Clients {
existing[cl.Name] = cl
}
var out []Client
changed := false
for _, d := range droplets {
for j := 1; j <= perDroplet; j++ {
for _, d := range hosts {
for j := 1; j <= perHost; j++ {
name := fmt.Sprintf("%s-c%d", d.Name, j)
if cl, ok := existing[name]; ok {
out = append(out, cl)
@@ -247,7 +246,7 @@ func (s *Session) ensureClients(ctx context.Context, minter RepoMinter, droplets
if err != nil {
return nil, fmt.Errorf("create repository %s: %w", repoName, err)
}
cl := Client{Name: name, Droplet: d.Name, RepoID: repo.ID, RepoName: repoName, Password: randHex(16)}
cl := Client{Name: name, Host: d.Name, RepoID: repo.ID, RepoName: repoName, Password: randHex(16)}
log.Info().Str("repo", repoName).Str("client", name).Msg("created bench repository (persists after the run)")
s.State.Clients = append(s.State.Clients, cl)
out = append(out, cl)
@@ -262,25 +261,25 @@ func (s *Session) ensureClients(ctx context.Context, minter RepoMinter, droplets
return out, nil
}
// Stop kills the load everywhere (droplets stay deployed) and collects each
// droplet's final status into a Result.
// Stop kills the load everywhere (hosts stay deployed) and collects each
// host's final status into a Result.
func (s *Session) Stop(ctx context.Context) (*Result, error) {
droplets, err := s.Droplets(ctx)
hosts, err := s.Hosts(ctx)
if err != nil {
return nil, err
}
if len(droplets) == 0 {
log.Info().Msg("no droplets; nothing to stop")
if len(hosts) == 0 {
log.Info().Msg("no hosts; nothing to stop")
return nil, nil
}
log.Info().Int("droplets", len(droplets)).Msg("stopping load on all droplets")
if err := eachDroplet(droplets, func(d provider.Host) error {
_, err := s.sshRun(ctx, d.PublicIP, killScript, nil)
log.Info().Int("hosts", len(hosts)).Msg("stopping load on all hosts")
if err := eachHost(hosts, func(d provider.Host) error {
_, err := s.ssh.Run(ctx, d.PublicIP, killScript, nil)
return err
}); err != nil {
return nil, err
}
res, err := s.collect(ctx, droplets)
res, err := s.collect(ctx, hosts)
if err != nil {
return nil, err
}
@@ -292,7 +291,7 @@ func (s *Session) Stop(ctx context.Context) (*Result, error) {
}
// Result is the fleet-aggregated outcome of a run, written to a local JSON on
// stop. Client numbers are restic's post-dedup data_added; droplet wire TX is
// stop. Client numbers are restic's post-dedup data_added; host wire TX is
// what the transfer allowance actually saw.
type Result struct {
Label string `json:"label"`
@@ -301,23 +300,23 @@ type Result struct {
StartedAt time.Time `json:"startedAt,omitempty"`
ElapsedSeconds float64 `json:"elapsedSeconds,omitempty"`
Params map[string]string `json:"params,omitempty"`
Droplets []DropletResult `json:"droplets"`
Hosts []HostResult `json:"droplets"`
TotalUploaded int64 `json:"totalUploaded"`
TotalWireTx int64 `json:"totalWireTx"`
TotalCycles int `json:"totalCycles"`
TotalErrors int `json:"totalErrors"`
}
// DropletResult is one droplet's final tallies.
type DropletResult struct {
Name string `json:"name"`
Region string `json:"region"`
State string `json:"state,omitempty"`
WireTxBytes int64 `json:"wireTxBytes"`
Clients []bench.LoadgenClientStatus `json:"clients,omitempty"`
// HostResult is one host's final tallies.
type HostResult struct {
Name string `json:"name"`
Region string `json:"region"`
State string `json:"state,omitempty"`
WireTxBytes int64 `json:"wireTxBytes"`
Clients []resticbench.LoadgenClientStatus `json:"clients,omitempty"`
}
func (s *Session) collect(ctx context.Context, droplets []provider.Host) (*Result, error) {
func (s *Session) collect(ctx context.Context, hosts []provider.Host) (*Result, error) {
res := &Result{Partition: s.Partition, Created: time.Now().UTC(), Label: "run"}
if s.State.Run != nil {
res.Label = s.State.Run.Label
@@ -326,27 +325,27 @@ func (s *Session) collect(ctx context.Context, droplets []provider.Host) (*Resul
res.Params = s.State.Run.Params
}
var mu sync.Mutex
err := eachDroplet(droplets, func(d provider.Host) error {
out, err := s.sshRun(ctx, d.PublicIP, "cat "+statusPath+" 2>/dev/null || true", nil)
err := eachHost(hosts, func(d provider.Host) error {
out, err := s.ssh.Run(ctx, d.PublicIP, "cat "+statusPath+" 2>/dev/null || true", nil)
if err != nil {
return err
}
dr := DropletResult{Name: d.Name, Region: d.Region}
var st bench.LoadgenStatus
dr := HostResult{Name: d.Name, Region: d.Region}
var st resticbench.LoadgenStatus
if json.Unmarshal([]byte(out), &st) == nil {
dr.State = st.State
dr.WireTxBytes = st.WireTxBytes
dr.Clients = st.Clients
}
mu.Lock()
res.Droplets = append(res.Droplets, dr)
res.Hosts = append(res.Hosts, dr)
mu.Unlock()
return nil
})
if err != nil {
return nil, err
}
for _, dr := range res.Droplets {
for _, dr := range res.Hosts {
res.TotalWireTx += dr.WireTxBytes
for _, cl := range dr.Clients {
res.TotalUploaded += cl.Uploaded
@@ -357,30 +356,30 @@ func (s *Session) collect(ctx context.Context, droplets []provider.Host) (*Resul
return res, nil
}
// DropletStatus is one droplet's live sample.
type DropletStatus struct {
// HostStatus is one host's live sample.
type HostStatus struct {
Name, Region, IP string
Reachable bool
Err string
AgentProcs, ResticProcs int
TxBps, RxBps float64
Status *bench.LoadgenStatus
Status *resticbench.LoadgenStatus
}
// StatusReport is what status/watch render.
type StatusReport struct {
Run *RunInfo
TransferCap int64 // per droplet
Droplets []DropletStatus
TransferCap int64 // per host
Hosts []HostStatus
}
// Status samples every droplet in parallel with one ssh round-trip each:
// Status samples every host in parallel with one ssh round-trip each:
// NIC counters over the window, the agent's status file, and process counts.
func (s *Session) Status(ctx context.Context, sampleSeconds int) (*StatusReport, error) {
if sampleSeconds <= 0 {
sampleSeconds = 5
}
droplets, err := s.Droplets(ctx)
hosts, err := s.Hosts(ctx)
if err != nil {
return nil, err
}
@@ -392,30 +391,26 @@ func (s *Session) Status(ctx context.Context, sampleSeconds int) (*StatusReport,
sampleSeconds, statusPath)
var mu sync.Mutex
_ = eachDroplet(droplets, func(d provider.Host) error {
ds := DropletStatus{Name: d.Name, Region: d.Region, IP: d.PublicIP}
out, err := s.sshRun(ctx, d.PublicIP, script, nil)
_ = eachHost(hosts, func(d provider.Host) error {
ds := HostStatus{Name: d.Name, Region: d.Region, IP: d.PublicIP}
out, err := s.ssh.Run(ctx, d.PublicIP, script, nil)
if err != nil {
ds.Err = tailStr(err.Error(), 120)
ds.Err = sshx.Tail(err.Error(), 120)
} else {
ds.Reachable = true
parseSample(out, sampleSeconds, &ds)
}
mu.Lock()
report.Droplets = append(report.Droplets, ds)
report.Hosts = append(report.Hosts, ds)
mu.Unlock()
return nil
})
sortDroplets(report.Droplets)
sort.Slice(report.Hosts, func(i, j int) bool { return report.Hosts[i].Name < report.Hosts[j].Name })
return report, nil
}
func sortDroplets(ds []DropletStatus) {
sort.Slice(ds, func(i, j int) bool { return ds[i].Name < ds[j].Name })
}
// parseSample decodes the three-marker sample script output.
func parseSample(out string, sampleSeconds int, ds *DropletStatus) {
func parseSample(out string, sampleSeconds int, ds *HostStatus) {
part1, rest, ok := strings.Cut(out, "---S1---")
if !ok {
return
@@ -431,7 +426,7 @@ func parseSample(out string, sampleSeconds int, ds *DropletStatus) {
ds.TxBps = float64(after.TX-before.TX) * 8 / float64(sampleSeconds)
ds.RxBps = float64(after.RX-before.RX) * 8 / float64(sampleSeconds)
var st bench.LoadgenStatus
var st resticbench.LoadgenStatus
if json.Unmarshal([]byte(strings.TrimSpace(part3)), &st) == nil && !st.StartedAt.IsZero() {
ds.Status = &st
}
@@ -441,44 +436,44 @@ func parseSample(out string, sampleSeconds int, ds *DropletStatus) {
}
}
// Cleanup forgets and prunes every bench-do snapshot, running the agent's
// cleanup op synchronously on each droplet (fresh URLs are minted; the
// droplets must still exist). Refuses while load is running unless force.
// Cleanup forgets and prunes every fleet-bench snapshot, running the agent's
// cleanup op synchronously on each host (fresh URLs are minted; the hosts
// must still exist). Refuses while load is running unless force.
func (s *Session) Cleanup(ctx context.Context, minter RepoMinter, force bool) error {
droplets, err := s.Droplets(ctx)
hosts, err := s.Hosts(ctx)
if err != nil {
return err
}
if len(droplets) == 0 {
return fmt.Errorf("no fleet droplets to clean from; repos can only be pruned while the fleet exists")
if len(hosts) == 0 {
return fmt.Errorf("no fleet hosts to clean from; repos can only be pruned while the fleet exists")
}
if len(s.State.Clients) == 0 {
log.Info().Msg("no bench repositories recorded; nothing to clean")
return nil
}
if !force {
if running, err := s.anyLoadRunning(ctx, droplets); err == nil && running {
return fmt.Errorf("load is still running; `yuctl tools bench-wide stop` first (or --force)")
if running, err := s.anyLoadRunning(ctx, hosts); err == nil && running {
return fmt.Errorf("load is still running; `yuctl tools fleet-bench stop` first (or --force)")
}
}
byDroplet := map[string][]Client{}
byHost := map[string][]Client{}
for _, cl := range s.State.Clients {
byDroplet[cl.Droplet] = append(byDroplet[cl.Droplet], cl)
byHost[cl.Host] = append(byHost[cl.Host], cl)
}
return eachDroplet(droplets, func(d provider.Host) error {
mine := byDroplet[d.Name]
return eachHost(hosts, func(d provider.Host) error {
mine := byHost[d.Name]
if len(mine) == 0 {
return nil
}
var lcs []bench.LoadgenClient
var lcs []resticbench.LoadgenClient
for _, cl := range mine {
u, err := minter.RepositoryURL(ctx, cl.RepoID)
if err != nil {
return fmt.Errorf("mint URL for %s: %w", cl.RepoName, err)
}
lcs = append(lcs, bench.LoadgenClient{Name: cl.Name, Repo: u, Password: cl.Password})
lcs = append(lcs, resticbench.LoadgenClient{Name: cl.Name, Repo: u, Password: cl.Password})
}
cfg := bench.LoadgenConfig{Op: bench.LoadgenOpCleanup, Clients: lcs, Workdir: Workdir}
cfg := resticbench.LoadgenConfig{Op: resticbench.LoadgenOpCleanup, Clients: lcs, Workdir: Workdir}
raw, err := json.Marshal(cfg)
if err != nil {
return err
@@ -490,8 +485,7 @@ func (s *Session) Cleanup(ctx context.Context, minter RepoMinter, force bool) er
// streamCleanup runs the agent cleanup op over a live ssh session, logging
// its event stream.
func (s *Session) streamCleanup(ctx context.Context, d provider.Host, cfg []byte) error {
cmd := exec.CommandContext(ctx, "ssh",
s.sshArgs(s.Provider.SSHUser()+"@"+d.PublicIP, "$HOME/"+bench.RemoteBinDir+"/bench-agent --loadgen")...)
cmd := s.ssh.Command(ctx, d.PublicIP, "$HOME/"+resticbench.RemoteBinDir+"/bench-agent --loadgen")
cmd.Stdin = strings.NewReader(string(cfg))
cmd.Stderr = os.Stderr
stdout, err := cmd.StdoutPipe()
@@ -502,27 +496,21 @@ func (s *Session) streamCleanup(ctx context.Context, d provider.Host, cfg []byte
return err
}
var fatal string
sc := bufio.NewScanner(stdout)
sc.Buffer(make([]byte, 1<<20), 8<<20)
for sc.Scan() {
var ev bench.Event
if json.Unmarshal(sc.Bytes(), &ev) != nil {
continue
}
_ = resticbench.ScanEvents(stdout, func(ev resticbench.Event) {
switch ev.Type {
case "phase_start":
log.Info().Str("droplet", d.Name).Msgf("%s: started", ev.Phase)
log.Info().Str("host", d.Name).Msgf("%s: started", ev.Phase)
case "phase_done":
e := log.Info().Str("droplet", d.Name)
e := log.Info().Str("host", d.Name)
if ev.PhaseResult != nil {
e = e.Str("duration", bench.FormatDuration(ev.PhaseResult.Seconds)).
e = e.Str("duration", resticbench.FormatDuration(ev.PhaseResult.Seconds)).
Int64("snapshots", ev.PhaseResult.Bytes)
}
e.Msgf("%s: done", ev.Phase)
case "fatal":
fatal = ev.Message
}
}
})
if err := cmd.Wait(); err != nil {
if fatal != "" {
return fmt.Errorf("cleanup on %s: %s", d.Name, fatal)
@@ -1,34 +1,35 @@
// Package benchwide deploys and drives a fleet of cloud VMs — across
// providers (DigitalOcean, Hetzner, …; see internal/provider) — running real
// Package fleetbench deploys and drives a fleet of cloud VMs — across
// providers (DigitalOcean, Hetzner, …; see yuctl/provider) — running real
// restic clients against michael, the external-user data path over the public
// internet. It is the VM-shaped sibling of internal/warp (fleet lifecycle:
// internet. It is the VM-shaped sibling of fleet/warp (fleet lifecycle:
// deploy/start/status/stop/cleanup/undeploy) built on the bench agent's
// loadgen mode: every host runs a detached supervisor looping seeded
// generate→backup cycles per client, hard-capped at the host's transfer
// allowance so a forgotten run cannot burn into paid overage. Each provider is
// a separate, independently-addressable fleet (state keyed by partition ×
// provider), so multiple providers can load michael concurrently.
package benchwide
package fleetbench
import (
"bytes"
"context"
"crypto/rand"
"encoding/hex"
"encoding/json"
"errors"
"fmt"
"os"
"os/exec"
"path/filepath"
"strings"
"sync"
"time"
yctx "yuctl/internal/context"
"yuctl/internal/provider"
"yuctl/ctxstore"
"yuctl/fleet"
"yuctl/provider"
"yuctl/sshx"
)
// Legacy identifiers from the tool's bench-do/bench-wide eras survive below
// (Workdir, the tag prefix, repo prefixes, the state dir and JSON tags):
// changing any of them would orphan deployed hosts, persisted repo passwords,
// or remote state on live fleets.
const (
// Workdir is the host-side scratch root: datasets, restic caches, the
// loadgen config (transient) and status file, and the agent log.
@@ -45,11 +46,11 @@ const (
)
// Client is one restic identity: its admin-api repository and password,
// pinned to a droplet by name. Passwords are generated once and persisted —
// pinned to a host by name. Passwords are generated once and persisted —
// without them the repos can never be reopened (cleanup, restarts).
type Client struct {
Name string `json:"name"` // <droplet>-c<n>
Droplet string `json:"droplet"` // droplet name it runs on
Name string `json:"name"` // <host>-c<n>
Host string `json:"droplet"`
RepoID string `json:"repoId"`
RepoName string `json:"repoName"`
Password string `json:"password"`
@@ -80,7 +81,7 @@ type State struct {
Run *RunInfo `json:"run,omitempty"`
}
// Session is the resolved handle for one bench-wide command invocation,
// Session is the resolved handle for one fleet-bench command invocation,
// scoped to one provider × partition fleet.
type Session struct {
Partition string
@@ -89,6 +90,7 @@ type Session struct {
providerName string // slug used in the fleet tag + local filenames
dir string // ${XDG_CONFIG_HOME}/yuctl/bench-wide
ssh *sshx.Client
}
// NewSession builds the named provider's client (env/op token) and loads any
@@ -98,7 +100,7 @@ func NewSession(ctx context.Context, partition, providerName string) (*Session,
if err != nil {
return nil, err
}
base, err := yctx.Dir()
base, err := ctxstore.Dir()
if err != nil {
return nil, err
}
@@ -106,6 +108,16 @@ func NewSession(ctx context.Context, partition, providerName string) (*Session,
if err := os.MkdirAll(filepath.Join(s.dir, "cm"), 0o700); err != nil {
return nil, fmt.Errorf("create ssh control dir: %w", err)
}
// Per-fleet known_hosts: provider IPs get recycled across deployments —
// the operator's global file would scream host-key-changed.
s.ssh = &sshx.Client{
User: prov.SSHUser(),
IdentityFile: s.PrivateKeyPath(),
KnownHostsFile: s.knownHostsPath(),
ControlDir: filepath.Join(s.dir, "cm"),
ConnectTimeoutSeconds: 10,
Retries: 2,
}
if err := s.loadState(); err != nil {
return nil, err
}
@@ -165,10 +177,10 @@ func (s *Session) SaveState() error {
return os.Rename(tmp, p)
}
// clearFleetState drops everything droplet-bound (ssh key material, run
// info) but keeps the client records when repos exist: their passwords are
// the only way to ever reopen those repos (e.g. `tools bench --repo-id` with
// a re-minted URL to prune them later).
// clearFleetState drops everything host-bound (ssh key material, run info)
// but keeps the client records when repos exist: their passwords are the only
// way to ever reopen those repos (e.g. `tools bench --repo-id` with a
// re-minted URL to prune them later).
func (s *Session) clearFleetState() error {
os.Remove(s.PrivateKeyPath())
os.Remove(s.knownHostsPath())
@@ -183,111 +195,20 @@ func (s *Session) clearFleetState() error {
return s.SaveState()
}
// sshArgs are the fleet ssh options: the ephemeral identity, a per-fleet
// known_hosts (droplet IPs get recycled across deployments — the global file
// would scream host-key-changed), TOFU on first contact, and connection
// multiplexing. Multiplexing matters beyond latency: status/watch/start
// otherwise open a fresh port-22 TCP flow per droplet per command, a burst
// pattern that trips ssh-targeted rate limiting on some paths (observed as
// flapping per-IP port-22 SYN drops while ICMP and other ports stay fine).
// One persistent master per droplet keeps the flow count flat.
func (s *Session) sshArgs(extra ...string) []string {
args := []string{
"-o", "BatchMode=yes",
"-o", "StrictHostKeyChecking=accept-new",
"-o", "UserKnownHostsFile=" + s.knownHostsPath(),
"-o", "ConnectTimeout=10",
"-o", "ServerAliveInterval=30",
"-o", "ServerAliveCountMax=8",
"-o", "ControlMaster=auto",
"-o", "ControlPath=" + filepath.Join(s.dir, "cm", "%C"),
"-o", "ControlPersist=300",
"-i", s.PrivateKeyPath(),
"-o", "IdentitiesOnly=yes",
}
return append(args, extra...)
}
// sshRun executes script on the droplet, returning combined stdout. stdin may
// be nil. The script travels as the ssh command argument (visible in remote
// ps — it must never contain secrets); secret payloads go through stdin.
// Connection-level failures (ssh exit 255: banner timeouts, resets — common
// when tens of sessions open against fresh droplets) are retried; remote
// command failures are not, ssh reports those as the command's own exit code.
func (s *Session) sshRun(ctx context.Context, ip, script string, stdin []byte) (string, error) {
var lastOut string
var lastErr error
for attempt := 1; attempt <= 3; attempt++ {
out, err := s.sshRunOnce(ctx, ip, script, stdin)
if err == nil {
return out, nil
}
lastOut, lastErr = out, err
var exit *exec.ExitError
if ctx.Err() != nil || !errors.As(err, &exit) || exit.ExitCode() != 255 {
break
}
select {
case <-ctx.Done():
return lastOut, lastErr
case <-time.After(time.Duration(attempt) * 5 * time.Second):
}
}
return lastOut, lastErr
}
// sshRunOnce is a single ssh attempt with no retry — used by callers that
// implement their own retry cadence (waitSSH), where nesting retries would
// multiply the delays.
func (s *Session) sshRunOnce(ctx context.Context, ip, script string, stdin []byte) (string, error) {
cmd := exec.CommandContext(ctx, "ssh", s.sshArgs(s.Provider.SSHUser()+"@"+ip, script)...)
if stdin != nil {
cmd.Stdin = bytes.NewReader(stdin)
}
var out, errb bytes.Buffer
cmd.Stdout = &out
cmd.Stderr = &errb
if err := cmd.Run(); err != nil {
return out.String(), fmt.Errorf("ssh %s: %w: %s", ip, err, tailStr(errb.String(), 1000))
}
return out.String(), nil
}
func tailStr(v string, n int) string {
v = strings.TrimSpace(v)
if len(v) > n {
return "…" + v[len(v)-n:]
}
return v
}
// Droplets returns the live fleet from the provider's tag listing. (Named for
// the historical DO fleet; a host is a host on any provider.)
func (s *Session) Droplets(ctx context.Context) ([]provider.Host, error) {
// Hosts returns the live fleet from the provider's tag listing.
func (s *Session) Hosts(ctx context.Context) ([]provider.Host, error) {
return s.Provider.List(ctx, s.Tag())
}
// eachDroplet runs fn against every host in parallel — staggered by 150ms each
// so a big fleet doesn't fire all its port-22 SYNs in one burst (see sshArgs)
// — and returns the first error (all hosts are still attempted).
func eachDroplet(hosts []provider.Host, fn func(d provider.Host) error) error {
var wg sync.WaitGroup
errs := make([]error, len(hosts))
for i, d := range hosts {
wg.Add(1)
go func(i int, d provider.Host) {
defer wg.Done()
time.Sleep(time.Duration(i) * 150 * time.Millisecond)
errs[i] = fn(d)
}(i, d)
}
wg.Wait()
for _, err := range errs {
if err != nil {
return err
}
}
return nil
// eachHost runs fn against every host in parallel — staggered by 150ms each
// so a big fleet doesn't fire all its port-22 SYNs in one burst (see
// sshx.Client.ControlDir) — and returns the first error (all hosts are still
// attempted).
func eachHost(hosts []provider.Host, fn func(d provider.Host) error) error {
return fleet.Each(len(hosts), func(i int) error {
time.Sleep(time.Duration(i) * 150 * time.Millisecond)
return fn(hosts[i])
})
}
func randHex(n int) string {
@@ -25,6 +25,7 @@ type CleanupOptions struct {
Force bool // purge even while load is running
Keep bool // keep the finished Job around for inspection
Timeout time.Duration
JobLogs io.Writer // where the finished Job's logs go (nil = discard)
}
// Cleanup purges warp test buckets via a hostNetwork Job running `mc` inside
@@ -130,8 +131,8 @@ echo cleanup done`, mci, target.Endpoint, strings.Join(patterns, "|"), strings.J
return false, nil
})
if logs := s.jobLogs(ctx, name); logs != "" {
fmt.Println(logs)
if logs := s.jobLogs(ctx, name); logs != "" && opts.JobLogs != nil {
fmt.Fprintln(opts.JobLogs, logs)
}
if waitErr != nil {
return fmt.Errorf("cleanup job did not complete (kept for inspection): %w", waitErr)
@@ -12,7 +12,8 @@ import (
corev1 "k8s.io/api/core/v1"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
"yuctl/internal/netdev"
"yuctl/fleet"
"yuctl/netdev"
)
// StartOptions shape the load. Zero values reproduce the proven per-pod shape
@@ -117,52 +118,42 @@ func (s *Session) Start(ctx context.Context, opts StartOptions) error {
Str("obj_size", opts.PutObjSize).Int("rgw_endpoints", len(ips)).Bool("nonstop", nonstop).
Msg("relaunching load on all pods in parallel (kill previous, then launch)")
var wg sync.WaitGroup
errs := make([]error, len(pods))
for i, pod := range pods {
if err := fleet.Each(len(pods), func(i int) error {
pod := pods[i]
put := share(totalPut, len(pods), i)
get := share(totalGet, len(pods), i)
wg.Add(1)
go func(i int, pod RunnerPod, put, get int) {
defer wg.Done()
putCmd := fmt.Sprintf(`/warp put --host=%s --host-select=roundrobin%s --bucket=%sput-%s `+
`--obj.size=%s --duration=%s --concurrent=%d --noclear --no-color >> /tmp/warp-put.log 2>&1`,
hostList, tlsFlags, opts.BucketPrefix, pod.Name, opts.PutObjSize, runDuration, put)
getCmd := fmt.Sprintf(`/warp get --host=%s --host-select=roundrobin%s --bucket=%sget-%s `+
`--objects=%d --obj.size=%s --duration=%s --concurrent=%d --noclear --no-color >> /tmp/warp-get.log 2>&1`,
hostList, tlsFlags, opts.BucketPrefix, pod.Name, opts.GetObjects, opts.GetObjSize, runDuration, get)
putCmd := fmt.Sprintf(`/warp put --host=%s --host-select=roundrobin%s --bucket=%sput-%s `+
`--obj.size=%s --duration=%s --concurrent=%d --noclear --no-color >> /tmp/warp-put.log 2>&1`,
hostList, tlsFlags, opts.BucketPrefix, pod.Name, opts.PutObjSize, runDuration, put)
getCmd := fmt.Sprintf(`/warp get --host=%s --host-select=roundrobin%s --bucket=%sget-%s `+
`--objects=%d --obj.size=%s --duration=%s --concurrent=%d --noclear --no-color >> /tmp/warp-get.log 2>&1`,
hostList, tlsFlags, opts.BucketPrefix, pod.Name, opts.GetObjects, opts.GetObjSize, runDuration, get)
// Kill and launch are separate execs: the launch script's literal
// "/warp put ..." text would otherwise match the kill patterns in
// the same command line and SIGTERM the script itself.
if _, err := s.podExec(ctx, pod.Name, killScript); err != nil {
if s.podGone(ctx, pod.Name) {
log.Warn().Str("pod", pod.Name).Msg("pod terminated mid-start; skipping it")
return
}
errs[i] = fmt.Errorf("stop previous load on %s: %w", pod.Name, err)
return
// Kill and launch are separate execs: the launch script's literal
// "/warp put ..." text would otherwise match the kill patterns in
// the same command line and SIGTERM the script itself.
if _, err := s.podExec(ctx, pod.Name, killScript); err != nil {
if s.podGone(ctx, pod.Name) {
log.Warn().Str("pod", pod.Name).Msg("pod terminated mid-start; skipping it")
return nil
}
var script strings.Builder
if put > 0 {
script.WriteString(launchLine("put", putCmd, nonstop))
}
if get > 0 {
script.WriteString(launchLine("get", getCmd, nonstop))
}
script.WriteString("true\n")
if _, err := s.podExec(ctx, pod.Name, script.String()); err != nil {
errs[i] = fmt.Errorf("start load on %s: %w", pod.Name, err)
return
}
log.Info().Str("pod", pod.Name).Str("node", pod.Node).Int("put", put).Int("get", get).Msg("load launched")
}(i, pod, put, get)
}
wg.Wait()
for _, err := range errs {
if err != nil {
return err
return fmt.Errorf("stop previous load on %s: %w", pod.Name, err)
}
var script strings.Builder
if put > 0 {
script.WriteString(launchLine("put", putCmd, nonstop))
}
if get > 0 {
script.WriteString(launchLine("get", getCmd, nonstop))
}
script.WriteString("true\n")
if _, err := s.podExec(ctx, pod.Name, script.String()); err != nil {
return fmt.Errorf("start load on %s: %w", pod.Name, err)
}
log.Info().Str("pod", pod.Name).Str("node", pod.Node).Int("put", put).Int("get", get).Msg("load launched")
return nil
}); err != nil {
return err
}
log.Info().Msg("all pods launched; `yuctl tools warp watch` for the live dashboard")
@@ -223,22 +214,13 @@ func (s *Session) Stop(ctx context.Context) error {
return nil
}
log.Info().Int("pods", len(pods)).Msg("killing warp processes on all pods in parallel")
var wg sync.WaitGroup
errs := make([]error, len(pods))
for i, pod := range pods {
wg.Add(1)
go func(i int, pod RunnerPod) {
defer wg.Done()
if _, err := s.podExec(ctx, pod.Name, killScript); err != nil {
errs[i] = fmt.Errorf("stop load on %s: %w", pod.Name, err)
}
}(i, pod)
}
wg.Wait()
for _, err := range errs {
if err != nil {
return err
if err := fleet.Each(len(pods), func(i int) error {
if _, err := s.podExec(ctx, pods[i].Name, killScript); err != nil {
return fmt.Errorf("stop load on %s: %w", pods[i].Name, err)
}
return nil
}); err != nil {
return err
}
log.Info().Msg("verifying nothing survived")
if running, err := s.anyLoadRunning(ctx); err == nil && running {
@@ -254,27 +236,19 @@ func (s *Session) anyLoadRunning(ctx context.Context) (bool, error) {
if err != nil {
return false, err
}
var wg sync.WaitGroup
counts := make([]int, len(pods))
errs := make([]error, len(pods))
for i, pod := range pods {
wg.Add(1)
go func(i int, pod RunnerPod) {
defer wg.Done()
out, err := s.podExec(ctx, pod.Name, `pgrep -f '/warp ([p]ut|[g]et)' 2>/dev/null | wc -l`)
if err != nil {
errs[i] = err
return
}
counts[i], _ = strconv.Atoi(strings.TrimSpace(out))
}(i, pod)
}
wg.Wait()
for i, err := range errs {
if err := fleet.Each(len(pods), func(i int) error {
out, err := s.podExec(ctx, pods[i].Name, `pgrep -f '/warp ([p]ut|[g]et)' 2>/dev/null | wc -l`)
if err != nil {
return false, err
return err
}
if counts[i] > 0 {
counts[i], _ = strconv.Atoi(strings.TrimSpace(out))
return nil
}); err != nil {
return false, err
}
for _, n := range counts {
if n > 0 {
return true, nil
}
}
@@ -320,14 +294,22 @@ func (s *Session) Status(ctx context.Context, sampleSeconds int) (*StatusReport,
report.Config = cm.Data
}
var mu sync.Mutex
var wg sync.WaitGroup
var nodePods []RunnerPod
seen := map[string]bool{}
for _, pod := range pods {
if !seen[pod.Node] {
seen[pod.Node] = true
nodePods = append(nodePods, pod)
}
}
// One fan-out over pod-status probes and per-node NIC samples together, so
// the fast probes overlap the sampleSeconds-long NIC windows.
var mu sync.Mutex
report.Pods = make([]PodStatus, len(pods))
for i, pod := range pods {
wg.Add(1)
go func(i int, pod RunnerPod) {
defer wg.Done()
_ = fleet.Each(len(pods)+len(nodePods), func(i int) error {
if i < len(pods) {
pod := pods[i]
ps := PodStatus{Name: pod.Name, Node: pod.Node}
out, err := s.podExec(ctx, pod.Name,
`echo "$(pgrep -f '/warp [p]ut' 2>/dev/null | wc -l) `+
@@ -348,32 +330,20 @@ func (s *Session) Status(ctx context.Context, sampleSeconds int) (*StatusReport,
ps.LastPut = strings.TrimSpace(lines[1])
}
}
mu.Lock()
report.Pods[i] = ps
mu.Unlock()
}(i, pod)
}
seen := map[string]bool{}
for _, pod := range pods {
if seen[pod.Node] {
continue
return nil
}
seen[pod.Node] = true
wg.Add(1)
go func(pod RunnerPod) {
defer wg.Done()
nt, err := s.sampleNode(ctx, pod, sampleSeconds)
if err != nil {
log.Warn().Err(err).Str("node", pod.Node).Msg("throughput sample failed")
return
}
mu.Lock()
report.Nodes = append(report.Nodes, *nt)
mu.Unlock()
}(pod)
}
wg.Wait()
pod := nodePods[i-len(pods)]
nt, err := s.sampleNode(ctx, pod, sampleSeconds)
if err != nil {
log.Warn().Err(err).Str("node", pod.Node).Msg("throughput sample failed")
return nil
}
mu.Lock()
report.Nodes = append(report.Nodes, *nt)
mu.Unlock()
return nil
})
return report, nil
}
@@ -31,8 +31,8 @@ import (
"k8s.io/client-go/tools/clientcmd"
"k8s.io/client-go/tools/remotecommand"
"yuctl/internal/op"
"yuctl/internal/state"
"yuctl/op"
"yuctl/state"
)
// fieldManager identifies this tool's server-side applies.
-169
View File
@@ -1,169 +0,0 @@
package cli
import (
"context"
"crypto/tls"
"fmt"
"net/http"
"net/url"
"os"
"os/exec"
"runtime"
"strings"
"time"
"github.com/spf13/cobra"
"yuctl/internal/adminapi"
yctx "yuctl/internal/context"
"yuctl/internal/discovery"
)
// adminFlags is the shared flag set for commands that talk to the admin-api.
type adminFlags struct {
adminURL string
insecure bool
reauth bool
noBrowser bool
}
func (f *adminFlags) register(c *cobra.Command) {
c.Flags().StringVar(&f.adminURL, "admin-url", "", "admin-api base URL (default: derived from discovery, or $YUCTL_ADMIN_API_URL)")
c.Flags().BoolVar(&f.insecure, "insecure-skip-tls-verify", false, "skip TLS verification")
c.Flags().BoolVar(&f.reauth, "reauth", false, "force a fresh browser login")
c.Flags().BoolVar(&f.noBrowser, "no-browser", false, "do not auto-open the browser; just print the login URL")
}
func (f *adminFlags) httpClient() *http.Client {
hc := &http.Client{Timeout: 30 * time.Second}
if f.insecure {
tr := http.DefaultTransport.(*http.Transport).Clone()
tr.TLSClientConfig = &tls.Config{InsecureSkipVerify: true}
hc.Transport = tr
}
return hc
}
// deriveAdminURL builds the admin-api origin for the partition's primary
// region from its k8s discovery payload: the Talos API endpoint lives at
// kube.<cluster>.<region>.<provider>.yucca.futo.network, and the admin-api is
// published on the same NetBird overlay zone as admin.<...> (the
// YUCCA_ADMIN_HOST cluster-setting follows the same convention).
func deriveAdminURL(topo *discovery.Topology, partition, region string) (string, error) {
k8s := topo.Kubernetes(partition, region)
if k8s == nil || k8s.APIEndpoint == "" {
return "", fmt.Errorf("no kubernetes discovery payload for %s@%s; pass --admin-url or set YUCTL_ADMIN_API_URL", partition, region)
}
u, err := url.Parse(k8s.APIEndpoint)
if err != nil {
return "", fmt.Errorf("parse api_endpoint %q: %w", k8s.APIEndpoint, err)
}
host, ok := strings.CutPrefix(u.Hostname(), "kube.")
if !ok {
return "", fmt.Errorf("api_endpoint host %q does not start with kube.; pass --admin-url or set YUCTL_ADMIN_API_URL", u.Hostname())
}
return "https://admin." + host, nil
}
// resolveAdminURL applies the override chain: --admin-url > $YUCTL_ADMIN_API_URL
// > derived from the primary region's discovery.
func (f *adminFlags) resolveAdminURL(topo *discovery.Topology, cc *yctx.Context) (string, error) {
if f.adminURL != "" {
return f.adminURL, nil
}
if env := os.Getenv("YUCTL_ADMIN_API_URL"); env != "" {
return env, nil
}
primary := topo.PrimaryRegion(cc.Partition)
if primary == "" {
return "", fmt.Errorf("no primary region found for partition %q (no stack with discovery.role==\"primary\")", cc.Partition)
}
return deriveAdminURL(topo, cc.Partition, primary)
}
// adminLogin returns an authenticated admin-api client, reusing the cached
// per-partition session when valid and running the browser loopback flow
// otherwise.
func (f *adminFlags) adminLogin(ctx context.Context, cmd *cobra.Command, cc *yctx.Context, topo *discovery.Topology) (*adminapi.Client, *adminapi.Token, error) {
adminURL, err := f.resolveAdminURL(topo, cc)
if err != nil {
return nil, nil, err
}
hc := f.httpClient()
token, err := adminapi.LoadToken(cc.Partition)
if err != nil {
return nil, nil, err
}
if f.reauth || !token.Valid() {
var openFn func(string) error
if !f.noBrowser {
openFn = openBrowser
}
fresh, err := adminapi.BrowserLogin(ctx, hc, adminURL, openFn, func(loginURL string) {
out := cmd.ErrOrStderr()
fmt.Fprintln(out, "Complete the login in your browser:")
fmt.Fprintf(out, " %s\n", loginURL)
})
if err != nil {
return nil, nil, fmt.Errorf("browser login: %w", err)
}
token = *fresh
if err := adminapi.SaveToken(cc.Partition, token); err != nil {
return nil, nil, err
}
}
return adminapi.NewClient(adminURL, token, hc), &token, nil
}
// openBrowser opens url in the OS default browser, best-effort.
func openBrowser(url string) error {
switch runtime.GOOS {
case "darwin":
return exec.Command("open", url).Start()
default:
return exec.Command("xdg-open", url).Start()
}
}
func newLoginCmd() *cobra.Command {
flags := &adminFlags{}
c := &cobra.Command{
Use: "login",
Short: "Log in to the partition's yucca-admin-api via the browser",
Long: "Authenticate against the selected partition's admin-api (primary region):\n" +
"opens the admin-api CLI login in your browser, receives a one-time code on a\n" +
"127.0.0.1 listener, exchanges it for a 24h session JWT, and caches it at\n" +
"${XDG_CONFIG_HOME:-~/.config}/yuctl/admin-token-<partition>.json.",
Args: cobra.NoArgs,
RunE: func(cmd *cobra.Command, _ []string) error {
ctx := cmd.Context()
cc, err := requireContext()
if err != nil {
return err
}
topo, err := resolveTopology(ctx)
if err != nil {
return err
}
client, token, err := flags.adminLogin(ctx, cmd, cc, topo)
if err != nil {
return err
}
// Round-trip the session so "login" only succeeds when the API
// actually accepts the token.
sub, err := client.GetAuth(ctx)
if err != nil {
return err
}
fmt.Fprintf(cmd.OutOrStdout(), "logged in to %s as %s (expires %s)\n",
cc.Partition, sub, token.Expiry.Local().Format(time.RFC3339))
return nil
},
}
flags.register(c)
return c
}
-227
View File
@@ -1,227 +0,0 @@
package cli
import (
"encoding/json"
"fmt"
"io"
"os"
"strings"
"text/tabwriter"
"time"
"yuctl/internal/bench"
"yuctl/internal/benchwide"
)
// benchDoView renders StatusReports as a styled dashboard; watch mode feeds
// it successive samples and it keeps a TX history for the sparkline. Styling
// and the meter/sparkline primitives are shared with the warp view.
type benchDoView struct {
label string
history []float64 // combined TX bps per sample
}
func (v *benchDoView) push(combined float64) {
v.history = append(v.history, combined)
if len(v.history) > 60 {
v.history = v.history[len(v.history)-60:]
}
}
func (v *benchDoView) render(r *benchwide.StatusReport, sampledAt time.Time, sampleSec int, watching bool) string {
var b strings.Builder
b.WriteString(warpBadge.Render("BENCH-WIDE") + " " + warpTitle.Render(v.label) + "\n")
if len(r.Droplets) == 0 {
b.WriteString(warpWarnSt.Render("no droplets deployed") +
warpDimSt.Render(" — yuctl tools bench-wide deploy") + "\n")
return warpFrame.Render(strings.TrimRight(b.String(), "\n"))
}
if r.Run != nil {
since := time.Since(r.Run.StartedAt).Round(time.Second).String() + " ago"
p := r.Run.Params
b.WriteString(fmt.Sprintf("%s %s · %s objects · %s/cycle · %s clients/droplet · %s\n",
warpOKSt.Render("● "+r.Run.Label),
warpDimSt.Render("started "+since),
p["obj_size_mib"]+"MiB", p["cycle_size"], p["clients_per_droplet"], p["duration"]))
b.WriteString(warpDimSt.Render("cap "+p["cap_per_droplet"]+" per droplet") + "\n")
} else {
b.WriteString(warpWarnSt.Render("○ no active run recorded") +
warpDimSt.Render(" — load stopped or never started") + "\n")
}
b.WriteString("\n")
// Droplets: TX bar scaled to the busiest droplet, transfer budget bar
// scaled to the cap.
var maxBps float64
nameW := 7
for _, d := range r.Droplets {
maxBps = max(maxBps, d.TxBps)
nameW = max(nameW, len(d.Name))
}
var wire, capTotal int64
var txBps float64
for _, d := range r.Droplets {
capBytes := r.TransferCap
var used int64
state := "-"
if d.Status != nil {
used = d.Status.WireTxBytes
if d.Status.CapBytes > 0 {
capBytes = d.Status.CapBytes
}
state = d.Status.State
}
budget := " "
if capBytes > 0 {
pct := float64(used) / float64(capBytes) * 100
st := warpOKSt
switch {
case pct >= 90:
st = warpErrSt
case pct >= 60:
st = warpWarnSt
}
budget = fmt.Sprintf("%s %s", st.Render(meter(float64(used), float64(capBytes), 10)), st.Render(fmt.Sprintf("%3.0f%%", pct)))
}
line := fmt.Sprintf("%-*s %s %s %s %s %s %s",
nameW, d.Name,
warpDimSt.Render(fmt.Sprintf("%-5s", d.Region)),
warpTxSt.Render(meter(d.TxBps, maxBps, 14)),
padGbps(d.TxBps),
budget,
warpDimSt.Render(fmt.Sprintf("%9s", bench.FormatBytes(used))),
stateCell(d, state))
b.WriteString(line + "\n")
txBps += d.TxBps
wire += used
capTotal += capBytes
}
b.WriteString(fmt.Sprintf("%s %s %s · %s\n",
warpDimSt.Render(fmt.Sprintf("%-*s", nameW+6, "aggregate")),
warpTxSt.Render("TX"), padGbps(txBps),
warpTotalSt.Render(fmt.Sprintf("pool %s / %s", bench.FormatBytes(wire), bench.FormatBytes(capTotal)))))
if len(v.history) > 1 {
b.WriteString(warpDimSt.Render("history ") + warpTotalSt.Render(sparkline(v.history, 40)) + "\n")
}
b.WriteString("\n")
// Clients: loop progress per restic identity.
cnameW := 6
for _, d := range r.Droplets {
if d.Status == nil {
continue
}
for _, c := range d.Status.Clients {
cnameW = max(cnameW, len(shortClient(c.Name)))
}
}
b.WriteString(warpDimSt.Render(fmt.Sprintf("%-*s %-9s %6s %10s %6s %s", cnameW, "CLIENT", "PHASE", "CYCLES", "UPLOADED", "ERR", "LAST ERROR")) + "\n")
for _, d := range r.Droplets {
if d.Status == nil {
if d.Err != "" {
b.WriteString(fmt.Sprintf("%-*s %s\n", cnameW, shortClient(d.Name), warpErrSt.Render(d.Err)))
}
continue
}
for _, c := range d.Status.Clients {
lastErr := ""
if c.LastError != "" {
lastErr = warpErrSt.Render(truncate(c.LastError, 48))
}
b.WriteString(fmt.Sprintf("%-*s %-9s %6d %10s %s %s\n",
cnameW, shortClient(c.Name), phaseCell(c.Phase), c.Cycles,
bench.FormatBytes(c.Uploaded+c.CurrentBytes), errCell6(c.Errors), lastErr))
}
}
footer := fmt.Sprintf("sampled %ds window at %s", sampleSec, sampledAt.Format("15:04:05"))
if watching {
footer += " · ctrl-c to quit"
}
b.WriteString("\n" + warpDimSt.Render(footer))
return warpFrame.Render(b.String())
}
// shortClient drops the shared fleet prefix so tables stay narrow.
func shortClient(name string) string {
return strings.TrimPrefix(name, "yucca-bench-")
}
func phaseCell(phase string) string {
switch phase {
case "backup":
return warpOKSt.Render(fmt.Sprintf("%-9s", phase))
case "generate":
return warpTxSt.Render(fmt.Sprintf("%-9s", phase))
default:
return warpDimSt.Render(fmt.Sprintf("%-9s", phase))
}
}
func errCell6(n int) string {
s := fmt.Sprintf("%6d", n)
if n > 0 {
return warpErrSt.Render(s)
}
return warpDimSt.Render(s)
}
// stateCell colors the droplet's agent state, flagging a dead agent while a
// run is recorded.
func stateCell(d benchwide.DropletStatus, state string) string {
switch {
case !d.Reachable:
return warpErrSt.Render("unreachable")
case state == "running" && d.AgentProcs > 0:
return warpOKSt.Render("running")
case state == "capped":
return warpWarnSt.Render("capped")
case state == "done":
return warpDimSt.Render("done")
case d.AgentProcs == 0 && state == "running":
return warpErrSt.Render("agent dead")
default:
return warpDimSt.Render(state)
}
}
func saveBenchDoResult(path string, r *benchwide.Result) error {
b, err := json.MarshalIndent(r, "", " ")
if err != nil {
return err
}
return os.WriteFile(path, append(b, '\n'), 0o644)
}
func renderBenchDoResult(w io.Writer, r *benchwide.Result) {
fmt.Fprintf(w, "\n%s partition=%s elapsed=%s\n", r.Label, r.Partition, bench.FormatDuration(r.ElapsedSeconds))
tw := tabwriter.NewWriter(w, 2, 4, 2, ' ', 0)
fmt.Fprintln(tw, "DROPLET\tREGION\tSTATE\tWIRE TX\tCLIENT\tCYCLES\tUPLOADED\tERRORS")
for _, d := range r.Droplets {
if len(d.Clients) == 0 {
fmt.Fprintf(tw, "%s\t%s\t%s\t%s\t-\t-\t-\t-\n",
shortClient(d.Name), d.Region, d.State, bench.FormatBytes(d.WireTxBytes))
continue
}
for i, c := range d.Clients {
name, region, state, wireTx := shortClient(d.Name), d.Region, d.State, bench.FormatBytes(d.WireTxBytes)
if i > 0 {
name, region, state, wireTx = "", "", "", ""
}
fmt.Fprintf(tw, "%s\t%s\t%s\t%s\t%s\t%d\t%s\t%d\n",
name, region, state, wireTx, "c"+strings.TrimPrefix(c.Name, d.Name+"-c"), c.Cycles,
bench.FormatBytes(c.Uploaded), c.Errors)
}
}
tw.Flush()
fmt.Fprintf(w, "totals: uploaded=%s wire-tx=%s cycles=%d errors=%d",
bench.FormatBytes(r.TotalUploaded), bench.FormatBytes(r.TotalWireTx), r.TotalCycles, r.TotalErrors)
if r.ElapsedSeconds > 0 && r.TotalUploaded > 0 {
fmt.Fprintf(w, " avg=%s", bench.FormatBPS(float64(r.TotalUploaded)/r.ElapsedSeconds))
}
fmt.Fprintln(w)
}
-506
View File
@@ -1,506 +0,0 @@
package cli
import (
"context"
"fmt"
"net/url"
"os"
"strconv"
"strings"
"text/tabwriter"
"github.com/spf13/cobra"
"yuctl/internal/adminapi"
)
// newUsersCmd builds the `users` subtree.
func newUsersCmd() *cobra.Command {
cmd := &cobra.Command{
Use: "users",
Short: "User administration via yucca-admin-api",
}
cmd.AddCommand(newUsersListCmd())
cmd.AddCommand(newUsersAllowlistCmd())
cmd.AddCommand(newUsersViewDashboardCmd())
cmd.AddCommand(newUsersFeaturesCmd())
cmd.AddCommand(newUsersConnectionsCmd())
return cmd
}
// newUsersFeaturesCmd builds `users features`: per-user feature-flag overrides.
func newUsersFeaturesCmd() *cobra.Command {
cmd := &cobra.Command{
Use: "features",
Short: "Per-user feature-flag overrides",
}
cmd.AddCommand(newUsersFeaturesListCmd())
cmd.AddCommand(newUsersFeaturesSetCmd())
cmd.AddCommand(newUsersFeaturesClearCmd())
return cmd
}
func newUsersFeaturesListCmd() *cobra.Command {
flags := &adminFlags{}
c := &cobra.Command{
Use: "list <email>",
Short: "Show a user's resolved feature flags and overrides",
Args: cobra.ExactArgs(1),
RunE: func(cmd *cobra.Command, args []string) error {
ctx := cmd.Context()
client, _, err := flags.allowlistClient(cmd)
if err != nil {
return err
}
userID, err := resolveUserID(ctx, client, args[0])
if err != nil {
return err
}
features, err := client.GetUserFeatures(ctx, userID)
if err != nil {
return err
}
overridden := map[string]adminapi.FeatureOverride{}
for _, o := range features.Overrides {
overridden[o.Flag] = o
}
w := tabwriter.NewWriter(cmd.OutOrStdout(), 0, 2, 2, ' ', 0)
fmt.Fprintln(w, "FLAG\tVALUE\tSOURCE\tSET BY\tREASON")
for flag, value := range features.Features {
if o, ok := overridden[flag]; ok {
reason := ""
if o.Reason != nil {
reason = *o.Reason
}
fmt.Fprintf(w, "%s\t%t\toverride\t%s\t%s\n", flag, value, o.SetBy, reason)
} else {
fmt.Fprintf(w, "%s\t%t\tdefault\t\t\n", flag, value)
}
}
w.Flush()
return nil
},
}
flags.register(c)
return c
}
func newUsersFeaturesSetCmd() *cobra.Command {
flags := &adminFlags{}
var reason string
c := &cobra.Command{
Use: "set <email> <flag> on|off",
Short: "Set a per-user feature-flag override",
Args: cobra.ExactArgs(3),
RunE: func(cmd *cobra.Command, args []string) error {
var value bool
switch args[2] {
case "on", "true":
value = true
case "off", "false":
value = false
default:
return fmt.Errorf("value must be on|off, got %q", args[2])
}
ctx := cmd.Context()
client, _, err := flags.allowlistClient(cmd)
if err != nil {
return err
}
userID, err := resolveUserID(ctx, client, args[0])
if err != nil {
return err
}
if _, err := client.SetUserFeature(ctx, userID, args[1], value, reason); err != nil {
return err
}
fmt.Fprintf(cmd.ErrOrStderr(), "%s set to %t for %s\n", args[1], value, args[0])
return nil
},
}
c.Flags().StringVar(&reason, "reason", "", "audit note stored on the override")
flags.register(c)
return c
}
func newUsersFeaturesClearCmd() *cobra.Command {
flags := &adminFlags{}
c := &cobra.Command{
Use: "clear <email> <flag>",
Short: "Clear an override (revert to the registry default)",
Args: cobra.ExactArgs(2),
RunE: func(cmd *cobra.Command, args []string) error {
ctx := cmd.Context()
client, _, err := flags.allowlistClient(cmd)
if err != nil {
return err
}
userID, err := resolveUserID(ctx, client, args[0])
if err != nil {
return err
}
if err := client.ClearUserFeature(ctx, userID, args[1]); err != nil {
return err
}
fmt.Fprintf(cmd.ErrOrStderr(), "%s override cleared for %s\n", args[1], args[0])
return nil
},
}
flags.register(c)
return c
}
// newUsersConnectionsCmd builds `users connections`.
func newUsersConnectionsCmd() *cobra.Command {
cmd := &cobra.Command{
Use: "connections",
Short: "A user's connection instances (immich/restic)",
}
flags := &adminFlags{}
list := &cobra.Command{
Use: "list <email>",
Short: "List a user's connections",
Args: cobra.ExactArgs(1),
RunE: func(cmd *cobra.Command, args []string) error {
ctx := cmd.Context()
client, _, err := flags.allowlistClient(cmd)
if err != nil {
return err
}
userID, err := resolveUserID(ctx, client, args[0])
if err != nil {
return err
}
connections, err := client.GetUserConnections(ctx, userID)
if err != nil {
return err
}
w := tabwriter.NewWriter(cmd.OutOrStdout(), 0, 2, 2, ' ', 0)
fmt.Fprintln(w, "ID\tTYPE\tNAME\tCREATED\tLAST SEEN")
for _, connection := range connections {
lastSeen := ""
if connection.LastSeenAt != nil {
lastSeen = *connection.LastSeenAt
}
fmt.Fprintf(w, "%s\t%s\t%s\t%s\t%s\n", connection.ID, connection.Type, connection.Name, connection.CreatedAt, lastSeen)
}
w.Flush()
return nil
},
}
flags.register(list)
cmd.AddCommand(list)
return cmd
}
const defaultGrafanaURL = "https://grafana.futostatus.com"
func newUsersViewDashboardCmd() *cobra.Command {
flags := &adminFlags{}
var userID, email, grafanaURL string
var noOpen bool
c := &cobra.Command{
Use: "view-dashboard",
Short: "Open the per-user Grafana dashboard for a user",
Long: "Build the Grafana per-user drill-down URL (dashboard uid yucca-per-user) for\n" +
"a user and open it in the browser. --id builds the URL without contacting the\n" +
"admin-api; --email resolves the user via the partition's admin-api first.",
Args: cobra.NoArgs,
RunE: func(cmd *cobra.Command, _ []string) error {
id := userID
if email != "" {
client, partition, err := flags.allowlistClient(cmd)
if err != nil {
return err
}
users, err := client.ListUsers(cmd.Context(), 0)
if err != nil {
return err
}
id = ""
for _, u := range users {
if strings.EqualFold(u.Email, email) {
id = u.ID
break
}
}
if id == "" {
return fmt.Errorf("no user with email %q in partition %s", email, partition)
}
}
base := grafanaURL
if base == "" {
base = os.Getenv("YUCTL_GRAFANA_URL")
}
if base == "" {
base = defaultGrafanaURL
}
dashboardURL := strings.TrimRight(base, "/") + "/d/yucca-per-user?var-user=" + url.QueryEscape(id)
fmt.Fprintln(cmd.OutOrStdout(), dashboardURL)
if noOpen {
return nil
}
if err := openBrowser(dashboardURL); err != nil {
return fmt.Errorf("open browser: %w", err)
}
return nil
},
}
c.Flags().StringVar(&userID, "id", "", "user id (uuid)")
c.Flags().StringVar(&email, "email", "", "user email; resolved to an id via the admin-api")
c.Flags().StringVar(&grafanaURL, "grafana-url", "", "Grafana base URL (default: $YUCTL_GRAFANA_URL or "+defaultGrafanaURL+")")
c.Flags().BoolVar(&noOpen, "no-open", false, "print the dashboard URL instead of opening the browser")
c.MarkFlagsOneRequired("id", "email")
c.MarkFlagsMutuallyExclusive("id", "email")
flags.register(c)
return c
}
func newUsersListCmd() *cobra.Command {
flags := &adminFlags{}
var limitFlag string
c := &cobra.Command{
Use: "list",
Short: "List users in the partition's primary region",
RunE: func(cmd *cobra.Command, args []string) error {
ctx := cmd.Context()
cc, err := requireContext()
if err != nil {
return err
}
limit, err := adminapi.ParseLimit(limitFlag)
if err != nil {
return err
}
topo, err := resolveTopology(ctx)
if err != nil {
return err
}
client, _, err := flags.adminLogin(ctx, cmd, cc, topo)
if err != nil {
return err
}
users, err := client.ListUsers(ctx, limit)
if err != nil {
return err
}
w := tabwriter.NewWriter(cmd.OutOrStdout(), 0, 2, 2, ' ', 0)
fmt.Fprintln(w, "ID\tSUB\tNAME\tEMAIL\tDISABLED")
for _, u := range users {
fmt.Fprintf(w, "%s\t%s\t%s\t%s\t%t\n", u.ID, u.Sub, u.Name, u.Email, u.Disabled)
}
w.Flush()
fmt.Fprintf(cmd.ErrOrStderr(), "\n%d user(s) in partition %s\n", len(users), cc.Partition)
return nil
},
}
c.Flags().StringVar(&limitFlag, "limit", "", "page size for the admin-api (default: server default)")
flags.register(c)
return c
}
// newUsersAllowlistCmd builds the `users allowlist` subtree (beta email
// allowlist + invites).
func newUsersAllowlistCmd() *cobra.Command {
cmd := &cobra.Command{
Use: "allowlist",
Short: "Manage the beta email allowlist and invites",
}
cmd.AddCommand(newAllowlistListCmd())
cmd.AddCommand(newAllowlistAddCmd())
cmd.AddCommand(newAllowlistRemoveCmd())
cmd.AddCommand(newAllowlistInviteCmd())
cmd.AddCommand(newAllowlistInviteBatchCmd())
return cmd
}
func (f *adminFlags) allowlistClient(cmd *cobra.Command) (*adminapi.Client, string, error) {
ctx := cmd.Context()
cc, err := requireContext()
if err != nil {
return nil, "", err
}
topo, err := resolveTopology(ctx)
if err != nil {
return nil, "", err
}
client, _, err := f.adminLogin(ctx, cmd, cc, topo)
if err != nil {
return nil, "", err
}
return client, cc.Partition, nil
}
func printAllowlistEntries(cmd *cobra.Command, entries []adminapi.AllowlistEntry) {
w := tabwriter.NewWriter(cmd.OutOrStdout(), 0, 2, 2, ' ', 0)
fmt.Fprintln(w, "EMAIL\tCODE\tINVITED\tUSED\tUSED AT\tCREATED")
for _, e := range entries {
usedAt := ""
if e.InviteUsedAt != nil {
usedAt = *e.InviteUsedAt
}
fmt.Fprintf(w, "%s\t%s\t%t\t%t\t%s\t%s\n", e.Email, e.InviteCode, e.Invited, e.InviteUsed, usedAt, e.CreatedAt)
}
w.Flush()
}
func newAllowlistListCmd() *cobra.Command {
flags := &adminFlags{}
var limitFlag string
c := &cobra.Command{
Use: "list",
Short: "List allowlist entries",
RunE: func(cmd *cobra.Command, args []string) error {
limit, err := adminapi.ParseLimit(limitFlag)
if err != nil {
return err
}
client, partition, err := flags.allowlistClient(cmd)
if err != nil {
return err
}
entries, err := client.ListAllowlist(cmd.Context(), limit)
if err != nil {
return err
}
printAllowlistEntries(cmd, entries)
fmt.Fprintf(cmd.ErrOrStderr(), "\n%d entries in partition %s\n", len(entries), partition)
return nil
},
}
c.Flags().StringVar(&limitFlag, "limit", "", "page size for the admin-api (default: server default)")
flags.register(c)
return c
}
func newAllowlistAddCmd() *cobra.Command {
flags := &adminFlags{}
var staged bool
c := &cobra.Command{
Use: "add <email>",
Short: "Allow an email to sign up (--staged to waitlist it instead)",
Args: cobra.ExactArgs(1),
RunE: func(cmd *cobra.Command, args []string) error {
client, _, err := flags.allowlistClient(cmd)
if err != nil {
return err
}
entry, err := client.AddAllowlistEntry(cmd.Context(), args[0], staged)
if err != nil {
return err
}
printAllowlistEntries(cmd, []adminapi.AllowlistEntry{*entry})
return nil
},
}
c.Flags().BoolVar(&staged, "staged", false, "stage the email without allowing login yet")
flags.register(c)
return c
}
func newAllowlistRemoveCmd() *cobra.Command {
flags := &adminFlags{}
c := &cobra.Command{
Use: "remove <email>",
Short: "Remove an email from the allowlist",
Args: cobra.ExactArgs(1),
RunE: func(cmd *cobra.Command, args []string) error {
client, _, err := flags.allowlistClient(cmd)
if err != nil {
return err
}
if err := client.RemoveAllowlistEntry(cmd.Context(), args[0]); err != nil {
return err
}
fmt.Fprintf(cmd.ErrOrStderr(), "removed %s\n", args[0])
return nil
},
}
flags.register(c)
return c
}
func newAllowlistInviteCmd() *cobra.Command {
flags := &adminFlags{}
c := &cobra.Command{
Use: "invite <email>[,<email>...]",
Short: "Invite emails: allow them to sign up, creating entries as needed",
Args: cobra.MinimumNArgs(1),
RunE: func(cmd *cobra.Command, args []string) error {
var emails []string
for _, arg := range args {
for _, email := range strings.Split(arg, ",") {
if email = strings.TrimSpace(email); email != "" {
emails = append(emails, email)
}
}
}
client, _, err := flags.allowlistClient(cmd)
if err != nil {
return err
}
entries, err := client.InviteEmails(cmd.Context(), emails)
if err != nil {
return err
}
printAllowlistEntries(cmd, entries)
return nil
},
}
flags.register(c)
return c
}
func newAllowlistInviteBatchCmd() *cobra.Command {
flags := &adminFlags{}
c := &cobra.Command{
Use: "invite-batch <count>",
Short: "Invite the oldest <count> staged (waitlisted) emails",
Args: cobra.ExactArgs(1),
RunE: func(cmd *cobra.Command, args []string) error {
count, err := strconv.Atoi(args[0])
if err != nil || count < 1 {
return fmt.Errorf("count must be a positive integer")
}
client, _, err := flags.allowlistClient(cmd)
if err != nil {
return err
}
entries, err := client.InviteBatch(cmd.Context(), count)
if err != nil {
return err
}
printAllowlistEntries(cmd, entries)
fmt.Fprintf(cmd.ErrOrStderr(), "\ninvited %d entries\n", len(entries))
return nil
},
}
flags.register(c)
return c
}
// resolveUserID turns an --user email into a user id via the admin-api.
func resolveUserID(ctx context.Context, client *adminapi.Client, email string) (string, error) {
users, err := client.ListUsers(ctx, 0)
if err != nil {
return "", err
}
for _, u := range users {
if strings.EqualFold(u.Email, email) {
return u.ID, nil
}
}
return "", fmt.Errorf("no user with email %q", email)
}
-194
View File
@@ -1,194 +0,0 @@
package cli
import (
"fmt"
"strings"
"time"
"github.com/charmbracelet/lipgloss"
"yuctl/internal/warp"
)
// warpView renders StatusReports as a styled dashboard; watch mode feeds it
// successive samples and it keeps a throughput history for the sparkline.
type warpView struct {
label string
history []float64 // combined bps per sample
}
var (
warpAccent = lipgloss.AdaptiveColor{Light: "#6C50FF", Dark: "#9D7CFF"}
warpOK = lipgloss.AdaptiveColor{Light: "#12A150", Dark: "#2ECC71"}
warpErr = lipgloss.AdaptiveColor{Light: "#D0021B", Dark: "#FF5F56"}
warpWarn = lipgloss.AdaptiveColor{Light: "#B8860B", Dark: "#F5C542"}
warpDim = lipgloss.AdaptiveColor{Light: "244", Dark: "241"}
warpTx = lipgloss.AdaptiveColor{Light: "#0087AF", Dark: "#33D1E0"}
warpRx = lipgloss.AdaptiveColor{Light: "#AF5F00", Dark: "#F5A623"}
warpBadge = lipgloss.NewStyle().Bold(true).Foreground(lipgloss.Color("#FFFFFF")).Background(warpAccent).Padding(0, 1)
warpTitle = lipgloss.NewStyle().Bold(true)
warpDimSt = lipgloss.NewStyle().Foreground(warpDim)
warpOKSt = lipgloss.NewStyle().Foreground(warpOK)
warpErrSt = lipgloss.NewStyle().Bold(true).Foreground(warpErr)
warpWarnSt = lipgloss.NewStyle().Foreground(warpWarn)
warpTxSt = lipgloss.NewStyle().Foreground(warpTx)
warpRxSt = lipgloss.NewStyle().Foreground(warpRx)
warpTotalSt = lipgloss.NewStyle().Bold(true).Foreground(warpAccent)
warpFrame = lipgloss.NewStyle().Border(lipgloss.RoundedBorder()).BorderForeground(warpAccent).Padding(0, 1)
)
func (v *warpView) push(combined float64) {
v.history = append(v.history, combined)
if len(v.history) > 60 {
v.history = v.history[len(v.history)-60:]
}
}
func (v *warpView) render(r *warp.StatusReport, sampledAt time.Time, sampleSec int, watching bool) string {
var b strings.Builder
b.WriteString(warpBadge.Render("WARP") + " " + warpTitle.Render(v.label) + "\n")
if len(r.Pods) == 0 {
b.WriteString(warpWarnSt.Render("no runner pods deployed") +
warpDimSt.Render(" — yuctl tools warp deploy") + "\n")
return warpFrame.Render(strings.TrimRight(b.String(), "\n"))
}
if r.Config != nil {
since := r.Config["started_at"]
if t, err := time.Parse(time.RFC3339, since); err == nil {
since = time.Since(t).Round(time.Second).String() + " ago"
}
b.WriteString(fmt.Sprintf("%s %s · %s PUT / %s GET · %s/%s objects · %s RGWs\n",
warpOKSt.Render("● "+r.Config["mode"]),
warpDimSt.Render("started "+since),
r.Config["put_streams"], r.Config["get_streams"],
r.Config["put_obj_size"], r.Config["get_obj_size"],
r.Config["rgw_endpoints"]))
b.WriteString(warpDimSt.Render("endpoint "+r.Config["endpoint"]) + "\n")
} else {
b.WriteString(warpWarnSt.Render("○ no active run recorded") +
warpDimSt.Render(" — load stopped or never started") + "\n")
}
b.WriteString("\n")
// Nodes: TX/RX bars scaled to the busiest direction in this frame.
var maxBps float64
for _, n := range r.Nodes {
maxBps = max(maxBps, max(n.TxBps, n.RxBps))
}
nodeW, ifaceW := 4, 5
for _, n := range r.Nodes {
nodeW = max(nodeW, len(n.Node))
ifaceW = max(ifaceW, len(n.Iface))
}
var tx, rx float64
for _, n := range r.Nodes {
b.WriteString(fmt.Sprintf("%-*s %s %s %s %s %s\n",
nodeW, n.Node,
warpDimSt.Render(fmt.Sprintf("%-*s", ifaceW, n.Iface)),
warpTxSt.Render(meter(n.TxBps, maxBps, 14)),
padGbps(n.TxBps),
warpRxSt.Render(meter(n.RxBps, maxBps, 14)),
padGbps(n.RxBps)))
tx += n.TxBps
rx += n.RxBps
}
if len(r.Nodes) > 0 {
b.WriteString(fmt.Sprintf("%s %s %s · %s %s · %s\n",
warpDimSt.Render(fmt.Sprintf("%-*s", nodeW+ifaceW-2, "aggregate")),
warpTxSt.Render("TX"), padGbps(tx),
warpRxSt.Render("RX"), padGbps(rx),
warpTotalSt.Render("combined "+fmtGbps(tx+rx))))
if len(v.history) > 1 {
b.WriteString(warpDimSt.Render("history ") + warpTotalSt.Render(sparkline(v.history, 40)) + "\n")
}
b.WriteString("\n")
}
// Pods: process liveness and log error counts.
podW := 3
for _, p := range r.Pods {
podW = max(podW, len(p.Name))
}
b.WriteString(warpDimSt.Render(fmt.Sprintf("%-*s %-*s %6s %6s %10s %10s", podW, "POD", nodeW, "NODE", "PUT", "GET", "ERR(PUT)", "ERR(GET)")) + "\n")
for _, p := range r.Pods {
b.WriteString(fmt.Sprintf("%-*s %-*s %s %s %s %s\n",
podW, p.Name, nodeW, p.Node,
procCell(p.PutProcs, r.Config != nil),
procCell(p.GetProcs, r.Config != nil),
errCell(p.PutErrors), errCell(p.GetErrors)))
}
footer := fmt.Sprintf("sampled %ds window at %s", sampleSec, sampledAt.Format("15:04:05"))
if watching {
footer += " · ctrl-c to quit"
}
b.WriteString("\n" + warpDimSt.Render(footer))
return warpFrame.Render(b.String())
}
// procCell colors a warp process count: green when alive, red when a run is
// recorded but nothing is running, dim otherwise.
func procCell(n int, runRecorded bool) string {
s := fmt.Sprintf("%6d", n)
switch {
case n > 0:
return warpOKSt.Render(s)
case runRecorded:
return warpErrSt.Render(s)
default:
return warpDimSt.Render(s)
}
}
func errCell(n int) string {
s := fmt.Sprintf("%10d", n)
if n > 0 {
return warpErrSt.Render(s)
}
return warpDimSt.Render(s)
}
func fmtGbps(bps float64) string {
return fmt.Sprintf("%.2f Gbps", bps/1e9)
}
func padGbps(bps float64) string {
return fmt.Sprintf("%11s", fmtGbps(bps))
}
// meter renders a horizontal bar of width w, filled to value/scale.
func meter(value, scale float64, w int) string {
filled := 0
if scale > 0 {
filled = int(value / scale * float64(w))
}
filled = min(max(filled, 0), w)
return strings.Repeat("█", filled) + strings.Repeat("░", w-filled)
}
// sparkline renders the last len(vals) samples with block glyphs, scaled to
// the window maximum, downsampled to at most w points.
func sparkline(vals []float64, w int) string {
if len(vals) > w {
vals = vals[len(vals)-w:]
}
var maxV float64
for _, v := range vals {
maxV = max(maxV, v)
}
if maxV == 0 {
return strings.Repeat("▁", len(vals))
}
glyphs := []rune("▁▂▃▄▅▆▇█")
var b strings.Builder
for _, v := range vals {
i := int(v / maxV * float64(len(glyphs)-1))
b.WriteRune(glyphs[min(max(i, 0), len(glyphs)-1)])
}
return b.String()
}
+2 -2
View File
@@ -1,6 +1,6 @@
// Command yuctl is the yucca operations CLI. It resolves the
// partition/region/ceph topology from Terraform "discovery" outputs (read
// directly from S3 state) and drives day-2 operations. See internal/cli for the
// directly from S3 state) and drives day-2 operations. See yuctl/cli for the
// command tree and packages/yuctl/README.md for usage.
package main
@@ -12,7 +12,7 @@ import (
"github.com/rs/zerolog/log"
"yuctl/internal/cli"
"yuctl/cli"
)
func main() {
@@ -1,6 +1,6 @@
// Package netdev parses /proc/net/dev counter snapshots. Shared by everything
// that measures NIC throughput from that file — the warp status sampler and
// the bench-do agent/orchestrator — so virtual-interface filtering stays in
// the fleet-bench agent/orchestrator — so virtual-interface filtering stays in
// one place.
package netdev
@@ -5,10 +5,10 @@ import (
"strconv"
"time"
"yuctl/internal/do"
"yuctl/do"
)
// doProvider adapts the DigitalOcean client (internal/do) to Provider. It owns
// doProvider adapts the DigitalOcean client (yuctl/do) to Provider. It owns
// no logic of its own — just the Host/Size translation and the int⇄string id
// bridging Provider's opaque ids require.
type doProvider struct{ c *do.Client }
@@ -14,19 +14,19 @@ import (
"github.com/rs/zerolog/log"
"yuctl/internal/op"
"yuctl/op"
)
// hetznerTokenRef is the 1Password item holding the Hetzner Cloud API token;
// $HCLOUD_TOKEN skips op entirely, $YUCTL_HCLOUD_TOKEN_REF overrides the ref.
const hetznerTokenRef = "op://yucca_tf_prod/HCLOUD_API_TOKEN/password"
// fleetLabel is the Hetzner label key carrying the bench-wide fleet tag; the
// fleetLabel is the Hetzner label key carrying the fleet-bench fleet tag; the
// tag is the value, and list/destroy select on it.
const fleetLabel = "yuctl-fleet"
// hetznerProvider talks to the Hetzner Cloud API (https://api.hetzner.cloud/v1)
// over plain net/http — the surface bench-wide needs is small enough that a
// over plain net/http — the surface fleet-bench needs is small enough that a
// full SDK dependency isn't worth it.
type hetznerProvider struct {
token string
@@ -1,4 +1,4 @@
// Package provider abstracts the cloud that hosts a bench-wide fleet so the
// Package provider abstracts the cloud that hosts a fleet-bench fleet so the
// fleet lifecycle (deploy/start/status/stop/cleanup/undeploy) is identical
// across providers. Each provider is pure VM plumbing — create/list/destroy
// tagged hosts, upload an ephemeral ssh key, resolve a size's price and
@@ -1,4 +1,4 @@
package bench
package resticbench
import (
"context"
@@ -1,4 +1,4 @@
package bench
package resticbench
import (
"crypto/sha256"
@@ -1,4 +1,4 @@
package bench
package resticbench
// Phase names, in execution order within a cell.
const (
@@ -1,4 +1,4 @@
package bench
package resticbench
import (
"bufio"
@@ -1,6 +1,6 @@
//go:build !embedagent
package bench
package resticbench
import "errors"
@@ -1,6 +1,6 @@
//go:build embedagent
package bench
package resticbench
import (
"bytes"
@@ -1,4 +1,10 @@
package bench
package resticbench
import (
"bufio"
"encoding/json"
"io"
)
// Event is one line of the agent→orchestrator stream (JSON per line on the
// agent's stdout).
@@ -13,3 +19,17 @@ type Event struct {
PhaseResult *PhaseResult `json:"phaseResult,omitempty"`
Result *RunResult `json:"result,omitempty"`
}
// Non-JSON lines are skipped: agent stderr never lands here, but a remote
// shell may still chirp.
func ScanEvents(r io.Reader, handle func(Event)) error {
sc := bufio.NewScanner(r)
sc.Buffer(make([]byte, 1<<20), 8<<20)
for sc.Scan() {
var ev Event
if json.Unmarshal(sc.Bytes(), &ev) == nil {
handle(ev)
}
}
return sc.Err()
}
@@ -1,4 +1,4 @@
package bench
package resticbench
import (
"bytes"
@@ -1,4 +1,4 @@
package bench
package resticbench
import (
"context"
@@ -11,10 +11,10 @@ import (
"sync"
"time"
"yuctl/internal/netdev"
"yuctl/netdev"
)
// Loadgen is the bench-do agent mode: a supervisor that runs one continuous
// Loadgen is the fleet-bench agent mode: a supervisor that runs one continuous
// restic backup loop per client until the duration elapses, the droplet's
// transfer cap is hit, or it is killed. Unlike the bench mode it is built to
// run detached (nohup) on a droplet: progress goes to a status file the
@@ -100,7 +100,7 @@ func (c *LoadgenConfig) defaults() error {
}
// LoadgenStatus is the droplet-local progress file, rewritten atomically every
// few seconds and read (plus a live NIC sample) by `bench-do status`.
// few seconds and read (plus a live NIC sample) by `fleet-bench status`.
type LoadgenStatus struct {
StartedAt time.Time `json:"startedAt"`
UpdatedAt time.Time `json:"updatedAt"`
@@ -344,7 +344,7 @@ func clientLoop(ctx context.Context, cfg LoadgenConfig, idx int, cl LoadgenClien
}
}
// RunLoadgenCleanup forgets and prunes every bench-do snapshot in each
// RunLoadgenCleanup forgets and prunes every fleet-bench snapshot in each
// client's repository, streaming bench Events (it runs synchronously over the
// orchestrator's ssh session, unlike the detached load mode).
func RunLoadgenCleanup(ctx context.Context, cfg LoadgenConfig, emit func(Event)) error {
@@ -1,4 +1,4 @@
package bench
package resticbench
import "testing"
@@ -1,20 +1,20 @@
package bench
package resticbench
import (
"bufio"
"bytes"
"context"
"encoding/json"
"fmt"
"io"
"os"
"os/exec"
"github.com/rs/zerolog/log"
"yuctl/sshx"
)
// RemoteBinDir is where the agent and restic land on a remote host, relative
// to $HOME. Shared with bench-do, which pushes the same binaries to droplets.
// to $HOME. Shared with fleet-bench, which pushes the same binaries to its
// hosts.
const RemoteBinDir = ".cache/yuctl-bench/bin"
const remoteDir = RemoteBinDir
@@ -25,7 +25,19 @@ type RunOpts struct {
SSHIdentity string // ssh private key file ("" = ssh defaults/agent)
AgentBin string // local linux/amd64 bench-agent; "" = use the embedded one
Config Config
Out string // local results path ("" = don't save)
Out string // local results path ("" = don't save)
Summary io.Writer // summary destination (nil = stdout)
}
func (o *RunOpts) ssh() *sshx.Client {
return &sshx.Client{IdentityFile: o.SSHIdentity, Retries: 2}
}
func (o *RunOpts) summary() io.Writer {
if o.Summary != nil {
return o.Summary
}
return os.Stdout
}
// Run pushes the agent + pinned restic to the host, streams the run, and
@@ -41,18 +53,19 @@ func Run(ctx context.Context, opts RunOpts) (*RunResult, error) {
return nil, err
}
ssh := opts.ssh()
log.Info().Str("host", opts.Host).Msg("pushing agent + restic " + ResticVersion)
if err := push(ctx, opts.SSHIdentity, opts.Host, agentBin, resticBin); err != nil {
if err := ssh.Push(ctx, opts.Host, remoteDir, map[string]string{agentBin: "bench-agent", resticBin: "restic"}); err != nil {
return nil, err
}
log.Info().Str("host", opts.Host).Ints("connections", opts.Config.Connections).
Str("size", FormatBytes(opts.Config.Size)).Msg("starting remote benchmark")
result, err := drive(ctx, opts.SSHIdentity, opts.Host, opts.Config)
result, err := drive(ctx, ssh, opts.Host, opts.Config)
if err != nil {
return nil, err
}
return finish(result, opts.Out)
return finish(result, opts.Out, opts.summary())
}
// RunHere executes the benchmark on the local machine — no ssh, the agent
@@ -74,50 +87,20 @@ func RunHere(ctx context.Context, opts RunOpts) (*RunResult, error) {
if sink.result == nil {
return nil, fmt.Errorf("benchmark finished without a result")
}
return finish(sink.result, opts.Out)
return finish(sink.result, opts.Out, opts.summary())
}
func finish(result *RunResult, out string) (*RunResult, error) {
func finish(result *RunResult, out string, summary io.Writer) (*RunResult, error) {
if out != "" {
if err := SaveResult(out, result); err != nil {
return nil, err
}
log.Info().Str("path", out).Msg("results saved")
}
RenderSummary(os.Stdout, result)
RenderSummary(summary, result)
return result, nil
}
// sshOpts builds the common ssh/scp options. accept-new (TOFU) keeps first
// contact with a discovery-resolved IP from failing BatchMode.
func sshOpts(identity string) []string {
args := []string{
"-o", "BatchMode=yes",
"-o", "StrictHostKeyChecking=accept-new",
"-o", "ServerAliveInterval=30",
"-o", "ServerAliveCountMax=8",
}
if identity != "" {
args = append(args, "-i", identity, "-o", "IdentitiesOnly=yes")
}
return args
}
func sshBase(identity, host string) []string {
return append(sshOpts(identity), host)
}
func run(ctx context.Context, name string, args ...string) error {
cmd := exec.CommandContext(ctx, name, args...)
var out bytes.Buffer
cmd.Stdout = &out
cmd.Stderr = &out
if err := cmd.Run(); err != nil {
return fmt.Errorf("%s %v: %w: %s", name, args, err, tail(out.String(), 1000))
}
return nil
}
// AgentBinary resolves the local linux/amd64 agent to push: an explicit path,
// or the embedded one materialized into a temp file. The returned cleanup
// removes the materialized copy.
@@ -144,24 +127,10 @@ func AgentBinary(explicit string) (string, func(), error) {
return path, func() { os.RemoveAll(dir) }, nil
}
func push(ctx context.Context, identity, host, agentBin, resticBin string) error {
if err := run(ctx, "ssh", append(sshBase(identity, host), "mkdir -p "+remoteDir)...); err != nil {
return err
}
scp := append([]string{"-q"}, sshOpts(identity)...)
if err := run(ctx, "scp", append(scp, agentBin, host+":"+remoteDir+"/bench-agent")...); err != nil {
return err
}
if err := run(ctx, "scp", append(scp, resticBin, host+":"+remoteDir+"/restic")...); err != nil {
return err
}
return run(ctx, "ssh", append(sshBase(identity, host), "chmod +x "+remoteDir+"/bench-agent "+remoteDir+"/restic")...)
}
// drive runs the remote agent, feeding Config over stdin and consuming the
// event stream from stdout. The agent's stderr passes straight through.
func drive(ctx context.Context, identity, host string, cfg Config) (*RunResult, error) {
cmd := exec.CommandContext(ctx, "ssh", append(sshBase(identity, host), remoteDir+"/bench-agent")...)
func drive(ctx context.Context, ssh *sshx.Client, host string, cfg Config) (*RunResult, error) {
cmd := ssh.Command(ctx, host, remoteDir+"/bench-agent")
cmd.Stderr = os.Stderr
stdin, err := cmd.StdinPipe()
@@ -182,15 +151,7 @@ func drive(ctx context.Context, identity, host string, cfg Config) (*RunResult,
}()
sink := &eventSink{}
sc := bufio.NewScanner(stdout)
sc.Buffer(make([]byte, 1<<20), 8<<20)
for sc.Scan() {
var ev Event
if err := json.Unmarshal(sc.Bytes(), &ev); err != nil {
continue
}
sink.handle(ev)
}
_ = ScanEvents(stdout, sink.handle)
waitErr := cmd.Wait()
if sink.fatal != "" {
return nil, fmt.Errorf("agent: %s", sink.fatal)
@@ -1,4 +1,4 @@
package bench
package resticbench
import (
"bufio"
@@ -1,4 +1,4 @@
package bench
package resticbench
import (
"encoding/json"
@@ -2,7 +2,7 @@
// orchestrator on the dev machine pushes an agent (yucca-bench itself, built
// for linux) plus a pinned restic binary to a management host, runs
// write/incremental/restore phases there, and reports results locally.
package bench
package resticbench
import (
"fmt"
+158
View File
@@ -0,0 +1,158 @@
// Package sshx is yuctl's single ssh/scp layer, shelling out to the system
// OpenSSH (agent support, ssh_config, ControlMaster — a Go ssh library would
// reimplement all three badly). Scripts travel as the ssh command argument —
// visible in remote ps, so they must never contain secrets; secret payloads
// go through stdin.
package sshx
import (
"bytes"
"context"
"errors"
"fmt"
"os/exec"
"path/filepath"
"sort"
"strconv"
"strings"
"time"
)
// Client carries one target family's connection policy. The zero value is a
// plain BatchMode/TOFU client using the operator's ssh defaults and agent.
type Client struct {
// User is prepended to hosts that don't already embed one ("" = ssh
// config).
User string
// IdentityFile pins a private key (with IdentitiesOnly); "" = defaults.
IdentityFile string
// KnownHostsFile overrides the known-hosts file. Fleets with recycled
// provider IPs need their own — the operator's global file would scream
// host-key-changed.
KnownHostsFile string
// ControlDir enables connection multiplexing with masters persisted
// under it. Multiplexing matters beyond latency: fan-out commands
// otherwise open a fresh port-22 TCP flow per host per command, a burst
// pattern that trips ssh-targeted rate limiting on some paths (observed
// as flapping per-IP port-22 SYN drops while ICMP and other ports stay
// fine). One persistent master per host keeps the flow count flat.
ControlDir string
// ConnectTimeoutSeconds bounds connection establishment (0 = ssh default).
ConnectTimeoutSeconds int
// Retries is how many times Run re-attempts after a connection-level
// failure (ssh exit 255: banner timeouts, resets — common when tens of
// sessions open against fresh hosts). Remote command failures are never
// retried; ssh reports those as the command's own exit code.
Retries int
}
func (c *Client) dest(host string) string {
if c.User != "" && !strings.Contains(host, "@") {
return c.User + "@" + host
}
return host
}
// options are the -o/-i arguments shared by ssh and scp. accept-new (TOFU)
// keeps first contact with a freshly resolved IP from failing BatchMode.
func (c *Client) options() []string {
args := []string{
"-o", "BatchMode=yes",
"-o", "StrictHostKeyChecking=accept-new",
"-o", "ServerAliveInterval=30",
"-o", "ServerAliveCountMax=8",
}
if c.KnownHostsFile != "" {
args = append(args, "-o", "UserKnownHostsFile="+c.KnownHostsFile)
}
if c.ConnectTimeoutSeconds > 0 {
args = append(args, "-o", "ConnectTimeout="+strconv.Itoa(c.ConnectTimeoutSeconds))
}
if c.ControlDir != "" {
args = append(args,
"-o", "ControlMaster=auto",
"-o", "ControlPath="+filepath.Join(c.ControlDir, "%C"),
"-o", "ControlPersist=300")
}
if c.IdentityFile != "" {
args = append(args, "-i", c.IdentityFile, "-o", "IdentitiesOnly=yes")
}
return args
}
func (c *Client) Run(ctx context.Context, host, script string, stdin []byte) (string, error) {
var lastOut string
var lastErr error
for attempt := 1; attempt <= c.Retries+1; attempt++ {
out, err := c.RunOnce(ctx, host, script, stdin)
if err == nil {
return out, nil
}
lastOut, lastErr = out, err
var exit *exec.ExitError
if ctx.Err() != nil || !errors.As(err, &exit) || exit.ExitCode() != 255 {
break
}
select {
case <-ctx.Done():
return lastOut, lastErr
case <-time.After(time.Duration(attempt) * 5 * time.Second):
}
}
return lastOut, lastErr
}
// RunOnce is a single ssh attempt with no retry — for callers with their own
// retry cadence (readiness polling), where nesting retries multiplies delays.
func (c *Client) RunOnce(ctx context.Context, host, script string, stdin []byte) (string, error) {
cmd := exec.CommandContext(ctx, "ssh", append(c.options(), c.dest(host), script)...)
if stdin != nil {
cmd.Stdin = bytes.NewReader(stdin)
}
var out, errb bytes.Buffer
cmd.Stdout = &out
cmd.Stderr = &errb
if err := cmd.Run(); err != nil {
return out.String(), fmt.Errorf("ssh %s: %w: %s", host, err, Tail(errb.String(), 1000))
}
return out.String(), nil
}
// Command is for long-lived streaming sessions; the caller owns the pipes and
// lifecycle.
func (c *Client) Command(ctx context.Context, host, remote string) *exec.Cmd {
return exec.CommandContext(ctx, "ssh", append(c.options(), c.dest(host), remote)...)
}
// Push copies local files (keys) into remoteDir on host under the given
// remote names (values), and marks them executable.
func (c *Client) Push(ctx context.Context, host, remoteDir string, files map[string]string) error {
if _, err := c.Run(ctx, host, "mkdir -p "+remoteDir, nil); err != nil {
return err
}
names := make([]string, 0, len(files))
for local, name := range files {
cmd := exec.CommandContext(ctx, "scp",
append(append([]string{"-q"}, c.options()...), local, c.dest(host)+":"+remoteDir+"/"+name)...)
if out, err := cmd.CombinedOutput(); err != nil {
return fmt.Errorf("scp %s to %s: %w: %s", local, host, err, Tail(string(out), 500))
}
names = append(names, remoteDir+"/"+name)
}
sort.Strings(names)
_, err := c.Run(ctx, host, "chmod +x "+strings.Join(names, " "), nil)
return err
}
func Tail(v string, n int) string {
v = strings.TrimSpace(v)
if len(v) > n {
return "…" + v[len(v)-n:]
}
return v
}
@@ -1,7 +1,7 @@
// Package k8s wraps Talos node operations (talosctl) for a region's K8s cluster,
// Package talos wraps Talos node operations (talosctl) for a region's K8s cluster,
// resolving the talosconfig from 1Password to a 0600 temp file and driving
// `talosctl upgrade` against the control-plane node IPs from discovery.
package k8s
package talos
import (
"context"
@@ -11,8 +11,8 @@ import (
"github.com/rs/zerolog"
"yuctl/internal/op"
"yuctl/internal/state"
"yuctl/op"
"yuctl/state"
)
// UpgradeOptions controls a talos upgrade run.
@@ -28,7 +28,7 @@ type UpgradeOptions struct {
// TalosUpgrade resolves the talosconfig referenced by the kubernetes payload and
// runs `talosctl upgrade` against each control-plane node. The temp talosconfig
// is removed on return. With DryRun the commands are only logged.
func TalosUpgrade(ctx context.Context, k state.Kubernetes, opts UpgradeOptions, logger zerolog.Logger) error {
func Upgrade(ctx context.Context, k state.Kubernetes, opts UpgradeOptions, logger zerolog.Logger) error {
nodes := opts.Nodes
if len(nodes) == 0 {
nodes = k.CPNodeIPs
+20
View File
@@ -0,0 +1,20 @@
// Package ui is the terminal presentation layer. Commands write through an
// IOStreams instead of os.Stdout/os.Stderr so tests can capture output; Out
// is for the command's payload (parseable, redirectable), Err for progress
// and human-only chatter.
package ui
import (
"io"
"os"
)
type IOStreams struct {
In io.Reader
Out io.Writer
Err io.Writer
}
func System() *IOStreams {
return &IOStreams{In: os.Stdin, Out: os.Stdout, Err: os.Stderr}
}
+26
View File
@@ -0,0 +1,26 @@
package ui
import "github.com/charmbracelet/lipgloss"
var (
accent = lipgloss.AdaptiveColor{Light: "#6C50FF", Dark: "#9D7CFF"}
good = lipgloss.AdaptiveColor{Light: "#12A150", Dark: "#2ECC71"}
bad = lipgloss.AdaptiveColor{Light: "#D0021B", Dark: "#FF5F56"}
warn = lipgloss.AdaptiveColor{Light: "#B8860B", Dark: "#F5C542"}
muted = lipgloss.AdaptiveColor{Light: "244", Dark: "241"}
tx = lipgloss.AdaptiveColor{Light: "#0087AF", Dark: "#33D1E0"}
rx = lipgloss.AdaptiveColor{Light: "#AF5F00", Dark: "#F5A623"}
)
var (
Badge = lipgloss.NewStyle().Bold(true).Foreground(lipgloss.Color("#FFFFFF")).Background(accent).Padding(0, 1)
Title = lipgloss.NewStyle().Bold(true)
Muted = lipgloss.NewStyle().Foreground(muted)
OK = lipgloss.NewStyle().Foreground(good)
Bad = lipgloss.NewStyle().Bold(true).Foreground(bad)
Warn = lipgloss.NewStyle().Foreground(warn)
TX = lipgloss.NewStyle().Foreground(tx)
RX = lipgloss.NewStyle().Foreground(rx)
Total = lipgloss.NewStyle().Bold(true).Foreground(accent)
Frame = lipgloss.NewStyle().Border(lipgloss.RoundedBorder()).BorderForeground(accent).Padding(0, 1)
)
+61
View File
@@ -0,0 +1,61 @@
package ui
import (
"fmt"
"strings"
)
// Meter renders a horizontal bar of width w, filled to value/scale.
func Meter(value, scale float64, w int) string {
filled := 0
if scale > 0 {
filled = int(value / scale * float64(w))
}
filled = min(max(filled, 0), w)
return strings.Repeat("█", filled) + strings.Repeat("░", w-filled)
}
// Sparkline renders the last len(vals) samples with block glyphs, scaled to
// the window maximum, downsampled to at most w points.
func Sparkline(vals []float64, w int) string {
if len(vals) > w {
vals = vals[len(vals)-w:]
}
var maxV float64
for _, v := range vals {
maxV = max(maxV, v)
}
if maxV == 0 {
return strings.Repeat("▁", len(vals))
}
glyphs := []rune("▁▂▃▄▅▆▇█")
var b strings.Builder
for _, v := range vals {
i := int(v / maxV * float64(len(glyphs)-1))
b.WriteRune(glyphs[min(max(i, 0), len(glyphs)-1)])
}
return b.String()
}
func FmtGbps(bps float64) string {
return fmt.Sprintf("%.2f Gbps", bps/1e9)
}
func PadGbps(bps float64) string {
return fmt.Sprintf("%11s", FmtGbps(bps))
}
func ErrCell(n, width int) string {
s := fmt.Sprintf("%*d", width, n)
if n > 0 {
return Bad.Render(s)
}
return Muted.Render(s)
}
func Truncate(s string, n int) string {
if len(s) <= n {
return s
}
return s[:n] + "…"
}
+18
View File
@@ -0,0 +1,18 @@
package ui
import (
"strings"
"testing"
)
func TestSparklineAndMeter(t *testing.T) {
if got := Sparkline([]float64{0, 1, 2, 4}, 10); len([]rune(got)) != 4 {
t.Errorf("sparkline length: %q", got)
}
if got := Meter(50, 100, 10); !strings.HasPrefix(got, "█████░") {
t.Errorf("meter: %q", got)
}
if got := Meter(0, 0, 4); got != "░░░░" {
t.Errorf("empty meter: %q", got)
}
}