mirror of
https://github.com/immich-app/yucca.git
synced 2026-09-30 13:33:00 +08:00
chore(yuctl): restructure and make pretty (#485)
This commit is contained in:
+1
-1
@@ -14,7 +14,7 @@ node_modules
|
||||
mise.local.toml
|
||||
|
||||
dist/
|
||||
packages/yuctl/internal/bench/bench-agent-linux-amd64.gz
|
||||
packages/yuctl/resticbench/bench-agent-linux-amd64.gz
|
||||
packages/michael/michael
|
||||
|
||||
# k3d/Tilt dev stack — Helm builds subchart snapshots on the fly; the .dev/
|
||||
|
||||
@@ -9,6 +9,6 @@ cd packages/yuctl
|
||||
# `yuctl tools bench` is self-contained. Plain `go build` still works without
|
||||
# the tag — the embedded agent is then absent and --agent-bin is required.
|
||||
CGO_ENABLED=0 GOOS=linux GOARCH=amd64 go build -o ../../dist/bench-agent-linux-amd64 ./cmd/bench-agent
|
||||
gzip -9 -n -c ../../dist/bench-agent-linux-amd64 > internal/bench/bench-agent-linux-amd64.gz
|
||||
gzip -9 -n -c ../../dist/bench-agent-linux-amd64 > resticbench/bench-agent-linux-amd64.gz
|
||||
|
||||
go build -tags embedagent -o ../../dist/yuctl .
|
||||
|
||||
+89
-63
@@ -12,12 +12,19 @@ references, never values).
|
||||
|
||||
## Conventions
|
||||
|
||||
Built to match `packages/michael`: `module yuctl`, Go 1.25, `main.go` +
|
||||
`internal/<pkg>`, `aws-sdk-go-v2` for S3, `rs/zerolog` for logging. The one
|
||||
documented divergence is **`spf13/cobra`** for the nested subcommand tree
|
||||
(michael is a single-purpose HTTP server and stays stdlib-only; yuctl is a
|
||||
multi-verb CLI). There is no Dockerfile — yuctl is an operator CLI, not a
|
||||
deployed service.
|
||||
`module yuctl`, Go 1.25, `aws-sdk-go-v2` for S3, `rs/zerolog` for logging —
|
||||
matching `packages/michael` — with two documented divergences: **`spf13/cobra`**
|
||||
for the nested subcommand tree (michael is a single-purpose HTTP server and
|
||||
stays stdlib-only; yuctl is a multi-verb CLI), and a **flat package layout**
|
||||
with no `internal/` (yuctl is an unpublishable standalone module nothing else
|
||||
imports, so the boundary bought nothing but a path segment). There is no
|
||||
Dockerfile — yuctl is an operator CLI, not a deployed service.
|
||||
|
||||
The command layer follows the gh/kubectl shape: `cli/` mirrors the command
|
||||
tree one package per topic (`yuctl ceph …` → `cli/ceph`, `yuctl tools warp …`
|
||||
→ `cli/tools/warp`), and every command receives a `cmdutil.Factory` carrying
|
||||
the lazily-resolved shared dependencies (IO streams, selected context,
|
||||
memoized topology, admin-api login) instead of re-deriving them per command.
|
||||
|
||||
## Build
|
||||
|
||||
@@ -32,16 +39,29 @@ Or directly: `cd packages/yuctl && go build -o ../../dist/yuctl .`
|
||||
|
||||
```
|
||||
packages/yuctl/
|
||||
main.go # entrypoint → cli.NewRootCmd().ExecuteContext
|
||||
internal/
|
||||
cli/ # cobra command tree (root, select, ceph, infra, users)
|
||||
discovery/ # S3 state reader + stack enumeration + topology queries
|
||||
state/ # discovery output contract structs + tfstate parsing
|
||||
op/ # `op read` / ReadToTempFile (0600) wrapper
|
||||
context/ # ~/.config/yuctl/context.json {partition,region,ceph_cluster}
|
||||
k8s/ # talosctl upgrade wrapper
|
||||
ceph/ # RGW/dashboard health probe
|
||||
adminapi/ # CLI loopback login + Bearer admin-api client
|
||||
main.go # yuctl entrypoint → cli.NewRootCmd().ExecuteContext
|
||||
cmd/bench-agent/ # second binary: the remote bench agent
|
||||
cli/ # cobra wiring, one package per topic, mirrors the command tree
|
||||
root.go select.go login.go
|
||||
ceph/ infra/ config/ features/
|
||||
users/{allowlist,features,connections}/
|
||||
tools/{bench,fleetbench,warp}/
|
||||
cmdutil/ # Factory (IO, context, topology, admin login) + Confirm/OpenBrowser
|
||||
ui/ # IOStreams, lipgloss theme, meter/sparkline widgets
|
||||
fleet/ # shared fleet-tool engine: watch loop, history, parallel fan-out
|
||||
warp/ # K8s-pod transport: warp runner fleet vs RGW
|
||||
fleetbench/ # cloud-VM transport: restic client fleet vs michael
|
||||
sshx/ # the one ssh/scp layer (multiplexing, retry, stdin secrets)
|
||||
resticbench/ # bench agent + orchestrator + loadgen + restic runner
|
||||
discovery/ # S3 state reader + stack enumeration + topology queries
|
||||
state/ # discovery output contract structs + tfstate parsing
|
||||
op/ # `op read` / ReadToTempFile (0600) wrapper
|
||||
ctxstore/ # ~/.config/yuctl/context.json {partition,region,ceph_cluster}
|
||||
talos/ # talosctl upgrade wrapper
|
||||
cephhealth/ # RGW/dashboard health probe
|
||||
adminapi/ # CLI loopback login + Bearer admin-api client
|
||||
provider/ # cloud-VM providers (DO, Hetzner) for fleet-bench
|
||||
do/ netdev/ # DigitalOcean plumbing; /proc/net/dev parsing
|
||||
```
|
||||
|
||||
## Command tree
|
||||
@@ -69,14 +89,14 @@ yuctl
|
||||
├── bench restic e2e benchmark against michael, run from a mgmt host
|
||||
│ ├── compare <a> <b> render before/after deltas from two results files
|
||||
│ └── cleanup forget+prune every bench snapshot (timed)
|
||||
├── bench-do restic client fleet on DigitalOcean droplets vs michael
|
||||
│ ├── deploy create/converge the droplet fleet (project yucca-bench)
|
||||
├── fleet-bench restic client fleet on cloud VMs (--provider do|hetzner) vs michael
|
||||
│ ├── deploy create/converge the host fleet (project yucca-bench)
|
||||
│ ├── start launch the per-client backup loops (graceful restart)
|
||||
│ ├── status one-shot dashboard (throughput, transfer budget, clients)
|
||||
│ ├── watch live dashboard, continuously sampled
|
||||
│ ├── stop kill the load, collect + save the results JSON
|
||||
│ ├── cleanup forget+prune every bench-do repo (from the droplets)
|
||||
│ └── undeploy destroy the droplets and the ephemeral ssh key
|
||||
│ ├── cleanup forget+prune every fleet-bench repo (from the hosts)
|
||||
│ └── undeploy destroy the hosts and the ephemeral ssh key
|
||||
└── warp S3 load test fleet against the region's RGW gateways
|
||||
├── deploy create/converge hostNetwork runner pods on the workers
|
||||
├── start launch the load (graceful restart; non-stop by default)
|
||||
@@ -100,7 +120,7 @@ Global flags: `--log-level` (trace|debug|info|warn|error), `--log-format`
|
||||
- **bucket fallback**: `ListObjectsV2` on `yucca-tf-state` under prefix
|
||||
`yucca/`, keeping `*/terraform.tfstate` keys.
|
||||
2. **Resolve live values** — `GetObject` each `terraform.tfstate` and parse
|
||||
`.outputs.discovery.value` into `internal/state.Discovery`. Stacks with no
|
||||
`.outputs.discovery.value` into `state.Discovery`. Stacks with no
|
||||
`discovery` output (pre-contract) or no applied state are skipped, not fatal.
|
||||
3. **Query** the merged `Topology` (`HasRegion`, `Kubernetes`, `CephClusters`,
|
||||
`PrimaryRegion`, `RegionMeta`).
|
||||
@@ -240,58 +260,62 @@ The ssh session stays open for the whole run (keepalives set); run multi-hour
|
||||
benchmarks inside tmux. Pair the client numbers with the michael dashboard
|
||||
(TTFB, connection churn, S3 client metrics) for the server-side view.
|
||||
|
||||
## `tools bench-do` — DigitalOcean restic client fleet
|
||||
## `tools fleet-bench` — cloud-VM restic client fleet
|
||||
|
||||
Drives michael from the outside: N DigitalOcean droplets each running real
|
||||
restic clients over the public internet — the actual external-user path (DNS,
|
||||
edge, michael, RGW), unlike `bench` (mgmt host on the fabric) and `warp`
|
||||
(in-cluster, straight at RGW). Fleet lifecycle mirrors `warp`.
|
||||
Drives michael from the outside: N cloud VMs (`--provider do|hetzner`) each
|
||||
running real restic clients over the public internet — the actual
|
||||
external-user path (DNS, edge, michael, RGW), unlike `bench` (mgmt host on
|
||||
the fabric) and `warp` (in-cluster, straight at RGW). Fleet lifecycle mirrors
|
||||
`warp`; fleets are per provider × partition, so several providers can load
|
||||
michael at once.
|
||||
|
||||
```bash
|
||||
yuctl select prod@htz-fsn1
|
||||
yuctl login
|
||||
yuctl tools bench-do start --droplets 6 --clients-per-droplet 2 \
|
||||
yuctl tools fleet-bench start --hosts 6 --clients-per-host 2 \
|
||||
--obj-size 64MiB --duration 2h --label big-packs # auto-deploys (confirms cost first)
|
||||
yuctl tools bench-do watch # live dashboard: Gbps, transfer budget bars, client loops
|
||||
yuctl tools bench-do stop # kill the load, save bench-do-<label>-<ts>.json
|
||||
yuctl tools bench-do cleanup # forget+prune the bench repos (while droplets exist)
|
||||
yuctl tools bench-do undeploy # destroy droplets + the ephemeral ssh key
|
||||
yuctl tools fleet-bench watch # live dashboard: Gbps, transfer budget bars, client loops
|
||||
yuctl tools fleet-bench stop # kill the load, save fleet-bench-<label>-<ts>.json
|
||||
yuctl tools fleet-bench cleanup # forget+prune the bench repos (while the hosts exist)
|
||||
yuctl tools fleet-bench undeploy # destroy the hosts + the ephemeral ssh key
|
||||
```
|
||||
|
||||
How it works:
|
||||
|
||||
- **Fleet**: droplets (`--droplets`, default 3 × `s-2vcpu-4gb`) are created via
|
||||
the DO API (token from `op://yucca/do_api_token/password`, or
|
||||
`$DIGITALOCEAN_TOKEN`), round-robined across `--do-region`
|
||||
(default `fra1,ams3,lon1,nyc3`), tagged `yuctl-bench-do-<partition>`, and
|
||||
filed under the **`yucca-bench`** project (created if missing). Deploy
|
||||
prints the hourly cost and transfer pool and asks before creating anything
|
||||
(`--yes` skips); it converges — rerunning reconciles the fleet to the
|
||||
requested size and re-pushes binaries.
|
||||
- **Fleet**: hosts (`--hosts`, default 3 of the provider's default size) are
|
||||
created via the provider API (DO token from
|
||||
`op://yucca/do_api_token/password` / `$DIGITALOCEAN_TOKEN`; Hetzner from
|
||||
`op://yucca_tf_prod/HCLOUD_API_TOKEN/password` / `$HCLOUD_TOKEN`),
|
||||
round-robined across `--region`, tagged `yuctl-bench-<provider>-<partition>`,
|
||||
and filed under the **`yucca-bench`** project where the provider has the
|
||||
concept. Deploy prints the hourly cost and transfer pool and asks before
|
||||
creating anything (`--yes` skips); it converges — rerunning reconciles the
|
||||
fleet to the requested size and re-pushes binaries.
|
||||
- **SSH**: an **ephemeral ed25519 keypair per fleet** — generated on deploy,
|
||||
registered via the API, private key + per-fleet known_hosts under
|
||||
`~/.config/yuctl/bench-do/`, deleted on undeploy. No personal keys involved.
|
||||
- **Clients**: one admin-api repository per client (`--clients-per-droplet`),
|
||||
named `yucca-benchdo-…`, created on first start and reused across restarts;
|
||||
restic URLs are re-minted on every start and travel to the droplet over ssh
|
||||
stdin (never argv). Repo passwords live in the 0600 fleet state file — they
|
||||
are the only way back into the repos, so `cleanup` before `undeploy`.
|
||||
`~/.config/yuctl/bench-wide/` (legacy dir name), deleted on undeploy. No
|
||||
personal keys involved.
|
||||
- **Clients**: one admin-api repository per client (`--clients-per-host`),
|
||||
named `yucca-benchdo-…` (legacy prefix), created on first start and reused
|
||||
across restarts; restic URLs are re-minted on every start and travel to the
|
||||
host over ssh stdin (never argv). Repo passwords live in the 0600 fleet
|
||||
state file — they are the only way back into the repos, so `cleanup` before
|
||||
`undeploy`.
|
||||
- **Load**: the bench agent's **loadgen mode** runs detached (nohup) on each
|
||||
droplet, looping seeded generate→backup cycles per client — fresh seed every
|
||||
host, looping seeded generate→backup cycles per client — fresh seed every
|
||||
cycle so nothing dedups — with `--obj-size` as the restic pack size
|
||||
(4–128 MiB, what michael sees as object size), `--size` per-cycle dataset,
|
||||
`--connections` rest.connections. `--duration` bounds the run (`0` =
|
||||
non-stop until `stop`). Progress goes to a droplet-local status file that
|
||||
`status`/`watch` sample over ssh alongside `/proc/net/dev`.
|
||||
- **Transfer cap (important)**: DO droplets have a monthly outbound transfer
|
||||
allowance (pooled; overage is billed per GiB) and a sustained restic load
|
||||
can burn through it in hours. The agent tracks wire TX and **hard-stops the
|
||||
droplet's load at the allowance** (counted conservatively as decimal TB);
|
||||
`--max-transfer` overrides. The dashboard shows a per-droplet budget bar and
|
||||
the fleet pool. If the fleet state is lost the cap is re-derived from the
|
||||
droplet size — the load never runs uncapped.
|
||||
- **Results**: `stop` collects each droplet's final status and writes a local
|
||||
JSON (per-client cycles, post-dedup uploaded bytes, errors; per-droplet wire
|
||||
(4–128 MiB, what michael sees as object size), `--cycle-size` per-cycle
|
||||
dataset, `--connections` rest.connections. `--duration` bounds the run
|
||||
(`0` = non-stop until `stop`). Progress goes to a host-local status file
|
||||
that `status`/`watch` sample over ssh alongside `/proc/net/dev`.
|
||||
- **Transfer cap (important)**: cloud VMs have a monthly outbound transfer
|
||||
allowance (overage is billed per GiB) and a sustained restic load can burn
|
||||
through it in hours. The agent tracks wire TX and **hard-stops the host's
|
||||
load at the allowance**; `--max-transfer` overrides. The dashboard shows a
|
||||
per-host budget bar and the fleet pool. If the fleet state is lost the cap
|
||||
is re-derived from the host size — the load never runs uncapped.
|
||||
- **Results**: `stop` collects each host's final status and writes a local
|
||||
JSON (per-client cycles, post-dedup uploaded bytes, errors; per-host wire
|
||||
TX) plus a rendered summary. Pair with the michael dashboards for the
|
||||
server-side view.
|
||||
|
||||
@@ -329,7 +353,7 @@ How it works:
|
||||
- **Runners**: `minio/warp` pods (2 per worker by default) on **hostNetwork**,
|
||||
so the load rides the workers' bonded NICs with no CNI hop. CPU request is
|
||||
derived from node allocatable. Manifests are `go:embed`ded templates
|
||||
(`internal/warp/manifests/`), server-side-applied via client-go — no
|
||||
(`fleet/warp/manifests/`), server-side-applied via client-go — no
|
||||
kubectl dependency. Credentials are copied from the `yucca-michael` secret
|
||||
(the same RGW svc user as the real data path).
|
||||
- **Gateway roster**: the ceph cluster's `rgw_s3_endpoint` is resolved to its
|
||||
@@ -362,12 +386,14 @@ about the fleet size, gateway count, or endpoints is hardcoded.
|
||||
| `OP_BIN` | 1Password CLI binary | `op` |
|
||||
| `YUCTL_ADMIN_API_URL` | admin-api base URL (`login`, `users`) | derived from discovery `api_endpoint` |
|
||||
| `YUCTL_GRAFANA_URL` | grafana base (`users view-dashboard`) | `https://grafana.futostatus.com` |
|
||||
| `DIGITALOCEAN_TOKEN` | DO API token (`tools bench-do`) | resolved via op |
|
||||
| `DIGITALOCEAN_TOKEN` | DO API token (`tools fleet-bench`) | resolved via op |
|
||||
| `YUCTL_DO_TOKEN_REF` | op ref for the DO token | `op://yucca/do_api_token/password` |
|
||||
| `HCLOUD_TOKEN` | Hetzner API token (`tools fleet-bench`)| resolved via op |
|
||||
| `YUCTL_HCLOUD_TOKEN_REF` | op ref for the Hetzner token | `op://yucca_tf_prod/HCLOUD_API_TOKEN/password` |
|
||||
|
||||
## Tests
|
||||
|
||||
`go test ./...` covers the load-bearing offline logic: discovery contract
|
||||
parsing (`internal/state`) and stack-key/topology queries
|
||||
(`internal/discovery`). The network/`op`/`talosctl`/admin-api paths are not unit
|
||||
parsing (`state`) and stack-key/topology queries
|
||||
(`discovery`). The network/`op`/`talosctl`/admin-api paths are not unit
|
||||
tested (they need live infra).
|
||||
|
||||
@@ -6,14 +6,14 @@ import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
|
||||
yctx "yuctl/internal/context"
|
||||
"yuctl/ctxstore"
|
||||
)
|
||||
|
||||
// tokenCachePath returns the per-partition token cache file inside the yuctl
|
||||
// config dir. Partition-scoped because each partition has its own primary
|
||||
// region / admin-api.
|
||||
func tokenCachePath(partition string) (string, error) {
|
||||
dir, err := yctx.Dir()
|
||||
dir, err := ctxstore.Dir()
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
@@ -43,7 +43,7 @@ func LoadToken(partition string) (Token, error) {
|
||||
|
||||
// SaveToken persists a CLI session token at 0600.
|
||||
func SaveToken(partition string, t Token) error {
|
||||
dir, err := yctx.Dir()
|
||||
dir, err := ctxstore.Dir()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
@@ -95,6 +95,21 @@ func (c *Client) ListUsers(ctx context.Context, limit int) ([]User, error) {
|
||||
return all, nil
|
||||
}
|
||||
|
||||
// ResolveUserID turns an email into a user id. The admin-api has no lookup
|
||||
// endpoint, so this lists every user and matches case-insensitively.
|
||||
func (c *Client) ResolveUserID(ctx context.Context, email string) (string, error) {
|
||||
users, err := c.ListUsers(ctx, 0)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
for _, u := range users {
|
||||
if strings.EqualFold(u.Email, email) {
|
||||
return u.ID, nil
|
||||
}
|
||||
}
|
||||
return "", fmt.Errorf("no user with email %q", email)
|
||||
}
|
||||
|
||||
func (c *Client) listUserPage(ctx context.Context, cursor string, limit int) (*userPage, error) {
|
||||
q := url.Values{}
|
||||
if cursor != "" {
|
||||
@@ -1,7 +1,7 @@
|
||||
// Package ceph implements health checks against a region's Ceph cluster RGW /
|
||||
// Package cephhealth implements health checks against a region's Ceph cluster RGW /
|
||||
// dashboard endpoint, using the `rgw_s3_endpoint` + `health_cred_ref` fields of
|
||||
// the ceph discovery payload (credential resolved via 1Password).
|
||||
package ceph
|
||||
package cephhealth
|
||||
|
||||
import (
|
||||
"context"
|
||||
@@ -12,8 +12,8 @@ import (
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"yuctl/internal/op"
|
||||
"yuctl/internal/state"
|
||||
"yuctl/op"
|
||||
"yuctl/state"
|
||||
)
|
||||
|
||||
// HealthResult summarizes a probe.
|
||||
@@ -1,92 +1,90 @@
|
||||
package cli
|
||||
package ceph
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"sort"
|
||||
"strings"
|
||||
|
||||
"github.com/spf13/cobra"
|
||||
|
||||
"yuctl/internal/ceph"
|
||||
yctx "yuctl/internal/context"
|
||||
"yuctl/cephhealth"
|
||||
"yuctl/cmdutil"
|
||||
"yuctl/ctxstore"
|
||||
"yuctl/ui"
|
||||
)
|
||||
|
||||
func newCephCmd() *cobra.Command {
|
||||
func New(f *cmdutil.Factory) *cobra.Command {
|
||||
cmd := &cobra.Command{
|
||||
Use: "ceph",
|
||||
Short: "Ceph cluster operations within the selected region",
|
||||
}
|
||||
cmd.AddCommand(newCephSelectCmd(), newCephGetCmd())
|
||||
cmd.AddCommand(newSelectCmd(f), newGetCmd(f))
|
||||
return cmd
|
||||
}
|
||||
|
||||
func newCephSelectCmd() *cobra.Command {
|
||||
func newSelectCmd(f *cmdutil.Factory) *cobra.Command {
|
||||
return &cobra.Command{
|
||||
Use: "select <name>",
|
||||
Short: "Select a Ceph cluster within the active region",
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
ctx := cmd.Context()
|
||||
name := args[0]
|
||||
|
||||
c, err := requireContext()
|
||||
c, err := f.Context()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
topo, err := resolveTopology(ctx)
|
||||
topo, err := f.Topology(cmd.Context())
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
clusters := topo.CephClusters(c.Partition, c.Region)
|
||||
if _, ok := clusters[name]; !ok {
|
||||
if _, ok := topo.CephClusters(c.Partition, c.Region)[name]; !ok {
|
||||
return fmt.Errorf("unknown ceph cluster %q in %s@%s; known: %s",
|
||||
name, c.Partition, c.Region, strings.Join(cephClusterNames(clusters), ", "))
|
||||
name, c.Partition, c.Region, strings.Join(topo.CephClusterNames(c.Partition, c.Region), ", "))
|
||||
}
|
||||
|
||||
c.CephCluster = name
|
||||
if err := yctx.Save(c); err != nil {
|
||||
if err := ctxstore.Save(c); err != nil {
|
||||
return err
|
||||
}
|
||||
fmt.Fprintf(cmd.OutOrStdout(), "selected ceph cluster %q in %s@%s\n", name, c.Partition, c.Region)
|
||||
fmt.Fprintf(f.IO.Out, "selected ceph cluster %q in %s@%s\n", name, c.Partition, c.Region)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
func newCephGetCmd() *cobra.Command {
|
||||
func newGetCmd(f *cmdutil.Factory) *cobra.Command {
|
||||
cmd := &cobra.Command{
|
||||
Use: "get",
|
||||
Short: "Read Ceph cluster state",
|
||||
}
|
||||
cmd.AddCommand(newCephGetHealthCmd())
|
||||
cmd.AddCommand(newGetHealthCmd(f))
|
||||
return cmd
|
||||
}
|
||||
|
||||
func newCephGetHealthCmd() *cobra.Command {
|
||||
func newGetHealthCmd(f *cmdutil.Factory) *cobra.Command {
|
||||
var insecure bool
|
||||
c := &cobra.Command{
|
||||
Use: "health",
|
||||
Short: "Probe the selected Ceph cluster's RGW/dashboard health",
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
ctx := cmd.Context()
|
||||
cc, err := requireContext()
|
||||
cc, err := f.Context()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if cc.CephCluster == "" {
|
||||
return fmt.Errorf("no ceph cluster selected; run `yuctl ceph select <name>` first")
|
||||
}
|
||||
topo, err := resolveTopology(ctx)
|
||||
topo, err := f.Topology(ctx)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
clusters := topo.CephClusters(cc.Partition, cc.Region)
|
||||
cluster, ok := clusters[cc.CephCluster]
|
||||
cluster, ok := topo.CephClusters(cc.Partition, cc.Region)[cc.CephCluster]
|
||||
if !ok {
|
||||
return fmt.Errorf("ceph cluster %q no longer present in %s@%s", cc.CephCluster, cc.Partition, cc.Region)
|
||||
}
|
||||
|
||||
res, err := ceph.CheckHealth(ctx, cluster, insecure)
|
||||
res, err := cephhealth.CheckHealth(ctx, cluster, insecure)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
@@ -94,12 +92,12 @@ func newCephGetHealthCmd() *cobra.Command {
|
||||
if !res.Healthy {
|
||||
status = "UNHEALTHY"
|
||||
}
|
||||
out := cmd.OutOrStdout()
|
||||
out := f.IO.Out
|
||||
fmt.Fprintf(out, "ceph cluster: %s (%s@%s)\n", cc.CephCluster, cc.Partition, cc.Region)
|
||||
fmt.Fprintf(out, "endpoint: %s\n", res.Endpoint)
|
||||
fmt.Fprintf(out, "status: %s (http %d)\n", status, res.StatusCode)
|
||||
if res.Detail != "" {
|
||||
fmt.Fprintf(out, "detail: %s\n", truncate(res.Detail, 200))
|
||||
fmt.Fprintf(out, "detail: %s\n", ui.Truncate(res.Detail, 200))
|
||||
}
|
||||
if !res.Healthy {
|
||||
return fmt.Errorf("ceph health check failed")
|
||||
@@ -110,20 +108,3 @@ func newCephGetHealthCmd() *cobra.Command {
|
||||
c.Flags().BoolVar(&insecure, "insecure-skip-tls-verify", false, "skip TLS verification (self-signed RGW/dashboard certs)")
|
||||
return c
|
||||
}
|
||||
|
||||
func truncate(s string, n int) string {
|
||||
if len(s) <= n {
|
||||
return s
|
||||
}
|
||||
return s[:n] + "…"
|
||||
}
|
||||
|
||||
// sortedKeys of the ceph cluster map.
|
||||
func cephClusterNames[T any](m map[string]T) []string {
|
||||
out := make([]string, 0, len(m))
|
||||
for k := range m {
|
||||
out = append(out, k)
|
||||
}
|
||||
sort.Strings(out)
|
||||
return out
|
||||
}
|
||||
@@ -1,6 +1,11 @@
|
||||
package cli
|
||||
// Package config implements `yuctl config`: scoped mutable configuration
|
||||
// (global / per-site / per-cluster overrides of restic_pack_size_mib,
|
||||
// connections_math, ...) stored by yucca-admin-api and served to clients via
|
||||
// GET /api/meta.
|
||||
package config
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"maps"
|
||||
@@ -10,22 +15,16 @@ import (
|
||||
|
||||
"github.com/spf13/cobra"
|
||||
|
||||
"yuctl/internal/adminapi"
|
||||
"yuctl/adminapi"
|
||||
"yuctl/cmdutil"
|
||||
)
|
||||
|
||||
// newConfigCmd builds the `config` subtree: scoped mutable configuration
|
||||
// (global / per-site / per-cluster overrides of restic_pack_size_mib,
|
||||
// connections_math, ...) stored by yucca-admin-api and served to clients via
|
||||
// GET /api/meta.
|
||||
func newConfigCmd() *cobra.Command {
|
||||
func New(f *cmdutil.Factory) *cobra.Command {
|
||||
cmd := &cobra.Command{
|
||||
Use: "config",
|
||||
Short: "Manage scoped config overrides (global / site / cluster) via yucca-admin-api",
|
||||
}
|
||||
cmd.AddCommand(newConfigListCmd())
|
||||
cmd.AddCommand(newConfigGetCmd())
|
||||
cmd.AddCommand(newConfigSetCmd())
|
||||
cmd.AddCommand(newConfigUnsetCmd())
|
||||
cmd.AddCommand(newListCmd(f), newGetCmd(f), newSetCmd(f), newUnsetCmd(f))
|
||||
return cmd
|
||||
}
|
||||
|
||||
@@ -52,14 +51,14 @@ func (s *scopeFlags) scope() string {
|
||||
}
|
||||
}
|
||||
|
||||
func newConfigListCmd() *cobra.Command {
|
||||
flags := &adminFlags{}
|
||||
func newListCmd(f *cmdutil.Factory) *cobra.Command {
|
||||
admin := &cmdutil.AdminFlags{}
|
||||
c := &cobra.Command{
|
||||
Use: "list",
|
||||
Short: "List every settings scope and its overrides",
|
||||
Args: cobra.NoArgs,
|
||||
RunE: func(cmd *cobra.Command, _ []string) error {
|
||||
client, partition, err := flags.allowlistClient(cmd)
|
||||
client, partition, err := admin.Client(cmd.Context(), f)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
@@ -68,7 +67,7 @@ func newConfigListCmd() *cobra.Command {
|
||||
return err
|
||||
}
|
||||
|
||||
w := tabwriter.NewWriter(cmd.OutOrStdout(), 0, 2, 2, ' ', 0)
|
||||
w := tabwriter.NewWriter(f.IO.Out, 0, 2, 2, ' ', 0)
|
||||
fmt.Fprintln(w, "SCOPE\tVALUE\tUPDATED")
|
||||
for _, e := range entries {
|
||||
value, err := json.Marshal(e.Value)
|
||||
@@ -78,52 +77,45 @@ func newConfigListCmd() *cobra.Command {
|
||||
fmt.Fprintf(w, "%s\t%s\t%s\n", e.Scope, value, e.UpdatedAt)
|
||||
}
|
||||
w.Flush()
|
||||
fmt.Fprintf(cmd.ErrOrStderr(), "\n%d scope(s) in partition %s\n", len(entries), partition)
|
||||
fmt.Fprintf(f.IO.Err, "\n%d scope(s) in partition %s\n", len(entries), partition)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
flags.register(c)
|
||||
admin.Register(c)
|
||||
return c
|
||||
}
|
||||
|
||||
func newConfigGetCmd() *cobra.Command {
|
||||
flags := &adminFlags{}
|
||||
func newGetCmd(f *cmdutil.Factory) *cobra.Command {
|
||||
admin := &cmdutil.AdminFlags{}
|
||||
scope := &scopeFlags{}
|
||||
c := &cobra.Command{
|
||||
Use: "get",
|
||||
Short: "Print the overrides stored for a scope",
|
||||
Args: cobra.NoArgs,
|
||||
RunE: func(cmd *cobra.Command, _ []string) error {
|
||||
client, _, err := flags.allowlistClient(cmd)
|
||||
client, _, err := admin.Client(cmd.Context(), f)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
entries, err := client.ListSettings(cmd.Context())
|
||||
value, err := currentValue(cmd.Context(), client, scope.scope())
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
value := map[string]any{}
|
||||
for _, e := range entries {
|
||||
if e.Scope == scope.scope() {
|
||||
value = e.Value
|
||||
break
|
||||
}
|
||||
}
|
||||
out, err := json.MarshalIndent(value, "", " ")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
fmt.Fprintln(cmd.OutOrStdout(), string(out))
|
||||
fmt.Fprintln(f.IO.Out, string(out))
|
||||
return nil
|
||||
},
|
||||
}
|
||||
scope.register(c)
|
||||
flags.register(c)
|
||||
admin.Register(c)
|
||||
return c
|
||||
}
|
||||
|
||||
func newConfigSetCmd() *cobra.Command {
|
||||
flags := &adminFlags{}
|
||||
func newSetCmd(f *cmdutil.Factory) *cobra.Command {
|
||||
admin := &cmdutil.AdminFlags{}
|
||||
scope := &scopeFlags{}
|
||||
c := &cobra.Command{
|
||||
Use: "set key=value [key=value ...]",
|
||||
@@ -144,13 +136,13 @@ func newConfigSetCmd() *cobra.Command {
|
||||
updates[key] = parseValue(raw)
|
||||
}
|
||||
|
||||
client, _, err := flags.allowlistClient(cmd)
|
||||
ctx := cmd.Context()
|
||||
client, _, err := admin.Client(ctx, f)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
ctx := cmd.Context()
|
||||
|
||||
value, err := currentValue(cmd, client, scope.scope())
|
||||
value, err := currentValue(ctx, client, scope.scope())
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
@@ -164,17 +156,17 @@ func newConfigSetCmd() *cobra.Command {
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
fmt.Fprintf(cmd.OutOrStdout(), "%s = %s\n", entry.Scope, out)
|
||||
fmt.Fprintf(f.IO.Out, "%s = %s\n", entry.Scope, out)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
scope.register(c)
|
||||
flags.register(c)
|
||||
admin.Register(c)
|
||||
return c
|
||||
}
|
||||
|
||||
func newConfigUnsetCmd() *cobra.Command {
|
||||
flags := &adminFlags{}
|
||||
func newUnsetCmd(f *cmdutil.Factory) *cobra.Command {
|
||||
admin := &cmdutil.AdminFlags{}
|
||||
scope := &scopeFlags{}
|
||||
var all bool
|
||||
c := &cobra.Command{
|
||||
@@ -185,21 +177,21 @@ func newConfigUnsetCmd() *cobra.Command {
|
||||
return fmt.Errorf("pass either keys to unset or --all")
|
||||
}
|
||||
|
||||
client, _, err := flags.allowlistClient(cmd)
|
||||
ctx := cmd.Context()
|
||||
client, _, err := admin.Client(ctx, f)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
ctx := cmd.Context()
|
||||
|
||||
if all {
|
||||
if err := client.DeleteSettings(ctx, scope.scope()); err != nil {
|
||||
return err
|
||||
}
|
||||
fmt.Fprintf(cmd.OutOrStdout(), "%s cleared\n", scope.scope())
|
||||
fmt.Fprintf(f.IO.Out, "%s cleared\n", scope.scope())
|
||||
return nil
|
||||
}
|
||||
|
||||
value, err := currentValue(cmd, client, scope.scope())
|
||||
value, err := currentValue(ctx, client, scope.scope())
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
@@ -211,7 +203,7 @@ func newConfigUnsetCmd() *cobra.Command {
|
||||
if err := client.DeleteSettings(ctx, scope.scope()); err != nil {
|
||||
return err
|
||||
}
|
||||
fmt.Fprintf(cmd.OutOrStdout(), "%s cleared\n", scope.scope())
|
||||
fmt.Fprintf(f.IO.Out, "%s cleared\n", scope.scope())
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -223,18 +215,18 @@ func newConfigUnsetCmd() *cobra.Command {
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
fmt.Fprintf(cmd.OutOrStdout(), "%s = %s\n", entry.Scope, out)
|
||||
fmt.Fprintf(f.IO.Out, "%s = %s\n", entry.Scope, out)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
c.Flags().BoolVar(&all, "all", false, "delete the whole scope instead of individual keys")
|
||||
scope.register(c)
|
||||
flags.register(c)
|
||||
admin.Register(c)
|
||||
return c
|
||||
}
|
||||
|
||||
func currentValue(cmd *cobra.Command, client *adminapi.Client, scope string) (map[string]any, error) {
|
||||
entries, err := client.ListSettings(cmd.Context())
|
||||
func currentValue(ctx context.Context, client *adminapi.Client, scope string) (map[string]any, error) {
|
||||
entries, err := client.ListSettings(ctx)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
@@ -1,4 +1,6 @@
|
||||
package cli
|
||||
// Package features implements `yuctl features`: the feature-flag registry and
|
||||
// fleet-wide operations (per-user set/clear lives under `users features`).
|
||||
package features
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
@@ -7,24 +9,21 @@ import (
|
||||
|
||||
"github.com/spf13/cobra"
|
||||
|
||||
"yuctl/internal/adminapi"
|
||||
"yuctl/adminapi"
|
||||
"yuctl/cmdutil"
|
||||
)
|
||||
|
||||
// newFeaturesCmd builds the `features` subtree: the feature-flag registry and
|
||||
// fleet-wide operations (per-user set/clear lives under `users features`).
|
||||
func newFeaturesCmd() *cobra.Command {
|
||||
func New(f *cmdutil.Factory) *cobra.Command {
|
||||
cmd := &cobra.Command{
|
||||
Use: "features",
|
||||
Short: "Feature-flag registry and batch enrollment",
|
||||
}
|
||||
cmd.AddCommand(newFeaturesListCmd())
|
||||
cmd.AddCommand(newFeaturesUsersCmd())
|
||||
cmd.AddCommand(newFeaturesEnableBatchCmd())
|
||||
cmd.AddCommand(newListCmd(f), newUsersCmd(f), newEnableBatchCmd(f))
|
||||
return cmd
|
||||
}
|
||||
|
||||
func printFeatureUsers(cmd *cobra.Command, items []adminapi.FeatureUser) {
|
||||
w := tabwriter.NewWriter(cmd.OutOrStdout(), 0, 2, 2, ' ', 0)
|
||||
func printFeatureUsers(f *cmdutil.Factory, items []adminapi.FeatureUser) {
|
||||
w := tabwriter.NewWriter(f.IO.Out, 0, 2, 2, ' ', 0)
|
||||
fmt.Fprintln(w, "EMAIL\tVALUE\tSET BY\tREASON\tUPDATED")
|
||||
for _, u := range items {
|
||||
reason := ""
|
||||
@@ -36,14 +35,14 @@ func printFeatureUsers(cmd *cobra.Command, items []adminapi.FeatureUser) {
|
||||
w.Flush()
|
||||
}
|
||||
|
||||
func newFeaturesListCmd() *cobra.Command {
|
||||
flags := &adminFlags{}
|
||||
func newListCmd(f *cmdutil.Factory) *cobra.Command {
|
||||
admin := &cmdutil.AdminFlags{}
|
||||
c := &cobra.Command{
|
||||
Use: "list",
|
||||
Short: "List the feature-flag registry with override counts",
|
||||
Args: cobra.NoArgs,
|
||||
RunE: func(cmd *cobra.Command, _ []string) error {
|
||||
client, _, err := flags.allowlistClient(cmd)
|
||||
client, _, err := admin.Client(cmd.Context(), f)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
@@ -52,27 +51,27 @@ func newFeaturesListCmd() *cobra.Command {
|
||||
return err
|
||||
}
|
||||
|
||||
w := tabwriter.NewWriter(cmd.OutOrStdout(), 0, 2, 2, ' ', 0)
|
||||
w := tabwriter.NewWriter(f.IO.Out, 0, 2, 2, ' ', 0)
|
||||
fmt.Fprintln(w, "FLAG\tDEFAULT\tSTAGE\tOVERRIDES\tSINCE\tDESCRIPTION")
|
||||
for _, f := range featureFlags {
|
||||
fmt.Fprintf(w, "%s\t%t\t%s\t%d\t%s\t%s\n", f.Key, f.Default, f.Stage, f.Overrides, f.Since, f.Description)
|
||||
for _, ff := range featureFlags {
|
||||
fmt.Fprintf(w, "%s\t%t\t%s\t%d\t%s\t%s\n", ff.Key, ff.Default, ff.Stage, ff.Overrides, ff.Since, ff.Description)
|
||||
}
|
||||
w.Flush()
|
||||
return nil
|
||||
},
|
||||
}
|
||||
flags.register(c)
|
||||
admin.Register(c)
|
||||
return c
|
||||
}
|
||||
|
||||
func newFeaturesUsersCmd() *cobra.Command {
|
||||
flags := &adminFlags{}
|
||||
func newUsersCmd(f *cmdutil.Factory) *cobra.Command {
|
||||
admin := &cmdutil.AdminFlags{}
|
||||
c := &cobra.Command{
|
||||
Use: "users <flag>",
|
||||
Short: "List users holding an override for a flag",
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
client, _, err := flags.allowlistClient(cmd)
|
||||
client, _, err := admin.Client(cmd.Context(), f)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
@@ -80,17 +79,17 @@ func newFeaturesUsersCmd() *cobra.Command {
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
printFeatureUsers(cmd, items)
|
||||
fmt.Fprintf(cmd.ErrOrStderr(), "\n%d override(s) for %s\n", len(items), args[0])
|
||||
printFeatureUsers(f, items)
|
||||
fmt.Fprintf(f.IO.Err, "\n%d override(s) for %s\n", len(items), args[0])
|
||||
return nil
|
||||
},
|
||||
}
|
||||
flags.register(c)
|
||||
admin.Register(c)
|
||||
return c
|
||||
}
|
||||
|
||||
func newFeaturesEnableBatchCmd() *cobra.Command {
|
||||
flags := &adminFlags{}
|
||||
func newEnableBatchCmd(f *cmdutil.Factory) *cobra.Command {
|
||||
admin := &cmdutil.AdminFlags{}
|
||||
c := &cobra.Command{
|
||||
Use: "enable-batch <flag> <count>",
|
||||
Short: "Enable a flag for the oldest <count> users without an override",
|
||||
@@ -100,7 +99,7 @@ func newFeaturesEnableBatchCmd() *cobra.Command {
|
||||
if err != nil || count < 1 {
|
||||
return fmt.Errorf("count must be a positive integer")
|
||||
}
|
||||
client, _, err := flags.allowlistClient(cmd)
|
||||
client, _, err := admin.Client(cmd.Context(), f)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
@@ -108,11 +107,11 @@ func newFeaturesEnableBatchCmd() *cobra.Command {
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
printFeatureUsers(cmd, enabled)
|
||||
fmt.Fprintf(cmd.ErrOrStderr(), "\nenabled %s for %d user(s)\n", args[0], len(enabled))
|
||||
printFeatureUsers(f, enabled)
|
||||
fmt.Fprintf(f.IO.Err, "\nenabled %s for %d user(s)\n", args[0], len(enabled))
|
||||
return nil
|
||||
},
|
||||
}
|
||||
flags.register(c)
|
||||
admin.Register(c)
|
||||
return c
|
||||
}
|
||||
@@ -1,36 +1,35 @@
|
||||
package cli
|
||||
package infra
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"fmt"
|
||||
"os"
|
||||
"strings"
|
||||
|
||||
"github.com/rs/zerolog/log"
|
||||
"github.com/spf13/cobra"
|
||||
|
||||
"yuctl/internal/k8s"
|
||||
"yuctl/cmdutil"
|
||||
"yuctl/talos"
|
||||
)
|
||||
|
||||
func newInfraCmd() *cobra.Command {
|
||||
func New(f *cmdutil.Factory) *cobra.Command {
|
||||
cmd := &cobra.Command{
|
||||
Use: "infra",
|
||||
Short: "Infrastructure / node operations for the active region",
|
||||
}
|
||||
cmd.AddCommand(newInfraTalosCmd())
|
||||
cmd.AddCommand(newTalosCmd(f))
|
||||
return cmd
|
||||
}
|
||||
|
||||
func newInfraTalosCmd() *cobra.Command {
|
||||
func newTalosCmd(f *cmdutil.Factory) *cobra.Command {
|
||||
cmd := &cobra.Command{
|
||||
Use: "talos",
|
||||
Short: "Talos node operations",
|
||||
}
|
||||
cmd.AddCommand(newInfraTalosUpgradeCmd())
|
||||
cmd.AddCommand(newTalosUpgradeCmd(f))
|
||||
return cmd
|
||||
}
|
||||
|
||||
func newInfraTalosUpgradeCmd() *cobra.Command {
|
||||
func newTalosUpgradeCmd(f *cmdutil.Factory) *cobra.Command {
|
||||
var (
|
||||
image string
|
||||
dryRun bool
|
||||
@@ -44,11 +43,11 @@ func newInfraTalosUpgradeCmd() *cobra.Command {
|
||||
"prompt unless --yes; use --dry-run to print the commands without running.",
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
ctx := cmd.Context()
|
||||
cc, err := requireContext()
|
||||
cc, err := f.Context()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
topo, err := resolveTopology(ctx)
|
||||
topo, err := f.Topology(ctx)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
@@ -57,7 +56,7 @@ func newInfraTalosUpgradeCmd() *cobra.Command {
|
||||
return fmt.Errorf("no kubernetes/talos discovery payload for %s@%s", cc.Partition, cc.Region)
|
||||
}
|
||||
|
||||
out := cmd.OutOrStdout()
|
||||
out := f.IO.Out
|
||||
fmt.Fprintf(out, "Talos upgrade target: %s@%s (cluster %s)\n", cc.Partition, cc.Region, kube.ClusterName)
|
||||
fmt.Fprintf(out, "control-plane nodes: %s\n", strings.Join(kube.CPNodeIPs, ", "))
|
||||
if image != "" {
|
||||
@@ -65,13 +64,13 @@ func newInfraTalosUpgradeCmd() *cobra.Command {
|
||||
}
|
||||
|
||||
if !dryRun && !yes {
|
||||
if !confirm(fmt.Sprintf("Upgrade Talos on %d node(s) in %s@%s?", len(kube.CPNodeIPs), cc.Partition, cc.Region)) {
|
||||
if !cmdutil.Confirm(f.IO, fmt.Sprintf("Upgrade Talos on %d node(s) in %s@%s?", len(kube.CPNodeIPs), cc.Partition, cc.Region)) {
|
||||
fmt.Fprintln(out, "aborted")
|
||||
return nil
|
||||
}
|
||||
}
|
||||
|
||||
return k8s.TalosUpgrade(ctx, *kube, k8s.UpgradeOptions{
|
||||
return talos.Upgrade(ctx, *kube, talos.UpgradeOptions{
|
||||
Image: image,
|
||||
DryRun: dryRun,
|
||||
}, log.Logger)
|
||||
@@ -82,15 +81,3 @@ func newInfraTalosUpgradeCmd() *cobra.Command {
|
||||
c.Flags().BoolVar(&yes, "yes", false, "skip the confirmation prompt")
|
||||
return c
|
||||
}
|
||||
|
||||
// confirm prompts the operator for a yes/no answer on stdin, defaulting to no.
|
||||
func confirm(prompt string) bool {
|
||||
fmt.Fprintf(os.Stderr, "%s [y/N]: ", prompt)
|
||||
reader := bufio.NewReader(os.Stdin)
|
||||
line, err := reader.ReadString('\n')
|
||||
if err != nil {
|
||||
return false
|
||||
}
|
||||
answer := strings.ToLower(strings.TrimSpace(line))
|
||||
return answer == "y" || answer == "yes"
|
||||
}
|
||||
@@ -0,0 +1,51 @@
|
||||
package cli
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"time"
|
||||
|
||||
"github.com/spf13/cobra"
|
||||
|
||||
"yuctl/cmdutil"
|
||||
)
|
||||
|
||||
func newLoginCmd(f *cmdutil.Factory) *cobra.Command {
|
||||
admin := &cmdutil.AdminFlags{}
|
||||
c := &cobra.Command{
|
||||
Use: "login",
|
||||
Short: "Log in to the partition's yucca-admin-api via the browser",
|
||||
Long: "Authenticate against the selected partition's admin-api (primary region):\n" +
|
||||
"opens the admin-api CLI login in your browser, receives a one-time code on a\n" +
|
||||
"127.0.0.1 listener, exchanges it for a 24h session JWT, and caches it at\n" +
|
||||
"${XDG_CONFIG_HOME:-~/.config}/yuctl/admin-token-<partition>.json.",
|
||||
Args: cobra.NoArgs,
|
||||
RunE: func(cmd *cobra.Command, _ []string) error {
|
||||
ctx := cmd.Context()
|
||||
cc, err := f.Context()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
topo, err := f.Topology(ctx)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
client, token, err := admin.Login(ctx, f, cc, topo)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
// Round-trip the session so "login" only succeeds when the API
|
||||
// actually accepts the token.
|
||||
sub, err := client.GetAuth(ctx)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
fmt.Fprintf(f.IO.Out, "logged in to %s as %s (expires %s)\n",
|
||||
cc.Partition, sub, token.Expiry.Local().Format(time.RFC3339))
|
||||
return nil
|
||||
},
|
||||
}
|
||||
admin.Register(c)
|
||||
return c
|
||||
}
|
||||
@@ -1,7 +1,6 @@
|
||||
// Package cli wires the yuctl command tree (spf13/cobra). cobra is a documented
|
||||
// divergence from michael's stdlib-only style, justified by yuctl's nested
|
||||
// subcommand surface (select / ceph / infra / users). Logging matches michael
|
||||
// (rs/zerolog).
|
||||
// subcommand surface. Logging matches michael (rs/zerolog).
|
||||
package cli
|
||||
|
||||
import (
|
||||
@@ -14,7 +13,16 @@ import (
|
||||
"github.com/rs/zerolog/log"
|
||||
"github.com/spf13/cobra"
|
||||
|
||||
"yuctl/internal/discovery"
|
||||
cephcmd "yuctl/cli/ceph"
|
||||
configcmd "yuctl/cli/config"
|
||||
featurescmd "yuctl/cli/features"
|
||||
infracmd "yuctl/cli/infra"
|
||||
toolscmd "yuctl/cli/tools"
|
||||
userscmd "yuctl/cli/users"
|
||||
"yuctl/cmdutil"
|
||||
"yuctl/ctxstore"
|
||||
"yuctl/discovery"
|
||||
"yuctl/ui"
|
||||
)
|
||||
|
||||
var (
|
||||
@@ -23,8 +31,9 @@ var (
|
||||
flagRefreshDiscovery bool
|
||||
)
|
||||
|
||||
// NewRootCmd builds the root command and registers every subcommand.
|
||||
func NewRootCmd() *cobra.Command {
|
||||
f := newFactory()
|
||||
|
||||
root := &cobra.Command{
|
||||
Use: "yuctl",
|
||||
Short: "yucca operations CLI",
|
||||
@@ -44,18 +53,46 @@ func NewRootCmd() *cobra.Command {
|
||||
"bypass the cached topology and re-read Terraform state from S3")
|
||||
|
||||
root.AddCommand(
|
||||
newSelectCmd(),
|
||||
newLoginCmd(),
|
||||
newCephCmd(),
|
||||
newInfraCmd(),
|
||||
newUsersCmd(),
|
||||
newConfigCmd(),
|
||||
newFeaturesCmd(),
|
||||
newToolsCmd(),
|
||||
newSelectCmd(f),
|
||||
newLoginCmd(f),
|
||||
cephcmd.New(f),
|
||||
infracmd.New(f),
|
||||
userscmd.New(f),
|
||||
configcmd.New(f),
|
||||
featurescmd.New(f),
|
||||
toolscmd.New(f),
|
||||
)
|
||||
return root
|
||||
}
|
||||
|
||||
// Topology is memoized: several layers of one command may need it (host
|
||||
// resolution, admin URL derivation) but state is read at most once.
|
||||
func newFactory() *cmdutil.Factory {
|
||||
f := &cmdutil.Factory{IO: ui.System()}
|
||||
|
||||
f.Context = func() (*ctxstore.Context, error) {
|
||||
c, err := ctxstore.Load()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if !c.Selected() {
|
||||
return nil, fmt.Errorf("no context selected; run `yuctl select <partition>@<region>` first")
|
||||
}
|
||||
return c, nil
|
||||
}
|
||||
|
||||
var topo *discovery.Topology
|
||||
f.Topology = func(ctx context.Context) (*discovery.Topology, error) {
|
||||
if topo != nil {
|
||||
return topo, nil
|
||||
}
|
||||
var err error
|
||||
topo, err = resolveTopology(ctx)
|
||||
return topo, err
|
||||
}
|
||||
return f
|
||||
}
|
||||
|
||||
func setupLogging() error {
|
||||
zerolog.TimeFieldFormat = time.RFC3339
|
||||
level, err := zerolog.ParseLevel(flagLogLevel)
|
||||
@@ -79,8 +116,7 @@ func setupLogging() error {
|
||||
// resolveTopology returns the cached topology when fresh (see
|
||||
// discovery.LoadCachedTopology; TTL 1h, YUCTL_DISCOVERY_TTL to override,
|
||||
// --refresh-discovery to bypass), otherwise builds the discovery client,
|
||||
// resolves live from S3 state, and refreshes the cache. Shared by every command
|
||||
// that needs to read state.
|
||||
// resolves live from S3 state, and refreshes the cache.
|
||||
func resolveTopology(ctx context.Context) (*discovery.Topology, error) {
|
||||
if !flagRefreshDiscovery {
|
||||
if topo, ok := discovery.LoadCachedTopology(log.Logger); ok {
|
||||
@@ -6,7 +6,8 @@ import (
|
||||
|
||||
"github.com/spf13/cobra"
|
||||
|
||||
yctx "yuctl/internal/context"
|
||||
"yuctl/cmdutil"
|
||||
"yuctl/ctxstore"
|
||||
)
|
||||
|
||||
// parseTarget splits the human form `partition@region`.
|
||||
@@ -18,7 +19,7 @@ func parseTarget(s string) (partition, region string, err error) {
|
||||
return p, r, nil
|
||||
}
|
||||
|
||||
func newSelectCmd() *cobra.Command {
|
||||
func newSelectCmd(f *cmdutil.Factory) *cobra.Command {
|
||||
return &cobra.Command{
|
||||
Use: "select <partition>@<region>",
|
||||
Short: "Select the active partition/region context",
|
||||
@@ -27,13 +28,12 @@ func newSelectCmd() *cobra.Command {
|
||||
"Ceph cluster.",
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
ctx := cmd.Context()
|
||||
partition, region, err := parseTarget(args[0])
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
topo, err := resolveTopology(ctx)
|
||||
topo, err := f.Topology(cmd.Context())
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
@@ -42,24 +42,12 @@ func newSelectCmd() *cobra.Command {
|
||||
partition, region, strings.Join(topo.Regions(), ", "))
|
||||
}
|
||||
|
||||
newCtx := &yctx.Context{Partition: partition, Region: region}
|
||||
if err := yctx.Save(newCtx); err != nil {
|
||||
newCtx := &ctxstore.Context{Partition: partition, Region: region}
|
||||
if err := ctxstore.Save(newCtx); err != nil {
|
||||
return err
|
||||
}
|
||||
fmt.Fprintf(cmd.OutOrStdout(), "selected %s@%s\n", partition, region)
|
||||
fmt.Fprintf(f.IO.Out, "selected %s@%s\n", partition, region)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// requireContext loads the persisted context, erroring if nothing is selected.
|
||||
func requireContext() (*yctx.Context, error) {
|
||||
c, err := yctx.Load()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if !c.Selected() {
|
||||
return nil, fmt.Errorf("no context selected; run `yuctl select <partition>@<region>` first")
|
||||
}
|
||||
return c, nil
|
||||
}
|
||||
@@ -1,6 +1,7 @@
|
||||
package cli
|
||||
package bench
|
||||
|
||||
import (
|
||||
"context"
|
||||
"crypto/rand"
|
||||
"encoding/binary"
|
||||
"encoding/hex"
|
||||
@@ -13,23 +14,79 @@ import (
|
||||
"github.com/rs/zerolog/log"
|
||||
"github.com/spf13/cobra"
|
||||
|
||||
"yuctl/internal/adminapi"
|
||||
"yuctl/internal/bench"
|
||||
yctx "yuctl/internal/context"
|
||||
"yuctl/internal/discovery"
|
||||
"yuctl/adminapi"
|
||||
"yuctl/cmdutil"
|
||||
"yuctl/resticbench"
|
||||
)
|
||||
|
||||
func newToolsCmd() *cobra.Command {
|
||||
tools := &cobra.Command{
|
||||
Use: "tools",
|
||||
Short: "Operational tooling for the selected context",
|
||||
func New(f *cmdutil.Factory) *cobra.Command {
|
||||
o := &options{}
|
||||
cmd := &cobra.Command{
|
||||
Use: "bench",
|
||||
Short: "Benchmark michael end-to-end with restic from a management host",
|
||||
Long: "Pushes a bench agent + a pinned restic to a management host of the selected\n" +
|
||||
"region and runs write/incremental/check/restore phases against michael,\n" +
|
||||
"streaming progress back and saving a results JSON locally. Without --repo,\n" +
|
||||
"a fresh benchmark repository is created via the admin-api (yuctl login).",
|
||||
RunE: func(cmd *cobra.Command, _ []string) error {
|
||||
return o.run(cmd.Context(), f, resticbench.DefaultPhases)
|
||||
},
|
||||
}
|
||||
tools.AddCommand(newBenchCmd(), newBenchWideCmd(), newWarpCmd())
|
||||
return tools
|
||||
o.registerCommon(cmd)
|
||||
cmd.Flags().StringVar(&o.size, "size", "32GiB", "dataset size per cell (e.g. 1TiB)")
|
||||
cmd.Flags().StringVar(&o.fileSize, "file-size", "64MiB", "size of each generated file")
|
||||
cmd.Flags().StringVar(&o.connections, "connections", "5", "rest.connections sweep, comma-separated (e.g. 5,16,32,64)")
|
||||
cmd.Flags().IntVar(&o.readConc, "read-concurrency", 4, "restic backup read concurrency")
|
||||
cmd.Flags().IntVar(&o.packSizeMiB, "pack-size", 16, "restic pack size in MiB (16 = restic default, max 128)")
|
||||
cmd.Flags().StringVar(&o.compression, "compression", "off", "restic compression (off keeps client CPU out of the way; bench data is incompressible)")
|
||||
cmd.Flags().IntVar(&o.incrementals, "incrementals", 2, "incremental backup rounds per cell")
|
||||
cmd.Flags().Float64Var(&o.mutatePercent, "mutate-percent", 2, "percent of files rewritten before each incremental")
|
||||
cmd.Flags().StringVar(&o.phases, "phases", strings.Join(resticbench.DefaultPhases, ","), "phases to run")
|
||||
cmd.Flags().Uint64Var(&o.seed, "seed", 0, "dataset seed (0 = random; reuse for identical content across runs)")
|
||||
cmd.Flags().StringVar(&o.label, "label", "run", "label stored in the results (e.g. before, after)")
|
||||
cmd.Flags().StringVar(&o.out, "out", "", "local results file (default bench-<label>-<timestamp>.json)")
|
||||
cmd.Flags().BoolVar(&o.keepData, "keep-data", false, "keep dataset and restore target on disk (doubles space needs)")
|
||||
cmd.Flags().BoolVar(&o.cleanup, "cleanup", false, "forget+prune this tool's snapshots after the run (also timed)")
|
||||
|
||||
cmd.AddCommand(newCompareCmd(f), newCleanupCmd(f))
|
||||
return cmd
|
||||
}
|
||||
|
||||
type benchFlags struct {
|
||||
admin adminFlags
|
||||
func newCompareCmd(f *cmdutil.Factory) *cobra.Command {
|
||||
return &cobra.Command{
|
||||
Use: "compare <before.json> <after.json>",
|
||||
Short: "Render before/after deltas from two results files",
|
||||
Args: cobra.ExactArgs(2),
|
||||
RunE: func(_ *cobra.Command, args []string) error {
|
||||
before, err := resticbench.LoadResult(args[0])
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
after, err := resticbench.LoadResult(args[1])
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
resticbench.RenderCompare(f.IO.Out, before, after)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
func newCleanupCmd(f *cmdutil.Factory) *cobra.Command {
|
||||
o := &options{}
|
||||
cmd := &cobra.Command{
|
||||
Use: "cleanup",
|
||||
Short: "Forget and prune every snapshot created by the bench (timed)",
|
||||
RunE: func(cmd *cobra.Command, _ []string) error {
|
||||
return o.run(cmd.Context(), f, []string{resticbench.PhaseCleanup})
|
||||
},
|
||||
}
|
||||
o.registerCommon(cmd)
|
||||
return cmd
|
||||
}
|
||||
|
||||
type options struct {
|
||||
admin cmdutil.AdminFlags
|
||||
|
||||
host string
|
||||
fromHere bool
|
||||
@@ -56,116 +113,40 @@ type benchFlags struct {
|
||||
cleanup bool
|
||||
}
|
||||
|
||||
func (f *benchFlags) registerCommon(c *cobra.Command) {
|
||||
f.admin.register(c)
|
||||
c.Flags().StringVar(&f.host, "host", "", "ssh destination of the management host (default: the region's first mgmt host from discovery)")
|
||||
c.Flags().BoolVar(&f.fromHere, "from-here", false, "run the benchmark on this machine (no ssh; agent runs in-process)")
|
||||
c.Flags().StringVar(&f.sshIdentity, "ssh-identity", "", "ssh private key for the management host (default: ssh agent/config)")
|
||||
c.Flags().StringVar(&f.sshUser, "ssh-user", "", "ssh username for the discovery-resolved mgmt host (default: your local username; identity-registry accounts differ)")
|
||||
c.Flags().StringVar(&f.agentBin, "agent-bin", "", "local linux/amd64 bench-agent binary (default: the embedded one)")
|
||||
c.Flags().StringVar(&f.repo, "repo", "", "restic repository URL; skips admin-api provisioning (default $RESTIC_REPOSITORY, else a repo is created via admin-api)")
|
||||
c.Flags().StringVar(&f.repoID, "repo-id", "", "existing repository id; a fresh URL is minted via admin-api")
|
||||
c.Flags().StringVar(&f.passwordFile, "password-file", "", "local file with the restic password (default $RESTIC_PASSWORD; auto-generated for auto-created repos)")
|
||||
c.Flags().StringVar(&f.workdir, "workdir", "/var/tmp/yucca-bench", "remote scratch directory (must fit --size)")
|
||||
func (o *options) registerCommon(c *cobra.Command) {
|
||||
o.admin.Register(c)
|
||||
c.Flags().StringVar(&o.host, "host", "", "ssh destination of the management host (default: the region's first mgmt host from discovery)")
|
||||
c.Flags().BoolVar(&o.fromHere, "from-here", false, "run the benchmark on this machine (no ssh; agent runs in-process)")
|
||||
c.Flags().StringVar(&o.sshIdentity, "ssh-identity", "", "ssh private key for the management host (default: ssh agent/config)")
|
||||
c.Flags().StringVar(&o.sshUser, "ssh-user", "", "ssh username for the discovery-resolved mgmt host (default: your local username; identity-registry accounts differ)")
|
||||
c.Flags().StringVar(&o.agentBin, "agent-bin", "", "local linux/amd64 bench-agent binary (default: the embedded one)")
|
||||
c.Flags().StringVar(&o.repo, "repo", "", "restic repository URL; skips admin-api provisioning (default $RESTIC_REPOSITORY, else a repo is created via admin-api)")
|
||||
c.Flags().StringVar(&o.repoID, "repo-id", "", "existing repository id; a fresh URL is minted via admin-api")
|
||||
c.Flags().StringVar(&o.passwordFile, "password-file", "", "local file with the restic password (default $RESTIC_PASSWORD; auto-generated for auto-created repos)")
|
||||
c.Flags().StringVar(&o.workdir, "workdir", "/var/tmp/yucca-bench", "remote scratch directory (must fit --size)")
|
||||
}
|
||||
|
||||
func newBenchCmd() *cobra.Command {
|
||||
f := &benchFlags{}
|
||||
cmd := &cobra.Command{
|
||||
Use: "bench",
|
||||
Short: "Benchmark michael end-to-end with restic from a management host",
|
||||
Long: "Pushes a bench agent + a pinned restic to a management host of the selected\n" +
|
||||
"region and runs write/incremental/check/restore phases against michael,\n" +
|
||||
"streaming progress back and saving a results JSON locally. Without --repo,\n" +
|
||||
"a fresh benchmark repository is created via the admin-api (yuctl login).",
|
||||
RunE: func(cmd *cobra.Command, _ []string) error {
|
||||
return f.runBench(cmd, bench.DefaultPhases)
|
||||
},
|
||||
}
|
||||
f.registerCommon(cmd)
|
||||
cmd.Flags().StringVar(&f.size, "size", "32GiB", "dataset size per cell (e.g. 1TiB)")
|
||||
cmd.Flags().StringVar(&f.fileSize, "file-size", "64MiB", "size of each generated file")
|
||||
cmd.Flags().StringVar(&f.connections, "connections", "5", "rest.connections sweep, comma-separated (e.g. 5,16,32,64)")
|
||||
cmd.Flags().IntVar(&f.readConc, "read-concurrency", 4, "restic backup read concurrency")
|
||||
cmd.Flags().IntVar(&f.packSizeMiB, "pack-size", 16, "restic pack size in MiB (16 = restic default, max 128)")
|
||||
cmd.Flags().StringVar(&f.compression, "compression", "off", "restic compression (off keeps client CPU out of the way; bench data is incompressible)")
|
||||
cmd.Flags().IntVar(&f.incrementals, "incrementals", 2, "incremental backup rounds per cell")
|
||||
cmd.Flags().Float64Var(&f.mutatePercent, "mutate-percent", 2, "percent of files rewritten before each incremental")
|
||||
cmd.Flags().StringVar(&f.phases, "phases", strings.Join(bench.DefaultPhases, ","), "phases to run")
|
||||
cmd.Flags().Uint64Var(&f.seed, "seed", 0, "dataset seed (0 = random; reuse for identical content across runs)")
|
||||
cmd.Flags().StringVar(&f.label, "label", "run", "label stored in the results (e.g. before, after)")
|
||||
cmd.Flags().StringVar(&f.out, "out", "", "local results file (default bench-<label>-<timestamp>.json)")
|
||||
cmd.Flags().BoolVar(&f.keepData, "keep-data", false, "keep dataset and restore target on disk (doubles space needs)")
|
||||
cmd.Flags().BoolVar(&f.cleanup, "cleanup", false, "forget+prune this tool's snapshots after the run (also timed)")
|
||||
|
||||
cmd.AddCommand(newBenchCompareCmd(), newBenchCleanupCmd())
|
||||
return cmd
|
||||
}
|
||||
|
||||
func newBenchCompareCmd() *cobra.Command {
|
||||
return &cobra.Command{
|
||||
Use: "compare <before.json> <after.json>",
|
||||
Short: "Render before/after deltas from two results files",
|
||||
Args: cobra.ExactArgs(2),
|
||||
RunE: func(_ *cobra.Command, args []string) error {
|
||||
before, err := bench.LoadResult(args[0])
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
after, err := bench.LoadResult(args[1])
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
bench.RenderCompare(os.Stdout, before, after)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
func newBenchCleanupCmd() *cobra.Command {
|
||||
f := &benchFlags{}
|
||||
cmd := &cobra.Command{
|
||||
Use: "cleanup",
|
||||
Short: "Forget and prune every snapshot created by the bench (timed)",
|
||||
RunE: func(cmd *cobra.Command, _ []string) error {
|
||||
return f.runBench(cmd, []string{bench.PhaseCleanup})
|
||||
},
|
||||
}
|
||||
f.registerCommon(cmd)
|
||||
return cmd
|
||||
}
|
||||
|
||||
// runBench resolves the context (host from discovery, repository via admin-api
|
||||
// unless supplied) and drives the run. Context/topology are resolved lazily so
|
||||
// a --from-here run with an explicit --repo needs neither state creds nor a
|
||||
// selected context (e.g. yuctl invoked directly on a mgmt host).
|
||||
func (f *benchFlags) runBench(cmd *cobra.Command, defaultPhases []string) error {
|
||||
ctx := cmd.Context()
|
||||
|
||||
var cc *yctx.Context
|
||||
var topo *discovery.Topology
|
||||
loadTopo := func() error {
|
||||
if topo != nil {
|
||||
return nil
|
||||
}
|
||||
var err error
|
||||
if cc, err = requireContext(); err != nil {
|
||||
return err
|
||||
}
|
||||
topo, err = resolveTopology(ctx)
|
||||
return err
|
||||
}
|
||||
|
||||
// run resolves the target (host from discovery, repository via admin-api
|
||||
// unless supplied) and drives the benchmark. Context/topology are resolved
|
||||
// lazily through the factory so a --from-here run with an explicit --repo
|
||||
// needs neither state creds nor a selected context (e.g. yuctl invoked
|
||||
// directly on a mgmt host).
|
||||
func (o *options) run(ctx context.Context, f *cmdutil.Factory, defaultPhases []string) error {
|
||||
var host string
|
||||
switch {
|
||||
case f.fromHere && f.host != "":
|
||||
case o.fromHere && o.host != "":
|
||||
return fmt.Errorf("--from-here and --host are mutually exclusive")
|
||||
case f.fromHere:
|
||||
case o.fromHere:
|
||||
// no remote host: the agent runs in-process on this machine
|
||||
case f.host != "":
|
||||
host = f.host
|
||||
case o.host != "":
|
||||
host = o.host
|
||||
default:
|
||||
if err := loadTopo(); err != nil {
|
||||
cc, err := f.Context()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
topo, err := f.Topology(ctx)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
hosts := topo.MgmtHosts(cc.Partition, cc.Region)
|
||||
@@ -173,35 +154,31 @@ func (f *benchFlags) runBench(cmd *cobra.Command, defaultPhases []string) error
|
||||
return fmt.Errorf("discovery has no mgmt hosts for %s@%s; pass --host (or --from-here)", cc.Partition, cc.Region)
|
||||
}
|
||||
host = hosts[0].PublicIP
|
||||
if f.sshUser != "" {
|
||||
host = f.sshUser + "@" + host
|
||||
if o.sshUser != "" {
|
||||
host = o.sshUser + "@" + host
|
||||
}
|
||||
log.Info().Str("mgmt", hosts[0].Name).Str("host", host).Msg("using mgmt host from discovery")
|
||||
}
|
||||
|
||||
cfg, err := f.buildConfig(defaultPhases)
|
||||
cfg, err := o.buildConfig(defaultPhases)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
if cfg.Repo == "" {
|
||||
// Topology is only needed to derive the admin URL; an explicit
|
||||
// --admin-url / $YUCTL_ADMIN_API_URL skips state access entirely
|
||||
// (the context file alone names the token-cache partition).
|
||||
if f.admin.adminURL == "" && os.Getenv("YUCTL_ADMIN_API_URL") == "" {
|
||||
if err := loadTopo(); err != nil {
|
||||
return err
|
||||
}
|
||||
} else if cc == nil {
|
||||
if cc, err = requireContext(); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
client, _, err := f.admin.adminLogin(ctx, cmd, cc, topo)
|
||||
cc, err := f.Context()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
repoID := f.repoID
|
||||
topo, err := o.admin.OptionalTopology(ctx, f)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
client, _, err := o.admin.Login(ctx, f, cc, topo)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
repoID := o.repoID
|
||||
if repoID == "" {
|
||||
name := fmt.Sprintf("yucca-bench-%s-%s", cfg.Label, time.Now().Format("20060102-150405"))
|
||||
repo, err := client.CreateRepository(ctx, name, false, adminapi.CreateRepositoryOptions{})
|
||||
@@ -228,47 +205,47 @@ func (f *benchFlags) runBench(cmd *cobra.Command, defaultPhases []string) error
|
||||
return fmt.Errorf("no password: set --password-file or RESTIC_PASSWORD")
|
||||
}
|
||||
|
||||
if f.out == "" {
|
||||
f.out = fmt.Sprintf("bench-%s-%s.json", cfg.Label, time.Now().Format("20060102-150405"))
|
||||
if o.out == "" {
|
||||
o.out = fmt.Sprintf("bench-%s-%s.json", cfg.Label, time.Now().Format("20060102-150405"))
|
||||
}
|
||||
outPath := f.out
|
||||
if len(cfg.Phases) == 1 && cfg.Phases[0] == bench.PhaseCleanup {
|
||||
outPath := o.out
|
||||
if len(cfg.Phases) == 1 && cfg.Phases[0] == resticbench.PhaseCleanup {
|
||||
outPath = ""
|
||||
}
|
||||
|
||||
opts := bench.RunOpts{Host: host, SSHIdentity: f.sshIdentity, AgentBin: f.agentBin, Config: cfg, Out: outPath}
|
||||
if f.fromHere {
|
||||
_, err = bench.RunHere(ctx, opts)
|
||||
opts := resticbench.RunOpts{Host: host, SSHIdentity: o.sshIdentity, AgentBin: o.agentBin, Config: cfg, Out: outPath, Summary: f.IO.Out}
|
||||
if o.fromHere {
|
||||
_, err = resticbench.RunHere(ctx, opts)
|
||||
} else {
|
||||
_, err = bench.Run(ctx, opts)
|
||||
_, err = resticbench.Run(ctx, opts)
|
||||
}
|
||||
return err
|
||||
}
|
||||
|
||||
func (f *benchFlags) buildConfig(defaultPhases []string) (bench.Config, error) {
|
||||
cfg := bench.Config{
|
||||
Workdir: f.workdir,
|
||||
ReadConcurrency: f.readConc,
|
||||
PackSizeMiB: f.packSizeMiB,
|
||||
Compression: f.compression,
|
||||
Incrementals: f.incrementals,
|
||||
MutatePercent: f.mutatePercent,
|
||||
Label: f.label,
|
||||
func (o *options) buildConfig(defaultPhases []string) (resticbench.Config, error) {
|
||||
cfg := resticbench.Config{
|
||||
Workdir: o.workdir,
|
||||
ReadConcurrency: o.readConc,
|
||||
PackSizeMiB: o.packSizeMiB,
|
||||
Compression: o.compression,
|
||||
Incrementals: o.incrementals,
|
||||
MutatePercent: o.mutatePercent,
|
||||
Label: o.label,
|
||||
Tag: "yucca-bench",
|
||||
KeepData: f.keepData,
|
||||
Seed: f.seed,
|
||||
KeepData: o.keepData,
|
||||
Seed: o.seed,
|
||||
Phases: defaultPhases,
|
||||
}
|
||||
if cfg.Label == "" {
|
||||
cfg.Label = "run"
|
||||
}
|
||||
|
||||
cfg.Repo = f.repo
|
||||
if cfg.Repo == "" && f.repoID == "" {
|
||||
cfg.Repo = o.repo
|
||||
if cfg.Repo == "" && o.repoID == "" {
|
||||
cfg.Repo = os.Getenv("RESTIC_REPOSITORY")
|
||||
}
|
||||
if f.passwordFile != "" {
|
||||
b, err := os.ReadFile(f.passwordFile)
|
||||
if o.passwordFile != "" {
|
||||
b, err := os.ReadFile(o.passwordFile)
|
||||
if err != nil {
|
||||
return cfg, err
|
||||
}
|
||||
@@ -277,21 +254,21 @@ func (f *benchFlags) buildConfig(defaultPhases []string) (bench.Config, error) {
|
||||
cfg.Password = os.Getenv("RESTIC_PASSWORD")
|
||||
}
|
||||
|
||||
if f.size != "" {
|
||||
size, err := bench.ParseSize(f.size)
|
||||
if o.size != "" {
|
||||
size, err := resticbench.ParseSize(o.size)
|
||||
if err != nil {
|
||||
return cfg, err
|
||||
}
|
||||
cfg.Size = size
|
||||
fileSize, err := bench.ParseSize(f.fileSize)
|
||||
fileSize, err := resticbench.ParseSize(o.fileSize)
|
||||
if err != nil {
|
||||
return cfg, err
|
||||
}
|
||||
cfg.FileSize = fileSize
|
||||
}
|
||||
|
||||
if f.connections != "" {
|
||||
for _, s := range strings.Split(f.connections, ",") {
|
||||
if o.connections != "" {
|
||||
for _, s := range strings.Split(o.connections, ",") {
|
||||
n, err := strconv.Atoi(strings.TrimSpace(s))
|
||||
if err != nil || n < 1 {
|
||||
return cfg, fmt.Errorf("invalid --connections value %q", s)
|
||||
@@ -300,11 +277,11 @@ func (f *benchFlags) buildConfig(defaultPhases []string) (bench.Config, error) {
|
||||
}
|
||||
}
|
||||
|
||||
if f.phases != "" {
|
||||
cfg.Phases = strings.Split(f.phases, ",")
|
||||
if o.phases != "" {
|
||||
cfg.Phases = strings.Split(o.phases, ",")
|
||||
}
|
||||
if f.cleanup && !cfg.HasPhase(bench.PhaseCleanup) {
|
||||
cfg.Phases = append(cfg.Phases, bench.PhaseCleanup)
|
||||
if o.cleanup && !cfg.HasPhase(resticbench.PhaseCleanup) {
|
||||
cfg.Phases = append(cfg.Phases, resticbench.PhaseCleanup)
|
||||
}
|
||||
|
||||
if cfg.Seed == 0 {
|
||||
+101
-131
@@ -1,43 +1,41 @@
|
||||
package cli
|
||||
package fleetbench
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"context"
|
||||
"fmt"
|
||||
"os"
|
||||
"os/signal"
|
||||
"strings"
|
||||
"syscall"
|
||||
"time"
|
||||
|
||||
"github.com/rs/zerolog/log"
|
||||
"github.com/spf13/cobra"
|
||||
|
||||
"yuctl/internal/adminapi"
|
||||
"yuctl/internal/bench"
|
||||
"yuctl/internal/benchwide"
|
||||
"yuctl/internal/provider"
|
||||
"yuctl/cmdutil"
|
||||
"yuctl/fleet"
|
||||
fbfleet "yuctl/fleet/fleetbench"
|
||||
"yuctl/provider"
|
||||
"yuctl/resticbench"
|
||||
)
|
||||
|
||||
// benchDoFlags are shared by every bench-wide subcommand.
|
||||
type benchDoFlags struct {
|
||||
// flags are shared by every fleet-bench subcommand.
|
||||
type flags struct {
|
||||
yes bool
|
||||
provider string
|
||||
}
|
||||
|
||||
func (f *benchDoFlags) register(c *cobra.Command) {
|
||||
c.PersistentFlags().BoolVar(&f.yes, "yes", false, "skip the host-creation confirmation prompt")
|
||||
c.PersistentFlags().StringVar(&f.provider, "provider", "do", "cloud provider for this fleet ("+strings.Join(provider.Names(), " | ")+")")
|
||||
func (o *flags) register(c *cobra.Command) {
|
||||
c.PersistentFlags().BoolVar(&o.yes, "yes", false, "skip the host-creation confirmation prompt")
|
||||
c.PersistentFlags().StringVar(&o.provider, "provider", "do", "cloud provider for this fleet ("+strings.Join(provider.Names(), " | ")+")")
|
||||
}
|
||||
|
||||
// session opens the fleet session for the selected partition + provider. The
|
||||
// label describes the target for display ("prod · yuctl-bench-do-prod").
|
||||
func (f *benchDoFlags) session(ctx context.Context) (*benchwide.Session, string, error) {
|
||||
cc, err := requireContext()
|
||||
func (o *flags) session(ctx context.Context, f *cmdutil.Factory) (*fbfleet.Session, string, error) {
|
||||
cc, err := f.Context()
|
||||
if err != nil {
|
||||
return nil, "", err
|
||||
}
|
||||
s, err := benchwide.NewSession(ctx, cc.Partition, f.provider)
|
||||
s, err := fbfleet.NewSession(ctx, cc.Partition, o.provider)
|
||||
if err != nil {
|
||||
return nil, "", err
|
||||
}
|
||||
@@ -45,43 +43,32 @@ func (f *benchDoFlags) session(ctx context.Context) (*benchwide.Session, string,
|
||||
}
|
||||
|
||||
// minter runs the admin-api login flow and returns the repo-minting client.
|
||||
func (f *benchDoFlags) minter(ctx context.Context, cmd *cobra.Command, admin *adminFlags) (benchwide.RepoMinter, error) {
|
||||
cc, err := requireContext()
|
||||
func (o *flags) minter(ctx context.Context, f *cmdutil.Factory, admin *cmdutil.AdminFlags) (fbfleet.RepoMinter, error) {
|
||||
cc, err := f.Context()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
// Topology is only needed to derive the admin URL; an explicit --admin-url
|
||||
// or $YUCTL_ADMIN_API_URL skips state access entirely.
|
||||
var client *adminapi.Client
|
||||
if admin.adminURL == "" && os.Getenv("YUCTL_ADMIN_API_URL") == "" {
|
||||
topo, err := resolveTopology(ctx)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
client, _, err = admin.adminLogin(ctx, cmd, cc, topo)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
} else {
|
||||
client, _, err = admin.adminLogin(ctx, cmd, cc, nil)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
topo, err := admin.OptionalTopology(ctx, f)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
client, _, err := admin.Login(ctx, f, cc, topo)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return client, nil
|
||||
}
|
||||
|
||||
// confirm returns the deploy confirmation callback: nil with --yes, otherwise
|
||||
// an interactive y/N prompt on the terminal.
|
||||
func (f *benchDoFlags) confirm(cmd *cobra.Command) func(string) bool {
|
||||
if f.yes {
|
||||
func (o *flags) confirm(f *cmdutil.Factory) func(string) bool {
|
||||
if o.yes {
|
||||
return nil
|
||||
}
|
||||
return func(plan string) bool {
|
||||
out := cmd.ErrOrStderr()
|
||||
fmt.Fprintln(out, "\nbench-wide will "+plan)
|
||||
fmt.Fprint(out, "\nProceed? [y/N] ")
|
||||
sc := bufio.NewScanner(cmd.InOrStdin())
|
||||
fmt.Fprintln(f.IO.Err, "\nfleet-bench will "+plan)
|
||||
fmt.Fprint(f.IO.Err, "\nProceed? [y/N] ")
|
||||
sc := bufio.NewScanner(f.IO.In)
|
||||
if !sc.Scan() {
|
||||
return false
|
||||
}
|
||||
@@ -90,10 +77,10 @@ func (f *benchDoFlags) confirm(cmd *cobra.Command) func(string) bool {
|
||||
}
|
||||
}
|
||||
|
||||
func newBenchWideCmd() *cobra.Command {
|
||||
f := &benchDoFlags{}
|
||||
func New(f *cmdutil.Factory) *cobra.Command {
|
||||
o := &flags{}
|
||||
cmd := &cobra.Command{
|
||||
Use: "bench-wide",
|
||||
Use: "fleet-bench",
|
||||
Short: "Restic client fleet across cloud providers writing against michael",
|
||||
Long: "Deploys a fleet of cloud VMs on a chosen --provider (DigitalOcean, Hetzner;\n" +
|
||||
"OVH later) with a per-fleet ephemeral ssh key and runs real restic clients\n" +
|
||||
@@ -104,48 +91,48 @@ func newBenchWideCmd() *cobra.Command {
|
||||
"overage. Fleets are per provider × partition, so several providers can load\n" +
|
||||
"michael at once (run deploy/start per --provider).",
|
||||
}
|
||||
f.register(cmd)
|
||||
o.register(cmd)
|
||||
cmd.AddCommand(
|
||||
newBenchDoDeployCmd(f),
|
||||
newBenchDoStartCmd(f),
|
||||
newBenchDoStatusCmd(f),
|
||||
newBenchDoWatchCmd(f),
|
||||
newBenchDoStopCmd(f),
|
||||
newBenchDoCleanupCmd(f),
|
||||
newBenchDoUndeployCmd(f),
|
||||
newDeployCmd(f, o),
|
||||
newStartCmd(f, o),
|
||||
newStatusCmd(f, o),
|
||||
newWatchCmd(f, o),
|
||||
newStopCmd(f, o),
|
||||
newCleanupCmd(f, o),
|
||||
newUndeployCmd(f, o),
|
||||
)
|
||||
return cmd
|
||||
}
|
||||
|
||||
func benchDoDeployFlags(c *cobra.Command, o *benchwide.DeployOptions) {
|
||||
c.Flags().IntVar(&o.Droplets, "droplets", 3, "fleet size")
|
||||
func deployFlags(c *cobra.Command, o *fbfleet.DeployOptions) {
|
||||
c.Flags().IntVar(&o.Hosts, "hosts", 3, "fleet size")
|
||||
c.Flags().StringSliceVar(&o.Regions, "region", nil, "regions, round-robined across the fleet (default: the provider's)")
|
||||
c.Flags().StringVar(&o.Size, "size", "", "instance size slug (default: the provider's)")
|
||||
c.Flags().StringVar(&o.Image, "image", "", "image slug (default: the provider's)")
|
||||
c.Flags().StringVar(&o.AgentBin, "agent-bin", "", "local linux/amd64 bench-agent binary (default: the embedded one)")
|
||||
}
|
||||
|
||||
func newBenchDoDeployCmd(f *benchDoFlags) *cobra.Command {
|
||||
o := benchwide.DeployOptions{}
|
||||
func newDeployCmd(f *cmdutil.Factory, fl *flags) *cobra.Command {
|
||||
o := fbfleet.DeployOptions{}
|
||||
cmd := &cobra.Command{
|
||||
Use: "deploy",
|
||||
Short: "Create (or converge) the droplet fleet and push the agent + restic",
|
||||
Short: "Create (or converge) the host fleet and push the agent + restic",
|
||||
RunE: func(cmd *cobra.Command, _ []string) error {
|
||||
s, _, err := f.session(cmd.Context())
|
||||
s, _, err := fl.session(cmd.Context(), f)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
o.Confirm = f.confirm(cmd)
|
||||
o.Confirm = fl.confirm(f)
|
||||
return s.Deploy(cmd.Context(), o)
|
||||
},
|
||||
}
|
||||
benchDoDeployFlags(cmd, &o)
|
||||
deployFlags(cmd, &o)
|
||||
return cmd
|
||||
}
|
||||
|
||||
func newBenchDoStartCmd(f *benchDoFlags) *cobra.Command {
|
||||
admin := &adminFlags{}
|
||||
d := benchwide.DeployOptions{}
|
||||
func newStartCmd(f *cmdutil.Factory, fl *flags) *cobra.Command {
|
||||
admin := &cmdutil.AdminFlags{}
|
||||
d := fbfleet.DeployOptions{}
|
||||
var (
|
||||
objSize string
|
||||
cycleSize string
|
||||
@@ -154,25 +141,25 @@ func newBenchDoStartCmd(f *benchDoFlags) *cobra.Command {
|
||||
maxTransfer string
|
||||
autoDeploy bool
|
||||
)
|
||||
o := benchwide.StartOptions{}
|
||||
o := fbfleet.StartOptions{}
|
||||
cmd := &cobra.Command{
|
||||
Use: "start",
|
||||
Short: "Start (or gracefully restart) the restic load on the fleet",
|
||||
Long: "Ensures one repository per client (admin-api; `yuctl login` session), mints\n" +
|
||||
"fresh restic URLs, and launches the detached load supervisor on every\n" +
|
||||
"droplet. A second start kills the previous load first and relaunches with\n" +
|
||||
"the new parameters. The load ends when --duration elapses, the per-droplet\n" +
|
||||
"transfer cap is hit, or `bench-wide stop`.",
|
||||
"host. A second start kills the previous load first and relaunches with\n" +
|
||||
"the new parameters. The load ends when --duration elapses, the per-host\n" +
|
||||
"transfer cap is hit, or `fleet-bench stop`.",
|
||||
RunE: func(cmd *cobra.Command, _ []string) error {
|
||||
ctx := cmd.Context()
|
||||
var err error
|
||||
if o.PackSizeMiB, err = parseMiB(objSize, "--obj-size"); err != nil {
|
||||
return err
|
||||
}
|
||||
if o.CycleSize, err = bench.ParseSize(cycleSize); err != nil {
|
||||
if o.CycleSize, err = resticbench.ParseSize(cycleSize); err != nil {
|
||||
return err
|
||||
}
|
||||
if o.FileSize, err = bench.ParseSize(fileSize); err != nil {
|
||||
if o.FileSize, err = resticbench.ParseSize(fileSize); err != nil {
|
||||
return err
|
||||
}
|
||||
if duration != "" && duration != "0" {
|
||||
@@ -183,49 +170,49 @@ func newBenchDoStartCmd(f *benchDoFlags) *cobra.Command {
|
||||
o.Duration = dur
|
||||
}
|
||||
if maxTransfer != "" {
|
||||
if o.MaxTransfer, err = bench.ParseSize(maxTransfer); err != nil {
|
||||
if o.MaxTransfer, err = resticbench.ParseSize(maxTransfer); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
|
||||
s, _, err := f.session(ctx)
|
||||
s, _, err := fl.session(ctx, f)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if autoDeploy {
|
||||
if droplets, err := s.Droplets(ctx); err == nil && len(droplets) == 0 {
|
||||
d.Confirm = f.confirm(cmd)
|
||||
if hosts, err := s.Hosts(ctx); err == nil && len(hosts) == 0 {
|
||||
d.Confirm = fl.confirm(f)
|
||||
if err := s.Deploy(ctx, d); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
}
|
||||
minter, err := f.minter(ctx, cmd, admin)
|
||||
minter, err := fl.minter(ctx, f, admin)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
return s.Start(ctx, minter, o)
|
||||
},
|
||||
}
|
||||
cmd.Flags().IntVar(&o.ClientsPerDroplet, "clients-per-droplet", 1, "restic clients per droplet (each with its own repository)")
|
||||
cmd.Flags().IntVar(&o.ClientsPerHost, "clients-per-host", 1, "restic clients per host (each with its own repository)")
|
||||
cmd.Flags().StringVar(&objSize, "obj-size", "16MiB", "restic pack size — the object size michael sees (4..128 MiB)")
|
||||
cmd.Flags().StringVar(&cycleSize, "cycle-size", "8GiB", "dataset per client per cycle (freshly seeded every cycle)")
|
||||
cmd.Flags().StringVar(&fileSize, "file-size", "64MiB", "size of each generated file")
|
||||
cmd.Flags().IntVar(&o.Connections, "connections", 5, "rest.connections per client")
|
||||
cmd.Flags().IntVar(&o.ReadConcurrency, "read-concurrency", 4, "restic backup read concurrency")
|
||||
cmd.Flags().StringVar(&o.Compression, "compression", "off", "restic compression (bench data is incompressible)")
|
||||
cmd.Flags().StringVar(&duration, "duration", "1h", "run length (e.g. 2h; 0 = non-stop until bench-wide stop)")
|
||||
cmd.Flags().StringVar(&maxTransfer, "max-transfer", "", "per-droplet wire-TX cap (default: the droplet size's transfer allowance)")
|
||||
cmd.Flags().StringVar(&duration, "duration", "1h", "run length (e.g. 2h; 0 = non-stop until fleet-bench stop)")
|
||||
cmd.Flags().StringVar(&maxTransfer, "max-transfer", "", "per-host wire-TX cap (default: the host size's transfer allowance)")
|
||||
cmd.Flags().StringVar(&o.Label, "label", "run", "label stored in the results")
|
||||
cmd.Flags().Uint64Var(&o.Seed, "seed", 0, "dataset seed (0 = random)")
|
||||
cmd.Flags().BoolVar(&autoDeploy, "auto-deploy", true, "deploy the fleet first if it is missing (asks before creating droplets)")
|
||||
benchDoDeployFlags(cmd, &d)
|
||||
admin.register(cmd)
|
||||
cmd.Flags().BoolVar(&autoDeploy, "auto-deploy", true, "deploy the fleet first if it is missing (asks before creating hosts)")
|
||||
deployFlags(cmd, &d)
|
||||
admin.Register(cmd)
|
||||
return cmd
|
||||
}
|
||||
|
||||
func parseMiB(v, flag string) (int, error) {
|
||||
n, err := bench.ParseSize(v)
|
||||
n, err := resticbench.ParseSize(v)
|
||||
if err != nil {
|
||||
return 0, fmt.Errorf("%s: %w", flag, err)
|
||||
}
|
||||
@@ -235,23 +222,23 @@ func parseMiB(v, flag string) (int, error) {
|
||||
return int(n >> 20), nil
|
||||
}
|
||||
|
||||
func newBenchDoStatusCmd(f *benchDoFlags) *cobra.Command {
|
||||
func newStatusCmd(f *cmdutil.Factory, fl *flags) *cobra.Command {
|
||||
var sample int
|
||||
cmd := &cobra.Command{
|
||||
Use: "status",
|
||||
Short: "Show per-droplet load state, throughput, and transfer budget",
|
||||
Short: "Show per-host load state, throughput, and transfer budget",
|
||||
RunE: func(cmd *cobra.Command, _ []string) error {
|
||||
s, label, err := f.session(cmd.Context())
|
||||
s, label, err := fl.session(cmd.Context(), f)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
log.Info().Int("window_s", sample).Msg("sampling droplets")
|
||||
log.Info().Int("window_s", sample).Msg("sampling hosts")
|
||||
report, err := s.Status(cmd.Context(), sample)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
v := &benchDoView{label: label}
|
||||
fmt.Fprintln(cmd.OutOrStdout(), v.render(report, time.Now(), sample, false))
|
||||
v := &view{label: label}
|
||||
fmt.Fprintln(f.IO.Out, v.render(report, time.Now(), sample, false))
|
||||
return nil
|
||||
},
|
||||
}
|
||||
@@ -259,61 +246,44 @@ func newBenchDoStatusCmd(f *benchDoFlags) *cobra.Command {
|
||||
return cmd
|
||||
}
|
||||
|
||||
func newBenchDoWatchCmd(f *benchDoFlags) *cobra.Command {
|
||||
func newWatchCmd(f *cmdutil.Factory, fl *flags) *cobra.Command {
|
||||
var sample int
|
||||
cmd := &cobra.Command{
|
||||
Use: "watch",
|
||||
Short: "Live dashboard: continuously sampled fleet state and throughput",
|
||||
RunE: func(cmd *cobra.Command, _ []string) error {
|
||||
ctx, stop := signal.NotifyContext(cmd.Context(), os.Interrupt, syscall.SIGTERM)
|
||||
defer stop()
|
||||
s, label, err := f.session(ctx)
|
||||
s, label, err := fl.session(cmd.Context(), f)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
out := cmd.OutOrStdout()
|
||||
fmt.Fprint(out, "\x1b[?1049h\x1b[?25l") // alt screen, hide cursor
|
||||
defer fmt.Fprint(out, "\x1b[?25h\x1b[?1049l")
|
||||
fmt.Fprintf(out, "%s connecting · sampling %ds window…\n", label, sample)
|
||||
|
||||
v := &benchDoView{label: label}
|
||||
for {
|
||||
v := &view{label: label}
|
||||
return fleet.Watch(cmd.Context(), f.IO.Out, label, sample, func(ctx context.Context) (string, error) {
|
||||
report, err := s.Status(ctx, sample)
|
||||
if ctx.Err() != nil {
|
||||
return nil
|
||||
}
|
||||
fmt.Fprint(out, "\x1b[H\x1b[2J")
|
||||
if err != nil {
|
||||
fmt.Fprintf(out, "status error (retrying): %v\n", err)
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return nil
|
||||
case <-time.After(3 * time.Second):
|
||||
}
|
||||
continue
|
||||
return "", err
|
||||
}
|
||||
var combined float64
|
||||
for _, d := range report.Droplets {
|
||||
for _, d := range report.Hosts {
|
||||
combined += d.TxBps
|
||||
}
|
||||
if len(report.Droplets) > 0 {
|
||||
v.push(combined)
|
||||
if len(report.Hosts) > 0 {
|
||||
v.history.Push(combined)
|
||||
}
|
||||
fmt.Fprintln(out, v.render(report, time.Now(), sample, true))
|
||||
}
|
||||
return v.render(report, time.Now(), sample, true), nil
|
||||
})
|
||||
},
|
||||
}
|
||||
cmd.Flags().IntVar(&sample, "sample", 5, "NIC sampling window per refresh in seconds")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func newBenchDoStopCmd(f *benchDoFlags) *cobra.Command {
|
||||
func newStopCmd(f *cmdutil.Factory, fl *flags) *cobra.Command {
|
||||
var out string
|
||||
cmd := &cobra.Command{
|
||||
Use: "stop",
|
||||
Short: "Stop the load everywhere and save the results JSON (droplets stay)",
|
||||
Short: "Stop the load everywhere and save the results JSON (hosts stay)",
|
||||
RunE: func(cmd *cobra.Command, _ []string) error {
|
||||
s, _, err := f.session(cmd.Context())
|
||||
s, _, err := fl.session(cmd.Context(), f)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
@@ -322,36 +292,36 @@ func newBenchDoStopCmd(f *benchDoFlags) *cobra.Command {
|
||||
return err
|
||||
}
|
||||
if out == "" {
|
||||
out = fmt.Sprintf("bench-wide-%s-%s.json", res.Label, time.Now().Format("20060102-150405"))
|
||||
out = fmt.Sprintf("fleet-bench-%s-%s.json", res.Label, time.Now().Format("20060102-150405"))
|
||||
}
|
||||
if err := saveBenchDoResult(out, res); err != nil {
|
||||
if err := fbfleet.SaveResult(out, res); err != nil {
|
||||
return err
|
||||
}
|
||||
log.Info().Str("path", out).Msg("results saved")
|
||||
renderBenchDoResult(cmd.OutOrStdout(), res)
|
||||
fbfleet.RenderResult(f.IO.Out, res)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().StringVar(&out, "out", "", "local results file (default bench-wide-<label>-<timestamp>.json)")
|
||||
cmd.Flags().StringVar(&out, "out", "", "local results file (default fleet-bench-<label>-<timestamp>.json)")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func newBenchDoCleanupCmd(f *benchDoFlags) *cobra.Command {
|
||||
admin := &adminFlags{}
|
||||
func newCleanupCmd(f *cmdutil.Factory, fl *flags) *cobra.Command {
|
||||
admin := &cmdutil.AdminFlags{}
|
||||
var force bool
|
||||
cmd := &cobra.Command{
|
||||
Use: "cleanup",
|
||||
Short: "Forget and prune every bench-wide snapshot (run before undeploy)",
|
||||
Long: "Runs restic forget+prune for every client repository, from its droplet —\n" +
|
||||
Short: "Forget and prune every fleet-bench snapshot (run before undeploy)",
|
||||
Long: "Runs restic forget+prune for every client repository, from its host —\n" +
|
||||
"the repos themselves persist (admin-api deletion is unimplemented) but end\n" +
|
||||
"~empty. Needs the fleet still deployed and a `yuctl login` session.",
|
||||
RunE: func(cmd *cobra.Command, _ []string) error {
|
||||
ctx := cmd.Context()
|
||||
s, _, err := f.session(ctx)
|
||||
s, _, err := fl.session(ctx, f)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
minter, err := f.minter(ctx, cmd, admin)
|
||||
minter, err := fl.minter(ctx, f, admin)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
@@ -359,17 +329,17 @@ func newBenchDoCleanupCmd(f *benchDoFlags) *cobra.Command {
|
||||
},
|
||||
}
|
||||
cmd.Flags().BoolVar(&force, "force", false, "clean even while load is running")
|
||||
admin.register(cmd)
|
||||
admin.Register(cmd)
|
||||
return cmd
|
||||
}
|
||||
|
||||
func newBenchDoUndeployCmd(f *benchDoFlags) *cobra.Command {
|
||||
func newUndeployCmd(f *cmdutil.Factory, fl *flags) *cobra.Command {
|
||||
var force bool
|
||||
cmd := &cobra.Command{
|
||||
Use: "undeploy",
|
||||
Short: "Destroy every fleet droplet and the ephemeral ssh key",
|
||||
Short: "Destroy every fleet host and the ephemeral ssh key",
|
||||
RunE: func(cmd *cobra.Command, _ []string) error {
|
||||
s, _, err := f.session(cmd.Context())
|
||||
s, _, err := fl.session(cmd.Context(), f)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
@@ -0,0 +1,163 @@
|
||||
package fleetbench
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"yuctl/fleet"
|
||||
fbfleet "yuctl/fleet/fleetbench"
|
||||
"yuctl/resticbench"
|
||||
"yuctl/ui"
|
||||
)
|
||||
|
||||
// view renders StatusReports as a styled dashboard; watch mode feeds it
|
||||
// successive samples and it keeps a TX history for the sparkline.
|
||||
type view struct {
|
||||
label string
|
||||
history fleet.History
|
||||
}
|
||||
|
||||
func (v *view) render(r *fbfleet.StatusReport, sampledAt time.Time, sampleSec int, watching bool) string {
|
||||
var b strings.Builder
|
||||
|
||||
b.WriteString(ui.Badge.Render("FLEET-BENCH") + " " + ui.Title.Render(v.label) + "\n")
|
||||
|
||||
if len(r.Hosts) == 0 {
|
||||
b.WriteString(ui.Warn.Render("no hosts deployed") +
|
||||
ui.Muted.Render(" — yuctl tools fleet-bench deploy") + "\n")
|
||||
return ui.Frame.Render(strings.TrimRight(b.String(), "\n"))
|
||||
}
|
||||
|
||||
if r.Run != nil {
|
||||
since := time.Since(r.Run.StartedAt).Round(time.Second).String() + " ago"
|
||||
p := r.Run.Params
|
||||
b.WriteString(fmt.Sprintf("%s %s · %s objects · %s/cycle · %s clients/host · %s\n",
|
||||
ui.OK.Render("● "+r.Run.Label),
|
||||
ui.Muted.Render("started "+since),
|
||||
p["obj_size_mib"]+"MiB", p["cycle_size"], p["clients_per_host"], p["duration"]))
|
||||
b.WriteString(ui.Muted.Render("cap "+p["cap_per_host"]+" per host") + "\n")
|
||||
} else {
|
||||
b.WriteString(ui.Warn.Render("○ no active run recorded") +
|
||||
ui.Muted.Render(" — load stopped or never started") + "\n")
|
||||
}
|
||||
b.WriteString("\n")
|
||||
|
||||
// Hosts: TX bar scaled to the busiest host, transfer budget bar scaled to
|
||||
// the cap.
|
||||
var maxBps float64
|
||||
nameW := 7
|
||||
for _, d := range r.Hosts {
|
||||
maxBps = max(maxBps, d.TxBps)
|
||||
nameW = max(nameW, len(d.Name))
|
||||
}
|
||||
var wire, capTotal int64
|
||||
var txBps float64
|
||||
for _, d := range r.Hosts {
|
||||
capBytes := r.TransferCap
|
||||
var used int64
|
||||
state := "-"
|
||||
if d.Status != nil {
|
||||
used = d.Status.WireTxBytes
|
||||
if d.Status.CapBytes > 0 {
|
||||
capBytes = d.Status.CapBytes
|
||||
}
|
||||
state = d.Status.State
|
||||
}
|
||||
budget := " "
|
||||
if capBytes > 0 {
|
||||
pct := float64(used) / float64(capBytes) * 100
|
||||
st := ui.OK
|
||||
switch {
|
||||
case pct >= 90:
|
||||
st = ui.Bad
|
||||
case pct >= 60:
|
||||
st = ui.Warn
|
||||
}
|
||||
budget = fmt.Sprintf("%s %s", st.Render(ui.Meter(float64(used), float64(capBytes), 10)), st.Render(fmt.Sprintf("%3.0f%%", pct)))
|
||||
}
|
||||
line := fmt.Sprintf("%-*s %s %s %s %s %s %s",
|
||||
nameW, d.Name,
|
||||
ui.Muted.Render(fmt.Sprintf("%-5s", d.Region)),
|
||||
ui.TX.Render(ui.Meter(d.TxBps, maxBps, 14)),
|
||||
ui.PadGbps(d.TxBps),
|
||||
budget,
|
||||
ui.Muted.Render(fmt.Sprintf("%9s", resticbench.FormatBytes(used))),
|
||||
stateCell(d, state))
|
||||
b.WriteString(line + "\n")
|
||||
txBps += d.TxBps
|
||||
wire += used
|
||||
capTotal += capBytes
|
||||
}
|
||||
b.WriteString(fmt.Sprintf("%s %s %s · %s\n",
|
||||
ui.Muted.Render(fmt.Sprintf("%-*s", nameW+6, "aggregate")),
|
||||
ui.TX.Render("TX"), ui.PadGbps(txBps),
|
||||
ui.Total.Render(fmt.Sprintf("pool %s / %s", resticbench.FormatBytes(wire), resticbench.FormatBytes(capTotal)))))
|
||||
if len(v.history.Values()) > 1 {
|
||||
b.WriteString(ui.Muted.Render("history ") + ui.Total.Render(ui.Sparkline(v.history.Values(), 40)) + "\n")
|
||||
}
|
||||
b.WriteString("\n")
|
||||
|
||||
// Clients: loop progress per restic identity.
|
||||
cnameW := 6
|
||||
for _, d := range r.Hosts {
|
||||
if d.Status == nil {
|
||||
continue
|
||||
}
|
||||
for _, c := range d.Status.Clients {
|
||||
cnameW = max(cnameW, len(fbfleet.ShortName(c.Name)))
|
||||
}
|
||||
}
|
||||
b.WriteString(ui.Muted.Render(fmt.Sprintf("%-*s %-9s %6s %10s %6s %s", cnameW, "CLIENT", "PHASE", "CYCLES", "UPLOADED", "ERR", "LAST ERROR")) + "\n")
|
||||
for _, d := range r.Hosts {
|
||||
if d.Status == nil {
|
||||
if d.Err != "" {
|
||||
b.WriteString(fmt.Sprintf("%-*s %s\n", cnameW, fbfleet.ShortName(d.Name), ui.Bad.Render(d.Err)))
|
||||
}
|
||||
continue
|
||||
}
|
||||
for _, c := range d.Status.Clients {
|
||||
lastErr := ""
|
||||
if c.LastError != "" {
|
||||
lastErr = ui.Bad.Render(ui.Truncate(c.LastError, 48))
|
||||
}
|
||||
b.WriteString(fmt.Sprintf("%-*s %-9s %6d %10s %s %s\n",
|
||||
cnameW, fbfleet.ShortName(c.Name), phaseCell(c.Phase), c.Cycles,
|
||||
resticbench.FormatBytes(c.Uploaded+c.CurrentBytes), ui.ErrCell(c.Errors, 6), lastErr))
|
||||
}
|
||||
}
|
||||
|
||||
b.WriteString("\n" + ui.Muted.Render(fleet.Footer(sampleSec, sampledAt, watching)))
|
||||
|
||||
return ui.Frame.Render(b.String())
|
||||
}
|
||||
|
||||
func phaseCell(phase string) string {
|
||||
switch phase {
|
||||
case "backup":
|
||||
return ui.OK.Render(fmt.Sprintf("%-9s", phase))
|
||||
case "generate":
|
||||
return ui.TX.Render(fmt.Sprintf("%-9s", phase))
|
||||
default:
|
||||
return ui.Muted.Render(fmt.Sprintf("%-9s", phase))
|
||||
}
|
||||
}
|
||||
|
||||
// stateCell colors the host's agent state, flagging a dead agent while a run
|
||||
// is recorded.
|
||||
func stateCell(d fbfleet.HostStatus, state string) string {
|
||||
switch {
|
||||
case !d.Reachable:
|
||||
return ui.Bad.Render("unreachable")
|
||||
case state == "running" && d.AgentProcs > 0:
|
||||
return ui.OK.Render("running")
|
||||
case state == "capped":
|
||||
return ui.Warn.Render("capped")
|
||||
case state == "done":
|
||||
return ui.Muted.Render("done")
|
||||
case d.AgentProcs == 0 && state == "running":
|
||||
return ui.Bad.Render("agent dead")
|
||||
default:
|
||||
return ui.Muted.Render(state)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,19 @@
|
||||
package tools
|
||||
|
||||
import (
|
||||
"github.com/spf13/cobra"
|
||||
|
||||
"yuctl/cli/tools/bench"
|
||||
"yuctl/cli/tools/fleetbench"
|
||||
"yuctl/cli/tools/warp"
|
||||
"yuctl/cmdutil"
|
||||
)
|
||||
|
||||
func New(f *cmdutil.Factory) *cobra.Command {
|
||||
tools := &cobra.Command{
|
||||
Use: "tools",
|
||||
Short: "Operational tooling for the selected context",
|
||||
}
|
||||
tools.AddCommand(bench.New(f), fleetbench.New(f), warp.New(f))
|
||||
return tools
|
||||
}
|
||||
@@ -0,0 +1,114 @@
|
||||
package warp
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"yuctl/fleet"
|
||||
warpfleet "yuctl/fleet/warp"
|
||||
"yuctl/ui"
|
||||
)
|
||||
|
||||
// view renders StatusReports as a styled dashboard; watch mode feeds it
|
||||
// successive samples and it keeps a throughput history for the sparkline.
|
||||
type view struct {
|
||||
label string
|
||||
history fleet.History
|
||||
}
|
||||
|
||||
func (v *view) render(r *warpfleet.StatusReport, sampledAt time.Time, sampleSec int, watching bool) string {
|
||||
var b strings.Builder
|
||||
|
||||
b.WriteString(ui.Badge.Render("WARP") + " " + ui.Title.Render(v.label) + "\n")
|
||||
|
||||
if len(r.Pods) == 0 {
|
||||
b.WriteString(ui.Warn.Render("no runner pods deployed") +
|
||||
ui.Muted.Render(" — yuctl tools warp deploy") + "\n")
|
||||
return ui.Frame.Render(strings.TrimRight(b.String(), "\n"))
|
||||
}
|
||||
|
||||
if r.Config != nil {
|
||||
since := r.Config["started_at"]
|
||||
if t, err := time.Parse(time.RFC3339, since); err == nil {
|
||||
since = time.Since(t).Round(time.Second).String() + " ago"
|
||||
}
|
||||
b.WriteString(fmt.Sprintf("%s %s · %s PUT / %s GET · %s/%s objects · %s RGWs\n",
|
||||
ui.OK.Render("● "+r.Config["mode"]),
|
||||
ui.Muted.Render("started "+since),
|
||||
r.Config["put_streams"], r.Config["get_streams"],
|
||||
r.Config["put_obj_size"], r.Config["get_obj_size"],
|
||||
r.Config["rgw_endpoints"]))
|
||||
b.WriteString(ui.Muted.Render("endpoint "+r.Config["endpoint"]) + "\n")
|
||||
} else {
|
||||
b.WriteString(ui.Warn.Render("○ no active run recorded") +
|
||||
ui.Muted.Render(" — load stopped or never started") + "\n")
|
||||
}
|
||||
b.WriteString("\n")
|
||||
|
||||
// Nodes: TX/RX bars scaled to the busiest direction in this frame.
|
||||
var maxBps float64
|
||||
for _, n := range r.Nodes {
|
||||
maxBps = max(maxBps, max(n.TxBps, n.RxBps))
|
||||
}
|
||||
nodeW, ifaceW := 4, 5
|
||||
for _, n := range r.Nodes {
|
||||
nodeW = max(nodeW, len(n.Node))
|
||||
ifaceW = max(ifaceW, len(n.Iface))
|
||||
}
|
||||
var tx, rx float64
|
||||
for _, n := range r.Nodes {
|
||||
b.WriteString(fmt.Sprintf("%-*s %s %s %s %s %s\n",
|
||||
nodeW, n.Node,
|
||||
ui.Muted.Render(fmt.Sprintf("%-*s", ifaceW, n.Iface)),
|
||||
ui.TX.Render(ui.Meter(n.TxBps, maxBps, 14)),
|
||||
ui.PadGbps(n.TxBps),
|
||||
ui.RX.Render(ui.Meter(n.RxBps, maxBps, 14)),
|
||||
ui.PadGbps(n.RxBps)))
|
||||
tx += n.TxBps
|
||||
rx += n.RxBps
|
||||
}
|
||||
if len(r.Nodes) > 0 {
|
||||
b.WriteString(fmt.Sprintf("%s %s %s · %s %s · %s\n",
|
||||
ui.Muted.Render(fmt.Sprintf("%-*s", nodeW+ifaceW-2, "aggregate")),
|
||||
ui.TX.Render("TX"), ui.PadGbps(tx),
|
||||
ui.RX.Render("RX"), ui.PadGbps(rx),
|
||||
ui.Total.Render("combined "+ui.FmtGbps(tx+rx))))
|
||||
if len(v.history.Values()) > 1 {
|
||||
b.WriteString(ui.Muted.Render("history ") + ui.Total.Render(ui.Sparkline(v.history.Values(), 40)) + "\n")
|
||||
}
|
||||
b.WriteString("\n")
|
||||
}
|
||||
|
||||
// Pods: process liveness and log error counts.
|
||||
podW := 3
|
||||
for _, p := range r.Pods {
|
||||
podW = max(podW, len(p.Name))
|
||||
}
|
||||
b.WriteString(ui.Muted.Render(fmt.Sprintf("%-*s %-*s %6s %6s %10s %10s", podW, "POD", nodeW, "NODE", "PUT", "GET", "ERR(PUT)", "ERR(GET)")) + "\n")
|
||||
for _, p := range r.Pods {
|
||||
b.WriteString(fmt.Sprintf("%-*s %-*s %s %s %s %s\n",
|
||||
podW, p.Name, nodeW, p.Node,
|
||||
procCell(p.PutProcs, r.Config != nil),
|
||||
procCell(p.GetProcs, r.Config != nil),
|
||||
ui.ErrCell(p.PutErrors, 10), ui.ErrCell(p.GetErrors, 10)))
|
||||
}
|
||||
|
||||
b.WriteString("\n" + ui.Muted.Render(fleet.Footer(sampleSec, sampledAt, watching)))
|
||||
|
||||
return ui.Frame.Render(b.String())
|
||||
}
|
||||
|
||||
// procCell colors a warp process count: green when alive, red when a run is
|
||||
// recorded but nothing is running, dim otherwise.
|
||||
func procCell(n int, runRecorded bool) string {
|
||||
s := fmt.Sprintf("%6d", n)
|
||||
switch {
|
||||
case n > 0:
|
||||
return ui.OK.Render(s)
|
||||
case runRecorded:
|
||||
return ui.Bad.Render(s)
|
||||
default:
|
||||
return ui.Muted.Render(s)
|
||||
}
|
||||
}
|
||||
+10
-22
@@ -1,33 +1,33 @@
|
||||
package cli
|
||||
package warp
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"yuctl/internal/warp"
|
||||
warpfleet "yuctl/fleet/warp"
|
||||
)
|
||||
|
||||
func TestWarpViewRender(t *testing.T) {
|
||||
v := &warpView{label: "prod@htz-fsn1 → spice · father"}
|
||||
r := &warp.StatusReport{
|
||||
func TestViewRender(t *testing.T) {
|
||||
v := &view{label: "prod@htz-fsn1 → spice · father"}
|
||||
r := &warpfleet.StatusReport{
|
||||
Config: map[string]string{
|
||||
"mode": "nonstop", "started_at": time.Now().Add(-2 * time.Hour).UTC().Format(time.RFC3339),
|
||||
"put_streams": "1002", "get_streams": "102",
|
||||
"put_obj_size": "16MiB", "get_obj_size": "16MiB",
|
||||
"rgw_endpoints": "47", "endpoint": "https://s3.example",
|
||||
},
|
||||
Pods: []warp.PodStatus{
|
||||
Pods: []warpfleet.PodStatus{
|
||||
{Name: "warp-runner-a", Node: "jeanne", PutProcs: 1, GetProcs: 1},
|
||||
{Name: "warp-runner-b", Node: "sheron", PutProcs: 0, GetProcs: 1, PutErrors: 3},
|
||||
},
|
||||
Nodes: []warp.NodeThroughput{
|
||||
Nodes: []warpfleet.NodeThroughput{
|
||||
{Node: "jeanne", Iface: "bond0", TxBps: 49e9, RxBps: 31e9},
|
||||
{Node: "sheron", Iface: "bond0", TxBps: 50e9, RxBps: 35e9},
|
||||
},
|
||||
}
|
||||
v.push(160e9)
|
||||
v.push(165e9)
|
||||
v.history.Push(160e9)
|
||||
v.history.Push(165e9)
|
||||
out := v.render(r, time.Now(), 5, true)
|
||||
for _, want := range []string{"WARP", "nonstop", "jeanne", "bond0", "combined", "ctrl-c"} {
|
||||
if !strings.Contains(out, want) {
|
||||
@@ -35,20 +35,8 @@ func TestWarpViewRender(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
empty := (&warpView{label: "x"}).render(&warp.StatusReport{}, time.Now(), 5, false)
|
||||
empty := (&view{label: "x"}).render(&warpfleet.StatusReport{}, time.Now(), 5, false)
|
||||
if !strings.Contains(empty, "no runner pods deployed") {
|
||||
t.Errorf("empty render: %s", empty)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSparklineAndMeter(t *testing.T) {
|
||||
if got := sparkline([]float64{0, 1, 2, 4}, 10); len([]rune(got)) != 4 {
|
||||
t.Errorf("sparkline length: %q", got)
|
||||
}
|
||||
if got := meter(50, 100, 10); !strings.HasPrefix(got, "█████░") {
|
||||
t.Errorf("meter: %q", got)
|
||||
}
|
||||
if got := meter(0, 0, 4); got != "░░░░" {
|
||||
t.Errorf("empty meter: %q", got)
|
||||
}
|
||||
}
|
||||
@@ -1,23 +1,22 @@
|
||||
package cli
|
||||
package warp
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"os"
|
||||
"os/signal"
|
||||
"strings"
|
||||
"syscall"
|
||||
"time"
|
||||
|
||||
"github.com/rs/zerolog/log"
|
||||
"github.com/spf13/cobra"
|
||||
|
||||
"yuctl/internal/warp"
|
||||
"yuctl/cmdutil"
|
||||
"yuctl/fleet"
|
||||
warpfleet "yuctl/fleet/warp"
|
||||
)
|
||||
|
||||
// warpFlags are shared by every warp subcommand: how to reach the cluster and
|
||||
// flags are shared by every warp subcommand: how to reach the cluster and
|
||||
// which ceph cluster's RGW fleet to target. Everything else is derived.
|
||||
type warpFlags struct {
|
||||
type flags struct {
|
||||
namespace string
|
||||
kubeconfig string
|
||||
cephCluster string
|
||||
@@ -25,23 +24,23 @@ type warpFlags struct {
|
||||
insecure bool
|
||||
}
|
||||
|
||||
func (f *warpFlags) register(c *cobra.Command) {
|
||||
c.PersistentFlags().StringVar(&f.namespace, "namespace", "loadtest", "namespace for the runner fleet")
|
||||
c.PersistentFlags().StringVar(&f.kubeconfig, "kubeconfig", "", "kubeconfig path (default: materialized from discovery via 1Password)")
|
||||
c.PersistentFlags().StringVar(&f.cephCluster, "ceph-cluster", "", "target ceph cluster (default: the selected one, or the region's only one)")
|
||||
c.PersistentFlags().StringVar(&f.s3Endpoint, "s3-endpoint", "", "override the RGW S3 endpoint from discovery")
|
||||
c.PersistentFlags().BoolVar(&f.insecure, "insecure", true, "skip RGW TLS verification (self-signed certs)")
|
||||
func (o *flags) register(c *cobra.Command) {
|
||||
c.PersistentFlags().StringVar(&o.namespace, "namespace", "loadtest", "namespace for the runner fleet")
|
||||
c.PersistentFlags().StringVar(&o.kubeconfig, "kubeconfig", "", "kubeconfig path (default: materialized from discovery via 1Password)")
|
||||
c.PersistentFlags().StringVar(&o.cephCluster, "ceph-cluster", "", "target ceph cluster (default: the selected one, or the region's only one)")
|
||||
c.PersistentFlags().StringVar(&o.s3Endpoint, "s3-endpoint", "", "override the RGW S3 endpoint from discovery")
|
||||
c.PersistentFlags().BoolVar(&o.insecure, "insecure", true, "skip RGW TLS verification (self-signed certs)")
|
||||
}
|
||||
|
||||
// session resolves context + topology and opens a warp session against the
|
||||
// region's K8s cluster and chosen ceph cluster. The label describes the target
|
||||
// for display ("prod@htz-fsn1 → spice · father").
|
||||
func (f *warpFlags) session(ctx context.Context) (*warp.Session, string, error) {
|
||||
cc, err := requireContext()
|
||||
func (o *flags) session(ctx context.Context, f *cmdutil.Factory) (*warpfleet.Session, string, error) {
|
||||
cc, err := f.Context()
|
||||
if err != nil {
|
||||
return nil, "", err
|
||||
}
|
||||
topo, err := resolveTopology(ctx)
|
||||
topo, err := f.Topology(ctx)
|
||||
if err != nil {
|
||||
return nil, "", err
|
||||
}
|
||||
@@ -50,14 +49,14 @@ func (f *warpFlags) session(ctx context.Context) (*warp.Session, string, error)
|
||||
return nil, "", fmt.Errorf("no kubernetes payload in discovery for %s@%s", cc.Partition, cc.Region)
|
||||
}
|
||||
clusters := topo.CephClusters(cc.Partition, cc.Region)
|
||||
name := f.cephCluster
|
||||
name := o.cephCluster
|
||||
if name == "" {
|
||||
name = cc.CephCluster
|
||||
}
|
||||
if name == "" {
|
||||
if len(clusters) != 1 {
|
||||
return nil, "", fmt.Errorf("region has %d ceph clusters (%s); `yuctl ceph select` or --ceph-cluster",
|
||||
len(clusters), strings.Join(cephClusterNames(clusters), ", "))
|
||||
len(clusters), strings.Join(topo.CephClusterNames(cc.Partition, cc.Region), ", "))
|
||||
}
|
||||
for n := range clusters {
|
||||
name = n
|
||||
@@ -66,15 +65,15 @@ func (f *warpFlags) session(ctx context.Context) (*warp.Session, string, error)
|
||||
ceph, ok := clusters[name]
|
||||
if !ok {
|
||||
return nil, "", fmt.Errorf("unknown ceph cluster %q in %s@%s; known: %s",
|
||||
name, cc.Partition, cc.Region, strings.Join(cephClusterNames(clusters), ", "))
|
||||
name, cc.Partition, cc.Region, strings.Join(topo.CephClusterNames(cc.Partition, cc.Region), ", "))
|
||||
}
|
||||
label := fmt.Sprintf("%s@%s → %s · %s", cc.Partition, cc.Region, name, k8s.ClusterName)
|
||||
s, err := warp.NewSession(ctx, *k8s, ceph, f.namespace, f.kubeconfig)
|
||||
s, err := warpfleet.NewSession(ctx, *k8s, ceph, o.namespace, o.kubeconfig)
|
||||
return s, label, err
|
||||
}
|
||||
|
||||
func newWarpCmd() *cobra.Command {
|
||||
f := &warpFlags{}
|
||||
func New(f *cmdutil.Factory) *cobra.Command {
|
||||
o := &flags{}
|
||||
cmd := &cobra.Command{
|
||||
Use: "warp",
|
||||
Short: "S3 load testing against the region's RGW fleet (MinIO warp)",
|
||||
@@ -84,20 +83,20 @@ func newWarpCmd() *cobra.Command {
|
||||
"(workers, CPU, RGW IPs, replica count) is discovered at runtime; defaults\n" +
|
||||
"reproduce the proven ~250Gbps father configuration per pod.",
|
||||
}
|
||||
f.register(cmd)
|
||||
o.register(cmd)
|
||||
cmd.AddCommand(
|
||||
newWarpDeployCmd(f),
|
||||
newWarpStartCmd(f),
|
||||
newWarpStatusCmd(f),
|
||||
newWarpWatchCmd(f),
|
||||
newWarpStopCmd(f),
|
||||
newWarpCleanupCmd(f),
|
||||
newWarpUndeployCmd(f),
|
||||
newDeployCmd(f, o),
|
||||
newStartCmd(f, o),
|
||||
newStatusCmd(f, o),
|
||||
newWatchCmd(f, o),
|
||||
newStopCmd(f, o),
|
||||
newCleanupCmd(f, o),
|
||||
newUndeployCmd(f, o),
|
||||
)
|
||||
return cmd
|
||||
}
|
||||
|
||||
func warpDeployFlags(c *cobra.Command, o *warp.DeployOptions) {
|
||||
func deployFlags(c *cobra.Command, o *warpfleet.DeployOptions) {
|
||||
c.Flags().StringVar(&o.Image, "image", "minio/warp:latest", "warp container image")
|
||||
c.Flags().IntVar(&o.PodsPerNode, "pods-per-node", 2, "runner pods per worker node")
|
||||
c.Flags().IntVar(&o.WorkerCount, "workers", 0, "use only the first N workers (0 = all)")
|
||||
@@ -106,26 +105,26 @@ func warpDeployFlags(c *cobra.Command, o *warp.DeployOptions) {
|
||||
c.Flags().StringVar(&o.SourceSecret, "creds-secret", "yucca-michael", "secret holding the RGW access/secret keys")
|
||||
}
|
||||
|
||||
func newWarpDeployCmd(f *warpFlags) *cobra.Command {
|
||||
o := warp.DeployOptions{}
|
||||
func newDeployCmd(f *cmdutil.Factory, fl *flags) *cobra.Command {
|
||||
o := warpfleet.DeployOptions{}
|
||||
cmd := &cobra.Command{
|
||||
Use: "deploy",
|
||||
Short: "Deploy (or converge) the runner fleet on the region's workers",
|
||||
RunE: func(cmd *cobra.Command, _ []string) error {
|
||||
s, _, err := f.session(cmd.Context())
|
||||
s, _, err := fl.session(cmd.Context(), f)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
return s.Deploy(cmd.Context(), o)
|
||||
},
|
||||
}
|
||||
warpDeployFlags(cmd, &o)
|
||||
deployFlags(cmd, &o)
|
||||
return cmd
|
||||
}
|
||||
|
||||
func newWarpStartCmd(f *warpFlags) *cobra.Command {
|
||||
o := warp.StartOptions{}
|
||||
d := warp.DeployOptions{}
|
||||
func newStartCmd(f *cmdutil.Factory, fl *flags) *cobra.Command {
|
||||
o := warpfleet.StartOptions{}
|
||||
d := warpfleet.DeployOptions{}
|
||||
var autoDeploy bool
|
||||
cmd := &cobra.Command{
|
||||
Use: "start",
|
||||
@@ -138,12 +137,12 @@ func newWarpStartCmd(f *warpFlags) *cobra.Command {
|
||||
"that is the proven 1002/102 ~250Gbps shape.",
|
||||
RunE: func(cmd *cobra.Command, _ []string) error {
|
||||
ctx := cmd.Context()
|
||||
s, _, err := f.session(ctx)
|
||||
s, _, err := fl.session(ctx, f)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
o.Insecure = f.insecure
|
||||
o.S3Override = f.s3Endpoint
|
||||
o.Insecure = fl.insecure
|
||||
o.S3Override = fl.s3Endpoint
|
||||
if autoDeploy {
|
||||
if pods, err := s.RunnerPods(ctx); err == nil && len(pods) == 0 {
|
||||
if err := s.Deploy(ctx, d); err != nil {
|
||||
@@ -165,17 +164,17 @@ func newWarpStartCmd(f *warpFlags) *cobra.Command {
|
||||
cmd.Flags().StringVar(&o.CycleDuration, "cycle", "6h", "warp run length inside the non-stop loop")
|
||||
cmd.Flags().StringVar(&o.BucketPrefix, "bucket-prefix", "yuctl-warp-", "bucket name prefix (cleanup purges by this)")
|
||||
cmd.Flags().BoolVar(&autoDeploy, "auto-deploy", true, "deploy the runner fleet first if it is missing")
|
||||
warpDeployFlags(cmd, &d)
|
||||
deployFlags(cmd, &d)
|
||||
return cmd
|
||||
}
|
||||
|
||||
func newWarpStatusCmd(f *warpFlags) *cobra.Command {
|
||||
func newStatusCmd(f *cmdutil.Factory, fl *flags) *cobra.Command {
|
||||
var sample int
|
||||
cmd := &cobra.Command{
|
||||
Use: "status",
|
||||
Short: "Show per-pod load state and per-node NIC throughput",
|
||||
RunE: func(cmd *cobra.Command, _ []string) error {
|
||||
s, label, err := f.session(cmd.Context())
|
||||
s, label, err := fl.session(cmd.Context(), f)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
@@ -184,8 +183,8 @@ func newWarpStatusCmd(f *warpFlags) *cobra.Command {
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
v := &warpView{label: label}
|
||||
fmt.Fprintln(cmd.OutOrStdout(), v.render(report, time.Now(), sample, false))
|
||||
v := &view{label: label}
|
||||
fmt.Fprintln(f.IO.Out, v.render(report, time.Now(), sample, false))
|
||||
return nil
|
||||
},
|
||||
}
|
||||
@@ -193,60 +192,43 @@ func newWarpStatusCmd(f *warpFlags) *cobra.Command {
|
||||
return cmd
|
||||
}
|
||||
|
||||
func newWarpWatchCmd(f *warpFlags) *cobra.Command {
|
||||
func newWatchCmd(f *cmdutil.Factory, fl *flags) *cobra.Command {
|
||||
var sample int
|
||||
cmd := &cobra.Command{
|
||||
Use: "watch",
|
||||
Short: "Live dashboard: continuously sampled load state and throughput",
|
||||
RunE: func(cmd *cobra.Command, _ []string) error {
|
||||
ctx, stop := signal.NotifyContext(cmd.Context(), os.Interrupt, syscall.SIGTERM)
|
||||
defer stop()
|
||||
s, label, err := f.session(ctx)
|
||||
s, label, err := fl.session(cmd.Context(), f)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
out := cmd.OutOrStdout()
|
||||
fmt.Fprint(out, "\x1b[?1049h\x1b[?25l") // alt screen, hide cursor
|
||||
defer fmt.Fprint(out, "\x1b[?25h\x1b[?1049l")
|
||||
fmt.Fprintf(out, "%s connecting · sampling %ds window…\n", label, sample)
|
||||
|
||||
v := &warpView{label: label}
|
||||
for {
|
||||
v := &view{label: label}
|
||||
return fleet.Watch(cmd.Context(), f.IO.Out, label, sample, func(ctx context.Context) (string, error) {
|
||||
report, err := s.Status(ctx, sample)
|
||||
if ctx.Err() != nil {
|
||||
return nil
|
||||
}
|
||||
fmt.Fprint(out, "\x1b[H\x1b[2J")
|
||||
if err != nil {
|
||||
fmt.Fprintf(out, "status error (retrying): %v\n", err)
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return nil
|
||||
case <-time.After(3 * time.Second):
|
||||
}
|
||||
continue
|
||||
return "", err
|
||||
}
|
||||
var combined float64
|
||||
for _, n := range report.Nodes {
|
||||
combined += n.TxBps + n.RxBps
|
||||
}
|
||||
if len(report.Nodes) > 0 {
|
||||
v.push(combined)
|
||||
v.history.Push(combined)
|
||||
}
|
||||
fmt.Fprintln(out, v.render(report, time.Now(), sample, true))
|
||||
}
|
||||
return v.render(report, time.Now(), sample, true), nil
|
||||
})
|
||||
},
|
||||
}
|
||||
cmd.Flags().IntVar(&sample, "sample", 5, "NIC sampling window per refresh in seconds")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func newWarpStopCmd(f *warpFlags) *cobra.Command {
|
||||
func newStopCmd(f *cmdutil.Factory, fl *flags) *cobra.Command {
|
||||
return &cobra.Command{
|
||||
Use: "stop",
|
||||
Short: "Stop the load everywhere (runners stay deployed for instant restart)",
|
||||
RunE: func(cmd *cobra.Command, _ []string) error {
|
||||
s, _, err := f.session(cmd.Context())
|
||||
s, _, err := fl.session(cmd.Context(), f)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
@@ -255,8 +237,8 @@ func newWarpStopCmd(f *warpFlags) *cobra.Command {
|
||||
}
|
||||
}
|
||||
|
||||
func newWarpCleanupCmd(f *warpFlags) *cobra.Command {
|
||||
o := warp.CleanupOptions{}
|
||||
func newCleanupCmd(f *cmdutil.Factory, fl *flags) *cobra.Command {
|
||||
o := warpfleet.CleanupOptions{}
|
||||
var timeout time.Duration
|
||||
cmd := &cobra.Command{
|
||||
Use: "cleanup",
|
||||
@@ -266,13 +248,14 @@ func newWarpCleanupCmd(f *warpFlags) *cobra.Command {
|
||||
"while load is active unless --force. Note: a mass delete is itself a load\n" +
|
||||
"event — RGW garbage collection churns for a while afterwards.",
|
||||
RunE: func(cmd *cobra.Command, _ []string) error {
|
||||
s, _, err := f.session(cmd.Context())
|
||||
s, _, err := fl.session(cmd.Context(), f)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
o.Insecure = f.insecure
|
||||
o.S3Override = f.s3Endpoint
|
||||
o.Insecure = fl.insecure
|
||||
o.S3Override = fl.s3Endpoint
|
||||
o.Timeout = timeout
|
||||
o.JobLogs = f.IO.Out
|
||||
return s.Cleanup(cmd.Context(), o)
|
||||
},
|
||||
}
|
||||
@@ -285,13 +268,13 @@ func newWarpCleanupCmd(f *warpFlags) *cobra.Command {
|
||||
return cmd
|
||||
}
|
||||
|
||||
func newWarpUndeployCmd(f *warpFlags) *cobra.Command {
|
||||
func newUndeployCmd(f *cmdutil.Factory, fl *flags) *cobra.Command {
|
||||
var force bool
|
||||
cmd := &cobra.Command{
|
||||
Use: "undeploy",
|
||||
Short: "Delete the runner fleet and its namespace entirely",
|
||||
RunE: func(cmd *cobra.Command, _ []string) error {
|
||||
s, _, err := f.session(cmd.Context())
|
||||
s, _, err := fl.session(cmd.Context(), f)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
@@ -0,0 +1,176 @@
|
||||
package allowlist
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strconv"
|
||||
"strings"
|
||||
"text/tabwriter"
|
||||
|
||||
"github.com/spf13/cobra"
|
||||
|
||||
"yuctl/adminapi"
|
||||
"yuctl/cmdutil"
|
||||
)
|
||||
|
||||
func New(f *cmdutil.Factory) *cobra.Command {
|
||||
cmd := &cobra.Command{
|
||||
Use: "allowlist",
|
||||
Short: "Manage the beta email allowlist and invites",
|
||||
}
|
||||
cmd.AddCommand(
|
||||
newListCmd(f),
|
||||
newAddCmd(f),
|
||||
newRemoveCmd(f),
|
||||
newInviteCmd(f),
|
||||
newInviteBatchCmd(f),
|
||||
)
|
||||
return cmd
|
||||
}
|
||||
|
||||
func printEntries(f *cmdutil.Factory, entries []adminapi.AllowlistEntry) {
|
||||
w := tabwriter.NewWriter(f.IO.Out, 0, 2, 2, ' ', 0)
|
||||
fmt.Fprintln(w, "EMAIL\tCODE\tINVITED\tUSED\tUSED AT\tCREATED")
|
||||
for _, e := range entries {
|
||||
usedAt := ""
|
||||
if e.InviteUsedAt != nil {
|
||||
usedAt = *e.InviteUsedAt
|
||||
}
|
||||
fmt.Fprintf(w, "%s\t%s\t%t\t%t\t%s\t%s\n", e.Email, e.InviteCode, e.Invited, e.InviteUsed, usedAt, e.CreatedAt)
|
||||
}
|
||||
w.Flush()
|
||||
}
|
||||
|
||||
func newListCmd(f *cmdutil.Factory) *cobra.Command {
|
||||
admin := &cmdutil.AdminFlags{}
|
||||
var limitFlag string
|
||||
c := &cobra.Command{
|
||||
Use: "list",
|
||||
Short: "List allowlist entries",
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
limit, err := adminapi.ParseLimit(limitFlag)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
client, partition, err := admin.Client(cmd.Context(), f)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
entries, err := client.ListAllowlist(cmd.Context(), limit)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
printEntries(f, entries)
|
||||
fmt.Fprintf(f.IO.Err, "\n%d entries in partition %s\n", len(entries), partition)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
c.Flags().StringVar(&limitFlag, "limit", "", "page size for the admin-api (default: server default)")
|
||||
admin.Register(c)
|
||||
return c
|
||||
}
|
||||
|
||||
func newAddCmd(f *cmdutil.Factory) *cobra.Command {
|
||||
admin := &cmdutil.AdminFlags{}
|
||||
var staged bool
|
||||
c := &cobra.Command{
|
||||
Use: "add <email>",
|
||||
Short: "Allow an email to sign up (--staged to waitlist it instead)",
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
client, _, err := admin.Client(cmd.Context(), f)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
entry, err := client.AddAllowlistEntry(cmd.Context(), args[0], staged)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
printEntries(f, []adminapi.AllowlistEntry{*entry})
|
||||
return nil
|
||||
},
|
||||
}
|
||||
c.Flags().BoolVar(&staged, "staged", false, "stage the email without allowing login yet")
|
||||
admin.Register(c)
|
||||
return c
|
||||
}
|
||||
|
||||
func newRemoveCmd(f *cmdutil.Factory) *cobra.Command {
|
||||
admin := &cmdutil.AdminFlags{}
|
||||
c := &cobra.Command{
|
||||
Use: "remove <email>",
|
||||
Short: "Remove an email from the allowlist",
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
client, _, err := admin.Client(cmd.Context(), f)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := client.RemoveAllowlistEntry(cmd.Context(), args[0]); err != nil {
|
||||
return err
|
||||
}
|
||||
fmt.Fprintf(f.IO.Err, "removed %s\n", args[0])
|
||||
return nil
|
||||
},
|
||||
}
|
||||
admin.Register(c)
|
||||
return c
|
||||
}
|
||||
|
||||
func newInviteCmd(f *cmdutil.Factory) *cobra.Command {
|
||||
admin := &cmdutil.AdminFlags{}
|
||||
c := &cobra.Command{
|
||||
Use: "invite <email>[,<email>...]",
|
||||
Short: "Invite emails: allow them to sign up, creating entries as needed",
|
||||
Args: cobra.MinimumNArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
var emails []string
|
||||
for _, arg := range args {
|
||||
for _, email := range strings.Split(arg, ",") {
|
||||
if email = strings.TrimSpace(email); email != "" {
|
||||
emails = append(emails, email)
|
||||
}
|
||||
}
|
||||
}
|
||||
client, _, err := admin.Client(cmd.Context(), f)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
entries, err := client.InviteEmails(cmd.Context(), emails)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
printEntries(f, entries)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
admin.Register(c)
|
||||
return c
|
||||
}
|
||||
|
||||
func newInviteBatchCmd(f *cmdutil.Factory) *cobra.Command {
|
||||
admin := &cmdutil.AdminFlags{}
|
||||
c := &cobra.Command{
|
||||
Use: "invite-batch <count>",
|
||||
Short: "Invite the oldest <count> staged (waitlisted) emails",
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
count, err := strconv.Atoi(args[0])
|
||||
if err != nil || count < 1 {
|
||||
return fmt.Errorf("count must be a positive integer")
|
||||
}
|
||||
client, _, err := admin.Client(cmd.Context(), f)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
entries, err := client.InviteBatch(cmd.Context(), count)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
printEntries(f, entries)
|
||||
fmt.Fprintf(f.IO.Err, "\ninvited %d entries\n", len(entries))
|
||||
return nil
|
||||
},
|
||||
}
|
||||
admin.Register(c)
|
||||
return c
|
||||
}
|
||||
@@ -0,0 +1,57 @@
|
||||
package connections
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"text/tabwriter"
|
||||
|
||||
"github.com/spf13/cobra"
|
||||
|
||||
"yuctl/cmdutil"
|
||||
)
|
||||
|
||||
func New(f *cmdutil.Factory) *cobra.Command {
|
||||
cmd := &cobra.Command{
|
||||
Use: "connections",
|
||||
Short: "A user's connection instances (immich/restic)",
|
||||
}
|
||||
cmd.AddCommand(newListCmd(f))
|
||||
return cmd
|
||||
}
|
||||
|
||||
func newListCmd(f *cmdutil.Factory) *cobra.Command {
|
||||
admin := &cmdutil.AdminFlags{}
|
||||
c := &cobra.Command{
|
||||
Use: "list <email>",
|
||||
Short: "List a user's connections",
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
ctx := cmd.Context()
|
||||
client, _, err := admin.Client(ctx, f)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
userID, err := client.ResolveUserID(ctx, args[0])
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
connections, err := client.GetUserConnections(ctx, userID)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
w := tabwriter.NewWriter(f.IO.Out, 0, 2, 2, ' ', 0)
|
||||
fmt.Fprintln(w, "ID\tTYPE\tNAME\tCREATED\tLAST SEEN")
|
||||
for _, connection := range connections {
|
||||
lastSeen := ""
|
||||
if connection.LastSeenAt != nil {
|
||||
lastSeen = *connection.LastSeenAt
|
||||
}
|
||||
fmt.Fprintf(w, "%s\t%s\t%s\t%s\t%s\n", connection.ID, connection.Type, connection.Name, connection.CreatedAt, lastSeen)
|
||||
}
|
||||
w.Flush()
|
||||
return nil
|
||||
},
|
||||
}
|
||||
admin.Register(c)
|
||||
return c
|
||||
}
|
||||
@@ -0,0 +1,133 @@
|
||||
package features
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"text/tabwriter"
|
||||
|
||||
"github.com/spf13/cobra"
|
||||
|
||||
"yuctl/adminapi"
|
||||
"yuctl/cmdutil"
|
||||
)
|
||||
|
||||
func New(f *cmdutil.Factory) *cobra.Command {
|
||||
cmd := &cobra.Command{
|
||||
Use: "features",
|
||||
Short: "Per-user feature-flag overrides",
|
||||
}
|
||||
cmd.AddCommand(newListCmd(f), newSetCmd(f), newClearCmd(f))
|
||||
return cmd
|
||||
}
|
||||
|
||||
func newListCmd(f *cmdutil.Factory) *cobra.Command {
|
||||
admin := &cmdutil.AdminFlags{}
|
||||
c := &cobra.Command{
|
||||
Use: "list <email>",
|
||||
Short: "Show a user's resolved feature flags and overrides",
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
ctx := cmd.Context()
|
||||
client, _, err := admin.Client(ctx, f)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
userID, err := client.ResolveUserID(ctx, args[0])
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
features, err := client.GetUserFeatures(ctx, userID)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
overridden := map[string]adminapi.FeatureOverride{}
|
||||
for _, o := range features.Overrides {
|
||||
overridden[o.Flag] = o
|
||||
}
|
||||
|
||||
w := tabwriter.NewWriter(f.IO.Out, 0, 2, 2, ' ', 0)
|
||||
fmt.Fprintln(w, "FLAG\tVALUE\tSOURCE\tSET BY\tREASON")
|
||||
for flag, value := range features.Features {
|
||||
if o, ok := overridden[flag]; ok {
|
||||
reason := ""
|
||||
if o.Reason != nil {
|
||||
reason = *o.Reason
|
||||
}
|
||||
fmt.Fprintf(w, "%s\t%t\toverride\t%s\t%s\n", flag, value, o.SetBy, reason)
|
||||
} else {
|
||||
fmt.Fprintf(w, "%s\t%t\tdefault\t\t\n", flag, value)
|
||||
}
|
||||
}
|
||||
w.Flush()
|
||||
return nil
|
||||
},
|
||||
}
|
||||
admin.Register(c)
|
||||
return c
|
||||
}
|
||||
|
||||
func newSetCmd(f *cmdutil.Factory) *cobra.Command {
|
||||
admin := &cmdutil.AdminFlags{}
|
||||
var reason string
|
||||
c := &cobra.Command{
|
||||
Use: "set <email> <flag> on|off",
|
||||
Short: "Set a per-user feature-flag override",
|
||||
Args: cobra.ExactArgs(3),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
var value bool
|
||||
switch args[2] {
|
||||
case "on", "true":
|
||||
value = true
|
||||
case "off", "false":
|
||||
value = false
|
||||
default:
|
||||
return fmt.Errorf("value must be on|off, got %q", args[2])
|
||||
}
|
||||
|
||||
ctx := cmd.Context()
|
||||
client, _, err := admin.Client(ctx, f)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
userID, err := client.ResolveUserID(ctx, args[0])
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if _, err := client.SetUserFeature(ctx, userID, args[1], value, reason); err != nil {
|
||||
return err
|
||||
}
|
||||
fmt.Fprintf(f.IO.Err, "%s set to %t for %s\n", args[1], value, args[0])
|
||||
return nil
|
||||
},
|
||||
}
|
||||
c.Flags().StringVar(&reason, "reason", "", "audit note stored on the override")
|
||||
admin.Register(c)
|
||||
return c
|
||||
}
|
||||
|
||||
func newClearCmd(f *cmdutil.Factory) *cobra.Command {
|
||||
admin := &cmdutil.AdminFlags{}
|
||||
c := &cobra.Command{
|
||||
Use: "clear <email> <flag>",
|
||||
Short: "Clear an override (revert to the registry default)",
|
||||
Args: cobra.ExactArgs(2),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
ctx := cmd.Context()
|
||||
client, _, err := admin.Client(ctx, f)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
userID, err := client.ResolveUserID(ctx, args[0])
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := client.ClearUserFeature(ctx, userID, args[1]); err != nil {
|
||||
return err
|
||||
}
|
||||
fmt.Fprintf(f.IO.Err, "%s override cleared for %s\n", args[1], args[0])
|
||||
return nil
|
||||
},
|
||||
}
|
||||
admin.Register(c)
|
||||
return c
|
||||
}
|
||||
@@ -0,0 +1,124 @@
|
||||
package users
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"net/url"
|
||||
"os"
|
||||
"strings"
|
||||
"text/tabwriter"
|
||||
|
||||
"github.com/spf13/cobra"
|
||||
|
||||
"yuctl/adminapi"
|
||||
"yuctl/cli/users/allowlist"
|
||||
"yuctl/cli/users/connections"
|
||||
ufeatures "yuctl/cli/users/features"
|
||||
"yuctl/cmdutil"
|
||||
)
|
||||
|
||||
func New(f *cmdutil.Factory) *cobra.Command {
|
||||
cmd := &cobra.Command{
|
||||
Use: "users",
|
||||
Short: "User administration via yucca-admin-api",
|
||||
}
|
||||
cmd.AddCommand(
|
||||
newListCmd(f),
|
||||
allowlist.New(f),
|
||||
newViewDashboardCmd(f),
|
||||
ufeatures.New(f),
|
||||
connections.New(f),
|
||||
)
|
||||
return cmd
|
||||
}
|
||||
|
||||
func newListCmd(f *cmdutil.Factory) *cobra.Command {
|
||||
admin := &cmdutil.AdminFlags{}
|
||||
var limitFlag string
|
||||
c := &cobra.Command{
|
||||
Use: "list",
|
||||
Short: "List users in the partition's primary region",
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
ctx := cmd.Context()
|
||||
limit, err := adminapi.ParseLimit(limitFlag)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
client, partition, err := admin.Client(ctx, f)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
users, err := client.ListUsers(ctx, limit)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
w := tabwriter.NewWriter(f.IO.Out, 0, 2, 2, ' ', 0)
|
||||
fmt.Fprintln(w, "ID\tSUB\tNAME\tEMAIL\tDISABLED")
|
||||
for _, u := range users {
|
||||
fmt.Fprintf(w, "%s\t%s\t%s\t%s\t%t\n", u.ID, u.Sub, u.Name, u.Email, u.Disabled)
|
||||
}
|
||||
w.Flush()
|
||||
fmt.Fprintf(f.IO.Err, "\n%d user(s) in partition %s\n", len(users), partition)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
c.Flags().StringVar(&limitFlag, "limit", "", "page size for the admin-api (default: server default)")
|
||||
admin.Register(c)
|
||||
return c
|
||||
}
|
||||
|
||||
const defaultGrafanaURL = "https://grafana.futostatus.com"
|
||||
|
||||
func newViewDashboardCmd(f *cmdutil.Factory) *cobra.Command {
|
||||
admin := &cmdutil.AdminFlags{}
|
||||
var userID, email, grafanaURL string
|
||||
var noOpen bool
|
||||
c := &cobra.Command{
|
||||
Use: "view-dashboard",
|
||||
Short: "Open the per-user Grafana dashboard for a user",
|
||||
Long: "Build the Grafana per-user drill-down URL (dashboard uid yucca-per-user) for\n" +
|
||||
"a user and open it in the browser. --id builds the URL without contacting the\n" +
|
||||
"admin-api; --email resolves the user via the partition's admin-api first.",
|
||||
Args: cobra.NoArgs,
|
||||
RunE: func(cmd *cobra.Command, _ []string) error {
|
||||
id := userID
|
||||
if email != "" {
|
||||
client, partition, err := admin.Client(cmd.Context(), f)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
id, err = client.ResolveUserID(cmd.Context(), email)
|
||||
if err != nil {
|
||||
return fmt.Errorf("%w in partition %s", err, partition)
|
||||
}
|
||||
}
|
||||
|
||||
base := grafanaURL
|
||||
if base == "" {
|
||||
base = os.Getenv("YUCTL_GRAFANA_URL")
|
||||
}
|
||||
if base == "" {
|
||||
base = defaultGrafanaURL
|
||||
}
|
||||
dashboardURL := strings.TrimRight(base, "/") + "/d/yucca-per-user?var-user=" + url.QueryEscape(id)
|
||||
|
||||
fmt.Fprintln(f.IO.Out, dashboardURL)
|
||||
if noOpen {
|
||||
return nil
|
||||
}
|
||||
if err := cmdutil.OpenBrowser(dashboardURL); err != nil {
|
||||
return fmt.Errorf("open browser: %w", err)
|
||||
}
|
||||
return nil
|
||||
},
|
||||
}
|
||||
c.Flags().StringVar(&userID, "id", "", "user id (uuid)")
|
||||
c.Flags().StringVar(&email, "email", "", "user email; resolved to an id via the admin-api")
|
||||
c.Flags().StringVar(&grafanaURL, "grafana-url", "", "Grafana base URL (default: $YUCTL_GRAFANA_URL or "+defaultGrafanaURL+")")
|
||||
c.Flags().BoolVar(&noOpen, "no-open", false, "print the dashboard URL instead of opening the browser")
|
||||
c.MarkFlagsOneRequired("id", "email")
|
||||
c.MarkFlagsMutuallyExclusive("id", "email")
|
||||
admin.Register(c)
|
||||
return c
|
||||
}
|
||||
@@ -1,8 +1,8 @@
|
||||
// Command bench-agent is the remote half of `yuctl tools bench` and
|
||||
// `yuctl tools bench-do`. In bench mode (no args) it reads a bench.Config as
|
||||
// `yuctl tools fleet-bench`. In bench mode (no args) it reads a resticbench.Config as
|
||||
// JSON on stdin, executes the benchmark phases, and streams JSON events on
|
||||
// stdout. In loadgen mode (--loadgen [config-path]) it reads a
|
||||
// bench.LoadgenConfig — from the file, which it deletes after reading, or
|
||||
// resticbench.LoadgenConfig — from the file, which it deletes after reading, or
|
||||
// from stdin when no path is given — and either supervises the detached
|
||||
// continuous load or runs the synchronous repo cleanup. The orchestrator
|
||||
// pushes it over ssh; it is embedded into yuctl by the build task.
|
||||
@@ -16,7 +16,7 @@ import (
|
||||
"os/signal"
|
||||
"syscall"
|
||||
|
||||
"yuctl/internal/bench"
|
||||
"yuctl/resticbench"
|
||||
)
|
||||
|
||||
func main() {
|
||||
@@ -31,20 +31,20 @@ func main() {
|
||||
return
|
||||
}
|
||||
|
||||
emit := bench.EmitJSON(os.Stdout)
|
||||
var cfg bench.Config
|
||||
emit := resticbench.EmitJSON(os.Stdout)
|
||||
var cfg resticbench.Config
|
||||
if err := json.NewDecoder(os.Stdin).Decode(&cfg); err != nil {
|
||||
emit(bench.Event{Type: "fatal", Message: "read config from stdin: " + err.Error()})
|
||||
emit(resticbench.Event{Type: "fatal", Message: "read config from stdin: " + err.Error()})
|
||||
os.Exit(1)
|
||||
}
|
||||
if err := bench.RunAgent(ctx, cfg, emit); err != nil {
|
||||
emit(bench.Event{Type: "fatal", Message: err.Error()})
|
||||
if err := resticbench.RunAgent(ctx, cfg, emit); err != nil {
|
||||
emit(resticbench.Event{Type: "fatal", Message: err.Error()})
|
||||
os.Exit(1)
|
||||
}
|
||||
}
|
||||
|
||||
func loadgen(ctx context.Context, args []string) error {
|
||||
var cfg bench.LoadgenConfig
|
||||
var cfg resticbench.LoadgenConfig
|
||||
if len(args) > 0 {
|
||||
// Config file: written 0600 by the launcher, consumed exactly once.
|
||||
b, err := os.ReadFile(args[0])
|
||||
@@ -59,14 +59,14 @@ func loadgen(ctx context.Context, args []string) error {
|
||||
return fmt.Errorf("read config from stdin: %w", err)
|
||||
}
|
||||
|
||||
if cfg.Op == bench.LoadgenOpCleanup {
|
||||
emit := bench.EmitJSON(os.Stdout)
|
||||
if err := bench.RunLoadgenCleanup(ctx, cfg, emit); err != nil {
|
||||
emit(bench.Event{Type: "fatal", Message: err.Error()})
|
||||
if cfg.Op == resticbench.LoadgenOpCleanup {
|
||||
emit := resticbench.EmitJSON(os.Stdout)
|
||||
if err := resticbench.RunLoadgenCleanup(ctx, cfg, emit); err != nil {
|
||||
emit(resticbench.Event{Type: "fatal", Message: err.Error()})
|
||||
os.Exit(1)
|
||||
}
|
||||
emit(bench.Event{Type: "result", Result: &bench.RunResult{}})
|
||||
emit(resticbench.Event{Type: "result", Result: &resticbench.RunResult{}})
|
||||
return nil
|
||||
}
|
||||
return bench.RunLoadgen(ctx, cfg)
|
||||
return resticbench.RunLoadgen(ctx, cfg)
|
||||
}
|
||||
|
||||
@@ -0,0 +1,147 @@
|
||||
package cmdutil
|
||||
|
||||
import (
|
||||
"context"
|
||||
"crypto/tls"
|
||||
"fmt"
|
||||
"net/http"
|
||||
"net/url"
|
||||
"os"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/spf13/cobra"
|
||||
|
||||
"yuctl/adminapi"
|
||||
"yuctl/ctxstore"
|
||||
"yuctl/discovery"
|
||||
)
|
||||
|
||||
// AdminFlags is the shared flag set for commands that talk to the admin-api.
|
||||
type AdminFlags struct {
|
||||
URL string
|
||||
Insecure bool
|
||||
Reauth bool
|
||||
NoBrowser bool
|
||||
}
|
||||
|
||||
func (a *AdminFlags) Register(c *cobra.Command) {
|
||||
c.Flags().StringVar(&a.URL, "admin-url", "", "admin-api base URL (default: derived from discovery, or $YUCTL_ADMIN_API_URL)")
|
||||
c.Flags().BoolVar(&a.Insecure, "insecure-skip-tls-verify", false, "skip TLS verification")
|
||||
c.Flags().BoolVar(&a.Reauth, "reauth", false, "force a fresh browser login")
|
||||
c.Flags().BoolVar(&a.NoBrowser, "no-browser", false, "do not auto-open the browser; just print the login URL")
|
||||
}
|
||||
|
||||
func (a *AdminFlags) httpClient() *http.Client {
|
||||
hc := &http.Client{Timeout: 30 * time.Second}
|
||||
if a.Insecure {
|
||||
tr := http.DefaultTransport.(*http.Transport).Clone()
|
||||
tr.TLSClientConfig = &tls.Config{InsecureSkipVerify: true}
|
||||
hc.Transport = tr
|
||||
}
|
||||
return hc
|
||||
}
|
||||
|
||||
// Client returns an authenticated admin-api client and the selected partition.
|
||||
func (a *AdminFlags) Client(ctx context.Context, f *Factory) (*adminapi.Client, string, error) {
|
||||
cc, err := f.Context()
|
||||
if err != nil {
|
||||
return nil, "", err
|
||||
}
|
||||
topo, err := f.Topology(ctx)
|
||||
if err != nil {
|
||||
return nil, "", err
|
||||
}
|
||||
client, _, err := a.Login(ctx, f, cc, topo)
|
||||
if err != nil {
|
||||
return nil, "", err
|
||||
}
|
||||
return client, cc.Partition, nil
|
||||
}
|
||||
|
||||
// Login returns an authenticated admin-api client, reusing the cached
|
||||
// per-partition session when valid and running the browser loopback flow
|
||||
// otherwise. topo may be nil when the admin URL comes from --admin-url or
|
||||
// $YUCTL_ADMIN_API_URL.
|
||||
func (a *AdminFlags) Login(ctx context.Context, f *Factory, cc *ctxstore.Context, topo *discovery.Topology) (*adminapi.Client, *adminapi.Token, error) {
|
||||
adminURL, err := a.resolveAdminURL(topo, cc)
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
hc := a.httpClient()
|
||||
|
||||
token, err := adminapi.LoadToken(cc.Partition)
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
if a.Reauth || !token.Valid() {
|
||||
var openFn func(string) error
|
||||
if !a.NoBrowser {
|
||||
openFn = OpenBrowser
|
||||
}
|
||||
fresh, err := adminapi.BrowserLogin(ctx, hc, adminURL, openFn, func(loginURL string) {
|
||||
fmt.Fprintln(f.IO.Err, "Complete the login in your browser:")
|
||||
fmt.Fprintf(f.IO.Err, " %s\n", loginURL)
|
||||
})
|
||||
if err != nil {
|
||||
return nil, nil, fmt.Errorf("browser login: %w", err)
|
||||
}
|
||||
token = *fresh
|
||||
if err := adminapi.SaveToken(cc.Partition, token); err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
}
|
||||
|
||||
return adminapi.NewClient(adminURL, token, hc), &token, nil
|
||||
}
|
||||
|
||||
// OptionalTopology resolves the topology only when Login will need it to
|
||||
// derive the admin URL — an explicit --admin-url or $YUCTL_ADMIN_API_URL
|
||||
// skips state access entirely (the context file alone names the token-cache
|
||||
// partition).
|
||||
func (a *AdminFlags) OptionalTopology(ctx context.Context, f *Factory) (*discovery.Topology, error) {
|
||||
if a.URL != "" || os.Getenv("YUCTL_ADMIN_API_URL") != "" {
|
||||
return nil, nil
|
||||
}
|
||||
return f.Topology(ctx)
|
||||
}
|
||||
|
||||
// resolveAdminURL applies the override chain: --admin-url > $YUCTL_ADMIN_API_URL
|
||||
// > derived from the primary region's discovery.
|
||||
func (a *AdminFlags) resolveAdminURL(topo *discovery.Topology, cc *ctxstore.Context) (string, error) {
|
||||
if a.URL != "" {
|
||||
return a.URL, nil
|
||||
}
|
||||
if env := os.Getenv("YUCTL_ADMIN_API_URL"); env != "" {
|
||||
return env, nil
|
||||
}
|
||||
if topo == nil {
|
||||
return "", fmt.Errorf("no topology available to derive the admin-api URL; pass --admin-url or set YUCTL_ADMIN_API_URL")
|
||||
}
|
||||
primary := topo.PrimaryRegion(cc.Partition)
|
||||
if primary == "" {
|
||||
return "", fmt.Errorf("no primary region found for partition %q (no stack with discovery.role==\"primary\")", cc.Partition)
|
||||
}
|
||||
return deriveAdminURL(topo, cc.Partition, primary)
|
||||
}
|
||||
|
||||
// deriveAdminURL builds the admin-api origin for the partition's primary
|
||||
// region from its k8s discovery payload: the Talos API endpoint lives at
|
||||
// kube.<cluster>.<region>.<provider>.yucca.futo.network, and the admin-api is
|
||||
// published on the same NetBird overlay zone as admin.<...> (the
|
||||
// YUCCA_ADMIN_HOST cluster-setting follows the same convention).
|
||||
func deriveAdminURL(topo *discovery.Topology, partition, region string) (string, error) {
|
||||
k8s := topo.Kubernetes(partition, region)
|
||||
if k8s == nil || k8s.APIEndpoint == "" {
|
||||
return "", fmt.Errorf("no kubernetes discovery payload for %s@%s; pass --admin-url or set YUCTL_ADMIN_API_URL", partition, region)
|
||||
}
|
||||
u, err := url.Parse(k8s.APIEndpoint)
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("parse api_endpoint %q: %w", k8s.APIEndpoint, err)
|
||||
}
|
||||
host, ok := strings.CutPrefix(u.Hostname(), "kube.")
|
||||
if !ok {
|
||||
return "", fmt.Errorf("api_endpoint host %q does not start with kube.; pass --admin-url or set YUCTL_ADMIN_API_URL", u.Hostname())
|
||||
}
|
||||
return "https://admin." + host, nil
|
||||
}
|
||||
@@ -0,0 +1,50 @@
|
||||
// Package cmdutil sits between cli and the domain packages: command packages
|
||||
// receive a *Factory instead of re-deriving shared dependencies per command,
|
||||
// and domain packages must never import it.
|
||||
package cmdutil
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"context"
|
||||
"fmt"
|
||||
"os/exec"
|
||||
"runtime"
|
||||
"strings"
|
||||
|
||||
"yuctl/ctxstore"
|
||||
"yuctl/discovery"
|
||||
"yuctl/ui"
|
||||
)
|
||||
|
||||
// Factory's closures are lazy — a command that never touches topology never
|
||||
// reads Terraform state — and memoized per invocation where resolution is
|
||||
// expensive.
|
||||
type Factory struct {
|
||||
IO *ui.IOStreams
|
||||
|
||||
// Context errors when nothing is selected (`yuctl select` first).
|
||||
Context func() (*ctxstore.Context, error)
|
||||
|
||||
Topology func(ctx context.Context) (*discovery.Topology, error)
|
||||
}
|
||||
|
||||
// Confirm prompts the operator for a yes/no answer, defaulting to no.
|
||||
func Confirm(io *ui.IOStreams, prompt string) bool {
|
||||
fmt.Fprintf(io.Err, "%s [y/N]: ", prompt)
|
||||
line, err := bufio.NewReader(io.In).ReadString('\n')
|
||||
if err != nil {
|
||||
return false
|
||||
}
|
||||
answer := strings.ToLower(strings.TrimSpace(line))
|
||||
return answer == "y" || answer == "yes"
|
||||
}
|
||||
|
||||
// OpenBrowser opens url in the OS default browser, best-effort.
|
||||
func OpenBrowser(url string) error {
|
||||
switch runtime.GOOS {
|
||||
case "darwin":
|
||||
return exec.Command("open", url).Start()
|
||||
default:
|
||||
return exec.Command("xdg-open", url).Start()
|
||||
}
|
||||
}
|
||||
@@ -1,8 +1,8 @@
|
||||
// Package context persists the operator's selected topology to
|
||||
// Package ctxstore persists the operator's selected topology to
|
||||
// ${XDG_CONFIG_HOME:-~/.config}/yuctl/context.json. It is the small bit of
|
||||
// sticky state that lets `yuctl select prod@htz-fsn1` be followed by
|
||||
// `yuctl ceph get health` without re-specifying the target each time.
|
||||
package context
|
||||
package ctxstore
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
@@ -8,7 +8,7 @@ import (
|
||||
|
||||
"github.com/rs/zerolog"
|
||||
|
||||
yctx "yuctl/internal/context"
|
||||
"yuctl/ctxstore"
|
||||
)
|
||||
|
||||
// The resolved topology is cached on disk so repeat invocations skip the
|
||||
@@ -31,7 +31,7 @@ type cacheEnvelope struct {
|
||||
}
|
||||
|
||||
func cachePath() (string, error) {
|
||||
dir, err := yctx.Dir()
|
||||
dir, err := ctxstore.Dir()
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
+1
-1
@@ -8,7 +8,7 @@ import (
|
||||
|
||||
"github.com/rs/zerolog"
|
||||
|
||||
"yuctl/internal/state"
|
||||
"yuctl/state"
|
||||
)
|
||||
|
||||
func testTopology() *Topology {
|
||||
+10
-3
@@ -1,6 +1,6 @@
|
||||
// Package discovery resolves the live yucca topology by reading Terraform state
|
||||
// objects directly from the shared `yucca-tf-state` S3 bucket and parsing each
|
||||
// stack's `discovery` output (see internal/state). It never shells out to
|
||||
// stack's `discovery` output (see yuctl/state). It never shells out to
|
||||
// `terragrunt output`, so it works without a checkout, provider plugins, or
|
||||
// `terragrunt init`.
|
||||
//
|
||||
@@ -14,16 +14,18 @@ import (
|
||||
"context"
|
||||
"fmt"
|
||||
"io"
|
||||
"maps"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"slices"
|
||||
"sort"
|
||||
"strings"
|
||||
"sync"
|
||||
|
||||
"github.com/rs/zerolog"
|
||||
|
||||
"yuctl/internal/op"
|
||||
"yuctl/internal/state"
|
||||
"yuctl/op"
|
||||
"yuctl/state"
|
||||
|
||||
"github.com/aws/aws-sdk-go-v2/aws"
|
||||
"github.com/aws/aws-sdk-go-v2/credentials"
|
||||
@@ -386,6 +388,11 @@ func (t *Topology) CephClusters(partition, region string) map[string]state.CephC
|
||||
return out
|
||||
}
|
||||
|
||||
// CephClusterNames returns the region's ceph cluster names, sorted.
|
||||
func (t *Topology) CephClusterNames(partition, region string) []string {
|
||||
return slices.Sorted(maps.Keys(t.CephClusters(partition, region)))
|
||||
}
|
||||
|
||||
// MgmtHosts returns the region's management-host roster (fabric payload), or
|
||||
// nil for regions without a fabric stack (e.g. austin).
|
||||
func (t *Topology) MgmtHosts(partition, region string) []state.MgmtHost {
|
||||
+1
-1
@@ -4,7 +4,7 @@ import (
|
||||
"reflect"
|
||||
"testing"
|
||||
|
||||
"yuctl/internal/state"
|
||||
"yuctl/state"
|
||||
)
|
||||
|
||||
func TestStackFromKey(t *testing.T) {
|
||||
@@ -1,4 +1,4 @@
|
||||
// Package do wraps the DigitalOcean API (godo) for the bench-do droplet
|
||||
// Package do wraps the DigitalOcean API (godo) for the fleet-bench droplet
|
||||
// fleet: token resolution via 1Password, the yucca-bench project, ephemeral
|
||||
// SSH keys, and tagged droplet lifecycle. Nothing here knows about restic or
|
||||
// michael — it is pure fleet plumbing.
|
||||
@@ -14,7 +14,7 @@ import (
|
||||
"github.com/digitalocean/godo"
|
||||
"github.com/rs/zerolog/log"
|
||||
|
||||
"yuctl/internal/op"
|
||||
"yuctl/op"
|
||||
)
|
||||
|
||||
const (
|
||||
@@ -74,7 +74,7 @@ func (c *Client) EnsureProject(ctx context.Context) (string, error) {
|
||||
p, _, err := c.do.Projects.Create(ctx, &godo.CreateProjectRequest{
|
||||
Name: ProjectName,
|
||||
Purpose: "Operational / Developer tooling",
|
||||
Description: "yuctl tools bench-do restic load-test fleets",
|
||||
Description: "yuctl tools fleet-bench restic load-test fleets",
|
||||
Environment: "Development",
|
||||
})
|
||||
if err != nil {
|
||||
@@ -185,7 +185,7 @@ func (c *Client) GetSize(ctx context.Context, slug string) (*Size, error) {
|
||||
return nil, fmt.Errorf("unknown DO size slug %q", slug)
|
||||
}
|
||||
|
||||
// Droplet is the subset of droplet state bench-do tracks.
|
||||
// Droplet is the subset of droplet state fleet-bench tracks.
|
||||
type Droplet struct {
|
||||
ID int
|
||||
Name string
|
||||
@@ -0,0 +1,82 @@
|
||||
// Package fleet holds what the load-test fleet tools (fleet/warp on K8s
|
||||
// pods, fleet/fleetbench on cloud VMs) share; anything transport-specific
|
||||
// stays in the subpackages.
|
||||
package fleet
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
"os/signal"
|
||||
"sync"
|
||||
"syscall"
|
||||
"time"
|
||||
)
|
||||
|
||||
type History struct {
|
||||
vals []float64
|
||||
}
|
||||
|
||||
func (h *History) Push(v float64) {
|
||||
h.vals = append(h.vals, v)
|
||||
if len(h.vals) > 60 {
|
||||
h.vals = h.vals[len(h.vals)-60:]
|
||||
}
|
||||
}
|
||||
|
||||
func (h *History) Values() []float64 { return h.vals }
|
||||
|
||||
func Footer(sampleSec int, sampledAt time.Time, watching bool) string {
|
||||
footer := fmt.Sprintf("sampled %ds window at %s", sampleSec, sampledAt.Format("15:04:05"))
|
||||
if watching {
|
||||
footer += " · ctrl-c to quit"
|
||||
}
|
||||
return footer
|
||||
}
|
||||
|
||||
// Watch redraws frame's result on the alternate screen until interrupted; a
|
||||
// frame error is shown and retried rather than ending the watch.
|
||||
func Watch(ctx context.Context, out io.Writer, label string, sample int, frame func(context.Context) (string, error)) error {
|
||||
ctx, stop := signal.NotifyContext(ctx, os.Interrupt, syscall.SIGTERM)
|
||||
defer stop()
|
||||
|
||||
fmt.Fprint(out, "\x1b[?1049h\x1b[?25l")
|
||||
defer fmt.Fprint(out, "\x1b[?25h\x1b[?1049l")
|
||||
fmt.Fprintf(out, "%s connecting · sampling %ds window…\n", label, sample)
|
||||
|
||||
for {
|
||||
s, err := frame(ctx)
|
||||
if ctx.Err() != nil {
|
||||
return nil
|
||||
}
|
||||
fmt.Fprint(out, "\x1b[H\x1b[2J")
|
||||
if err != nil {
|
||||
fmt.Fprintf(out, "status error (retrying): %v\n", err)
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return nil
|
||||
case <-time.After(3 * time.Second):
|
||||
}
|
||||
continue
|
||||
}
|
||||
fmt.Fprintln(out, s)
|
||||
}
|
||||
}
|
||||
|
||||
// Each runs fn for every index concurrently; every index completes even when
|
||||
// one fails, and the first error is returned.
|
||||
func Each(n int, fn func(int) error) error {
|
||||
var wg sync.WaitGroup
|
||||
errs := make([]error, n)
|
||||
for i := range n {
|
||||
wg.Go(func() { errs[i] = fn(i) })
|
||||
}
|
||||
wg.Wait()
|
||||
for _, err := range errs {
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
+52
-77
@@ -1,4 +1,4 @@
|
||||
package benchwide
|
||||
package fleetbench
|
||||
|
||||
import (
|
||||
"context"
|
||||
@@ -7,7 +7,6 @@ import (
|
||||
"encoding/pem"
|
||||
"fmt"
|
||||
"os"
|
||||
"os/exec"
|
||||
"sort"
|
||||
"strings"
|
||||
"sync/atomic"
|
||||
@@ -16,14 +15,15 @@ import (
|
||||
"github.com/rs/zerolog/log"
|
||||
"golang.org/x/crypto/ssh"
|
||||
|
||||
"yuctl/internal/bench"
|
||||
"yuctl/internal/provider"
|
||||
"yuctl/provider"
|
||||
"yuctl/resticbench"
|
||||
"yuctl/sshx"
|
||||
)
|
||||
|
||||
// DeployOptions shape the host fleet. Regions/Size/Image default to the
|
||||
// provider's own defaults when left empty (resolved in Deploy).
|
||||
type DeployOptions struct {
|
||||
Droplets int // fleet size (default 3)
|
||||
Hosts int // fleet size (default 3)
|
||||
Regions []string // round-robined; provider default if empty
|
||||
Size string // size slug; provider default if empty
|
||||
Image string // image slug; provider default if empty
|
||||
@@ -35,8 +35,8 @@ type DeployOptions struct {
|
||||
}
|
||||
|
||||
func (o *DeployOptions) defaults(d provider.Defaults) {
|
||||
if o.Droplets <= 0 {
|
||||
o.Droplets = 3
|
||||
if o.Hosts <= 0 {
|
||||
o.Hosts = 3
|
||||
}
|
||||
if len(o.Regions) == 0 {
|
||||
o.Regions = d.Regions
|
||||
@@ -49,13 +49,13 @@ func (o *DeployOptions) defaults(d provider.Defaults) {
|
||||
}
|
||||
}
|
||||
|
||||
// dropletName is the fleet-stable name of host n (1-based) — reconciliation
|
||||
// hostName is the fleet-stable name of host n (1-based) — reconciliation
|
||||
// keys on these names, so deploy converges instead of duplicating.
|
||||
func (s *Session) dropletName(n int) string {
|
||||
func (s *Session) hostName(n int) string {
|
||||
return fmt.Sprintf("yucca-bench-%s-%02d", s.slug(), n)
|
||||
}
|
||||
|
||||
// Deploy converges the fleet: creates missing droplets (confirmed, with the
|
||||
// Deploy converges the fleet: creates missing hosts (confirmed, with the
|
||||
// hourly cost and transfer pool spelled out), trims surplus ones, files
|
||||
// everything under the yucca-bench project, waits for ssh, and pushes the
|
||||
// bench agent + pinned restic everywhere.
|
||||
@@ -66,15 +66,15 @@ func (s *Session) Deploy(ctx context.Context, opts DeployOptions) error {
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
live, err := s.Droplets(ctx)
|
||||
live, err := s.Hosts(ctx)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
want := map[string]string{} // name -> region
|
||||
var wantNames []string
|
||||
for i := range opts.Droplets {
|
||||
name := s.dropletName(i + 1)
|
||||
for i := range opts.Hosts {
|
||||
name := s.hostName(i + 1)
|
||||
want[name] = opts.Regions[i%len(opts.Regions)]
|
||||
wantNames = append(wantNames, name)
|
||||
}
|
||||
@@ -112,8 +112,8 @@ func (s *Session) Deploy(ctx context.Context, opts DeployOptions) error {
|
||||
" transfer allowance: %s per host → %s pool (hard cap enforced by the agent)",
|
||||
len(missing), size.Slug, opts.Image, s.Provider.Name(),
|
||||
strings.Join(regions, ", "),
|
||||
size.PriceHourly, size.PriceHourly*float64(opts.Droplets), size.PriceHourly*float64(opts.Droplets)*730, opts.Droplets,
|
||||
bench.FormatBytes(size.TransferBytes), bench.FormatBytes(size.TransferBytes*int64(opts.Droplets)))
|
||||
size.PriceHourly, size.PriceHourly*float64(opts.Hosts), size.PriceHourly*float64(opts.Hosts)*730, opts.Hosts,
|
||||
resticbench.FormatBytes(size.TransferBytes), resticbench.FormatBytes(size.TransferBytes*int64(opts.Hosts)))
|
||||
if opts.Confirm != nil && !opts.Confirm(plan) {
|
||||
return fmt.Errorf("deploy aborted")
|
||||
}
|
||||
@@ -143,42 +143,43 @@ func (s *Session) Deploy(ctx context.Context, opts DeployOptions) error {
|
||||
}
|
||||
}
|
||||
|
||||
droplets, err := s.Provider.WaitActive(ctx, s.Tag(), opts.Droplets, 5*time.Minute)
|
||||
hosts, err := s.Provider.WaitActive(ctx, s.Tag(), opts.Hosts, 5*time.Minute)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
// Filing under the project is organizational only — the tag is what every
|
||||
// fleet operation keys on — so a scoped token without project permissions
|
||||
// must not strand freshly created (billing!) droplets mid-deploy. Assign
|
||||
// the whole fleet on every converge; it is idempotent, so a run after a
|
||||
// token fix picks the droplets up.
|
||||
s.assignProject(ctx, droplets)
|
||||
// must not strand freshly created (billing!) hosts mid-deploy. Assign the
|
||||
// whole fleet on every converge; it is idempotent, so a run after a token
|
||||
// fix picks the hosts up.
|
||||
s.assignProject(ctx, hosts)
|
||||
|
||||
log.Info().Int("droplets", len(droplets)).Msg("waiting for ssh")
|
||||
log.Info().Int("hosts", len(hosts)).Msg("waiting for ssh")
|
||||
var sshReady atomic.Int32
|
||||
if err := eachDroplet(droplets, func(d provider.Host) error {
|
||||
if err := eachHost(hosts, func(d provider.Host) error {
|
||||
if err := s.waitSSH(ctx, d, 4*time.Minute); err != nil {
|
||||
return err
|
||||
}
|
||||
log.Info().Str("droplet", d.Name).Int32("ready", sshReady.Add(1)).Int("want", len(droplets)).Msg("ssh ready")
|
||||
log.Info().Str("host", d.Name).Int32("ready", sshReady.Add(1)).Int("want", len(hosts)).Msg("ssh ready")
|
||||
return nil
|
||||
}); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
agentBin, cleanup, err := bench.AgentBinary(opts.AgentBin)
|
||||
agentBin, cleanup, err := resticbench.AgentBinary(opts.AgentBin)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer cleanup()
|
||||
resticBin, err := bench.EnsureResticLinux(ctx)
|
||||
resticBin, err := resticbench.EnsureResticLinux(ctx)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
log.Info().Int("droplets", len(droplets)).Str("restic", bench.ResticVersion).Msg("pushing bench agent + restic")
|
||||
if err := eachDroplet(droplets, func(d provider.Host) error {
|
||||
return s.pushBinaries(ctx, d.PublicIP, agentBin, resticBin)
|
||||
log.Info().Int("hosts", len(hosts)).Str("restic", resticbench.ResticVersion).Msg("pushing bench agent + restic")
|
||||
if err := eachHost(hosts, func(d provider.Host) error {
|
||||
return s.ssh.Push(ctx, d.PublicIP, resticbench.RemoteBinDir,
|
||||
map[string]string{agentBin: "bench-agent", resticBin: "restic"})
|
||||
}); err != nil {
|
||||
return err
|
||||
}
|
||||
@@ -192,7 +193,7 @@ func (s *Session) Deploy(ctx context.Context, opts DeployOptions) error {
|
||||
if err := s.SaveState(); err != nil {
|
||||
return err
|
||||
}
|
||||
log.Info().Int("droplets", len(droplets)).Msg("fleet deployed and ready")
|
||||
log.Info().Int("hosts", len(hosts)).Msg("fleet deployed and ready")
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -218,7 +219,7 @@ func (s *Session) ensureKey(ctx context.Context) (string, error) {
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
block, err := ssh.MarshalPrivateKey(priv, "yuctl bench-wide "+s.slug())
|
||||
block, err := ssh.MarshalPrivateKey(priv, "yuctl fleet-bench "+s.slug())
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("marshal ssh private key: %w", err)
|
||||
}
|
||||
@@ -243,84 +244,58 @@ func (s *Session) ensureKey(ctx context.Context) (string, error) {
|
||||
return keyID, s.SaveState()
|
||||
}
|
||||
|
||||
// waitSSH retries a no-op ssh until the droplet accepts the fleet key. Fresh
|
||||
// droplets take a little while between "active" and sshd+key readiness.
|
||||
// waitSSH retries a no-op ssh until the host accepts the fleet key. Fresh
|
||||
// hosts take a little while between "active" and sshd+key readiness.
|
||||
// Single-shot probes (the loop is the retry) with the latest failure surfaced
|
||||
// periodically, so a stuck droplet is visible instead of a silent hang.
|
||||
// periodically, so a stuck host is visible instead of a silent hang.
|
||||
func (s *Session) waitSSH(ctx context.Context, d provider.Host, timeout time.Duration) error {
|
||||
start := time.Now()
|
||||
deadline := start.Add(timeout)
|
||||
lastLog := start
|
||||
for {
|
||||
_, err := s.sshRunOnce(ctx, d.PublicIP, "true", nil)
|
||||
_, err := s.ssh.RunOnce(ctx, d.PublicIP, "true", nil)
|
||||
if err == nil {
|
||||
return nil
|
||||
}
|
||||
if time.Now().After(deadline) {
|
||||
return fmt.Errorf("droplet %s (%s) not reachable over ssh after %s: %w", d.Name, d.PublicIP, timeout, err)
|
||||
return fmt.Errorf("host %s (%s) not reachable over ssh after %s: %w", d.Name, d.PublicIP, timeout, err)
|
||||
}
|
||||
if time.Since(lastLog) > 30*time.Second {
|
||||
lastLog = time.Now()
|
||||
log.Warn().Str("droplet", d.Name).Str("ip", d.PublicIP).
|
||||
log.Warn().Str("host", d.Name).Str("ip", d.PublicIP).
|
||||
Str("waited", time.Since(start).Round(time.Second).String()).
|
||||
Str("error", tailStr(err.Error(), 160)).Msg("still waiting for ssh")
|
||||
Str("error", sshx.Tail(err.Error(), 160)).Msg("still waiting for ssh")
|
||||
}
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return ctx.Err()
|
||||
// A gentle cadence: on paths that graylist port-22 SYN bursts,
|
||||
// hammering a not-yet-reachable droplet only extends the block.
|
||||
// hammering a not-yet-reachable host only extends the block.
|
||||
case <-time.After(10 * time.Second):
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// pushBinaries scps the agent and restic into RemoteBinDir on the droplet.
|
||||
func (s *Session) pushBinaries(ctx context.Context, ip, agentBin, resticBin string) error {
|
||||
if _, err := s.sshRun(ctx, ip, "mkdir -p "+bench.RemoteBinDir, nil); err != nil {
|
||||
return err
|
||||
}
|
||||
dest := s.Provider.SSHUser() + "@" + ip + ":" + bench.RemoteBinDir + "/"
|
||||
scp := func(local string) error {
|
||||
cmd := exec.CommandContext(ctx, "scp", append([]string{"-q"}, s.sshArgs(local, dest)...)...)
|
||||
if out, err := cmd.CombinedOutput(); err != nil {
|
||||
return fmt.Errorf("scp %s to %s: %w: %s", local, ip, err, tailStr(string(out), 500))
|
||||
}
|
||||
return nil
|
||||
}
|
||||
// The local temp names are already bench-agent / restic_<ver>; normalize
|
||||
// on the remote side so launch paths are stable.
|
||||
if err := scp(agentBin); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := scp(resticBin); err != nil {
|
||||
return err
|
||||
}
|
||||
script := fmt.Sprintf("cd %s && mv -f $(basename %q) bench-agent 2>/dev/null; mv -f $(basename %q) restic 2>/dev/null; chmod +x bench-agent restic",
|
||||
bench.RemoteBinDir, agentBin, resticBin)
|
||||
_, err := s.sshRun(ctx, ip, script, nil)
|
||||
return err
|
||||
}
|
||||
|
||||
// Undeploy destroys the whole fleet: droplets by tag, the DO ssh key, and the
|
||||
// local key/known_hosts/state files. Repositories (and their data) persist —
|
||||
// run `bench-do cleanup` first to prune them while the droplets still exist.
|
||||
// Undeploy destroys the whole fleet: hosts by tag, the provider ssh key, and
|
||||
// the local key/known_hosts/state files. Repositories (and their data)
|
||||
// persist — run `fleet-bench cleanup` first to prune them while the hosts
|
||||
// still exist.
|
||||
func (s *Session) Undeploy(ctx context.Context, force bool) error {
|
||||
droplets, err := s.Droplets(ctx)
|
||||
hosts, err := s.Hosts(ctx)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if len(droplets) == 0 {
|
||||
log.Info().Msg("no droplets tagged for this fleet")
|
||||
if len(hosts) == 0 {
|
||||
log.Info().Msg("no hosts tagged for this fleet")
|
||||
} else {
|
||||
if !force {
|
||||
if running, err := s.anyLoadRunning(ctx, droplets); err != nil {
|
||||
if running, err := s.anyLoadRunning(ctx, hosts); err != nil {
|
||||
log.Warn().Err(err).Msg("could not check for running load")
|
||||
} else if running {
|
||||
return fmt.Errorf("load is still running; `yuctl tools bench-wide stop` first (or --force)")
|
||||
return fmt.Errorf("load is still running; `yuctl tools fleet-bench stop` first (or --force)")
|
||||
}
|
||||
}
|
||||
log.Info().Int("hosts", len(droplets)).Msg("destroying fleet hosts")
|
||||
log.Info().Int("hosts", len(hosts)).Msg("destroying fleet hosts")
|
||||
if err := s.Provider.DeleteByTag(ctx, s.Tag()); err != nil {
|
||||
return err
|
||||
}
|
||||
@@ -343,10 +318,10 @@ func (s *Session) Undeploy(ctx context.Context, force bool) error {
|
||||
|
||||
// anyLoadRunning reports whether any host still has an agent or restic
|
||||
// process alive.
|
||||
func (s *Session) anyLoadRunning(ctx context.Context, droplets []provider.Host) (bool, error) {
|
||||
func (s *Session) anyLoadRunning(ctx context.Context, hosts []provider.Host) (bool, error) {
|
||||
running := false
|
||||
err := eachDroplet(droplets, func(d provider.Host) error {
|
||||
out, err := s.sshRun(ctx, d.PublicIP,
|
||||
err := eachHost(hosts, func(d provider.Host) error {
|
||||
out, err := s.ssh.Run(ctx, d.PublicIP,
|
||||
`echo "$(pgrep -fc 'bench-agent --[l]oadgen') $(pgrep -xc restic)" 2>/dev/null || true`, nil)
|
||||
if err != nil {
|
||||
return err
|
||||
+7
-7
@@ -1,4 +1,4 @@
|
||||
package benchwide
|
||||
package fleetbench
|
||||
|
||||
import (
|
||||
"regexp"
|
||||
@@ -6,10 +6,10 @@ import (
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestDropletName(t *testing.T) {
|
||||
func TestHostName(t *testing.T) {
|
||||
s := &Session{Partition: "staging", providerName: "do"}
|
||||
if got := s.dropletName(7); got != "yucca-bench-do-staging-07" {
|
||||
t.Fatalf("dropletName = %q", got)
|
||||
if got := s.hostName(7); got != "yucca-bench-do-staging-07" {
|
||||
t.Fatalf("hostName = %q", got)
|
||||
}
|
||||
if got := s.Tag(); got != "yuctl-bench-do-staging" {
|
||||
t.Fatalf("Tag = %q", got)
|
||||
@@ -44,7 +44,7 @@ Inter-| Receive | Transmit
|
||||
---S3---
|
||||
1 2
|
||||
`
|
||||
ds := DropletStatus{}
|
||||
ds := HostStatus{}
|
||||
parseSample(out, 5, &ds)
|
||||
if ds.TxBps != float64(1900000-900000)*8/5 {
|
||||
t.Fatalf("TxBps = %v", ds.TxBps)
|
||||
@@ -65,7 +65,7 @@ Inter-| Receive | Transmit
|
||||
|
||||
func TestParseSampleMissingStatus(t *testing.T) {
|
||||
out := "x---S1---y---S2---\n\n---S3---\n0 0\n"
|
||||
ds := DropletStatus{}
|
||||
ds := HostStatus{}
|
||||
parseSample(out, 5, &ds)
|
||||
if ds.Status != nil {
|
||||
t.Fatalf("expected nil status, got %+v", ds.Status)
|
||||
@@ -78,7 +78,7 @@ func TestParseSampleMissingStatus(t *testing.T) {
|
||||
func TestStartOptionsDefaults(t *testing.T) {
|
||||
o := StartOptions{}
|
||||
o.defaults()
|
||||
if o.ClientsPerDroplet != 1 || o.PackSizeMiB != 16 || o.CycleSize != 8<<30 || o.Seed == 0 {
|
||||
if o.ClientsPerHost != 1 || o.PackSizeMiB != 16 || o.CycleSize != 8<<30 || o.Seed == 0 {
|
||||
t.Fatalf("defaults = %+v", o)
|
||||
}
|
||||
if o.Label != "run" || o.Compression != "off" {
|
||||
@@ -0,0 +1,54 @@
|
||||
package fleetbench
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
"strings"
|
||||
"text/tabwriter"
|
||||
|
||||
"yuctl/resticbench"
|
||||
)
|
||||
|
||||
// ShortName drops the shared fleet prefix so tables stay narrow.
|
||||
func ShortName(name string) string {
|
||||
return strings.TrimPrefix(name, "yucca-bench-")
|
||||
}
|
||||
|
||||
func SaveResult(path string, r *Result) error {
|
||||
b, err := json.MarshalIndent(r, "", " ")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
return os.WriteFile(path, append(b, '\n'), 0o644)
|
||||
}
|
||||
|
||||
func RenderResult(w io.Writer, r *Result) {
|
||||
fmt.Fprintf(w, "\n%s partition=%s elapsed=%s\n", r.Label, r.Partition, resticbench.FormatDuration(r.ElapsedSeconds))
|
||||
tw := tabwriter.NewWriter(w, 2, 4, 2, ' ', 0)
|
||||
fmt.Fprintln(tw, "HOST\tREGION\tSTATE\tWIRE TX\tCLIENT\tCYCLES\tUPLOADED\tERRORS")
|
||||
for _, d := range r.Hosts {
|
||||
if len(d.Clients) == 0 {
|
||||
fmt.Fprintf(tw, "%s\t%s\t%s\t%s\t-\t-\t-\t-\n",
|
||||
ShortName(d.Name), d.Region, d.State, resticbench.FormatBytes(d.WireTxBytes))
|
||||
continue
|
||||
}
|
||||
for i, c := range d.Clients {
|
||||
name, region, state, wireTx := ShortName(d.Name), d.Region, d.State, resticbench.FormatBytes(d.WireTxBytes)
|
||||
if i > 0 {
|
||||
name, region, state, wireTx = "", "", "", ""
|
||||
}
|
||||
fmt.Fprintf(tw, "%s\t%s\t%s\t%s\t%s\t%d\t%s\t%d\n",
|
||||
name, region, state, wireTx, "c"+strings.TrimPrefix(c.Name, d.Name+"-c"), c.Cycles,
|
||||
resticbench.FormatBytes(c.Uploaded), c.Errors)
|
||||
}
|
||||
}
|
||||
tw.Flush()
|
||||
fmt.Fprintf(w, "totals: uploaded=%s wire-tx=%s cycles=%d errors=%d",
|
||||
resticbench.FormatBytes(r.TotalUploaded), resticbench.FormatBytes(r.TotalWireTx), r.TotalCycles, r.TotalErrors)
|
||||
if r.ElapsedSeconds > 0 && r.TotalUploaded > 0 {
|
||||
fmt.Fprintf(w, " avg=%s", resticbench.FormatBPS(float64(r.TotalUploaded)/r.ElapsedSeconds))
|
||||
}
|
||||
fmt.Fprintln(w)
|
||||
}
|
||||
+119
-131
@@ -1,14 +1,12 @@
|
||||
package benchwide
|
||||
package fleetbench
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"context"
|
||||
"crypto/rand"
|
||||
"encoding/binary"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"os"
|
||||
"os/exec"
|
||||
"sort"
|
||||
"strconv"
|
||||
"strings"
|
||||
@@ -17,10 +15,11 @@ import (
|
||||
|
||||
"github.com/rs/zerolog/log"
|
||||
|
||||
"yuctl/internal/adminapi"
|
||||
"yuctl/internal/bench"
|
||||
"yuctl/internal/netdev"
|
||||
"yuctl/internal/provider"
|
||||
"yuctl/adminapi"
|
||||
"yuctl/netdev"
|
||||
"yuctl/provider"
|
||||
"yuctl/resticbench"
|
||||
"yuctl/sshx"
|
||||
)
|
||||
|
||||
// RepoMinter is the slice of the admin-api client Start needs: repository
|
||||
@@ -31,24 +30,24 @@ type RepoMinter interface {
|
||||
RepositoryURL(ctx context.Context, id string) (string, error)
|
||||
}
|
||||
|
||||
// StartOptions shape the load each droplet runs.
|
||||
// StartOptions shape the load each host runs.
|
||||
type StartOptions struct {
|
||||
ClientsPerDroplet int // default 1
|
||||
CycleSize int64 // dataset per client per cycle (default 8GiB)
|
||||
FileSize int64 // default 64MiB
|
||||
PackSizeMiB int // restic pack ("object") size, 4..128 (default 16)
|
||||
Connections int // rest.connections per client (default 5)
|
||||
ReadConcurrency int // default 4
|
||||
Compression string // default off
|
||||
Duration time.Duration // 0 = until `stop`
|
||||
MaxTransfer int64 // per-droplet wire cap; 0 = the size's allowance
|
||||
Label string // results label
|
||||
Seed uint64 // 0 = random
|
||||
ClientsPerHost int // default 1
|
||||
CycleSize int64 // dataset per client per cycle (default 8GiB)
|
||||
FileSize int64 // default 64MiB
|
||||
PackSizeMiB int // restic pack ("object") size, 4..128 (default 16)
|
||||
Connections int // rest.connections per client (default 5)
|
||||
ReadConcurrency int // default 4
|
||||
Compression string // default off
|
||||
Duration time.Duration // 0 = until `stop`
|
||||
MaxTransfer int64 // per-host wire cap; 0 = the size's allowance
|
||||
Label string // results label
|
||||
Seed uint64 // 0 = random
|
||||
}
|
||||
|
||||
func (o *StartOptions) defaults() {
|
||||
if o.ClientsPerDroplet <= 0 {
|
||||
o.ClientsPerDroplet = 1
|
||||
if o.ClientsPerHost <= 0 {
|
||||
o.ClientsPerHost = 1
|
||||
}
|
||||
if o.CycleSize <= 0 {
|
||||
o.CycleSize = 8 << 30
|
||||
@@ -83,7 +82,7 @@ func (o *StartOptions) defaults() {
|
||||
// pattern from matching this very script's command line.
|
||||
const killScript = `pkill -f 'bench-agent --[l]oadgen' 2>/dev/null; sleep 1; pkill -x restic 2>/dev/null; true`
|
||||
|
||||
// prepScript and launchScript are one droplet's start, as two SEPARATE ssh
|
||||
// prepScript and launchScript are one host's start, as two SEPARATE ssh
|
||||
// execs. They must not be merged: prep runs the kill patterns, and launch's
|
||||
// command line spells out "bench-agent --loadgen" verbatim — in one script
|
||||
// the pkill would match its own shell's command line and SIGTERM it mid-run
|
||||
@@ -98,11 +97,11 @@ func prepScript() string {
|
||||
func launchScript() string {
|
||||
return fmt.Sprintf(
|
||||
"setsid nohup $HOME/%s/bench-agent --loadgen %s </dev/null >> %s 2>&1 & echo launched",
|
||||
bench.RemoteBinDir, configPath, agentLog)
|
||||
resticbench.RemoteBinDir, configPath, agentLog)
|
||||
}
|
||||
|
||||
// Start (re)launches the load on every droplet: ensures one repository per
|
||||
// client via the admin-api, re-mints restic URLs, and hands each droplet a
|
||||
// Start (re)launches the load on every host: ensures one repository per
|
||||
// client via the admin-api, re-mints restic URLs, and hands each host a
|
||||
// LoadgenConfig over ssh stdin (secrets never in argv). A second start is a
|
||||
// graceful restart with new parameters.
|
||||
func (s *Session) Start(ctx context.Context, minter RepoMinter, opts StartOptions) error {
|
||||
@@ -110,15 +109,15 @@ func (s *Session) Start(ctx context.Context, minter RepoMinter, opts StartOption
|
||||
if opts.PackSizeMiB < 4 || opts.PackSizeMiB > 128 {
|
||||
return fmt.Errorf("--obj-size %dMiB out of restic's pack-size range (4..128 MiB)", opts.PackSizeMiB)
|
||||
}
|
||||
droplets, err := s.Droplets(ctx)
|
||||
hosts, err := s.Hosts(ctx)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if len(droplets) == 0 {
|
||||
return fmt.Errorf("no fleet droplets; run `yuctl tools bench-wide deploy` first")
|
||||
if len(hosts) == 0 {
|
||||
return fmt.Errorf("no fleet hosts; run `yuctl tools fleet-bench deploy` first")
|
||||
}
|
||||
|
||||
clients, err := s.ensureClients(ctx, minter, droplets, opts.ClientsPerDroplet)
|
||||
clients, err := s.ensureClients(ctx, minter, hosts, opts.ClientsPerHost)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
@@ -136,9 +135,9 @@ func (s *Session) Start(ctx context.Context, minter RepoMinter, opts StartOption
|
||||
capBytes = s.State.TransferBytes
|
||||
}
|
||||
if capBytes <= 0 {
|
||||
// State was lost but droplets exist: re-derive the allowance rather
|
||||
// State was lost but hosts exist: re-derive the allowance rather
|
||||
// than ever running uncapped into paid overage.
|
||||
size, err := s.Provider.ResolveSize(ctx, droplets[0].SizeSlug)
|
||||
size, err := s.Provider.ResolveSize(ctx, hosts[0].SizeSlug)
|
||||
if err != nil {
|
||||
return fmt.Errorf("no recorded transfer allowance and size lookup failed: %w", err)
|
||||
}
|
||||
@@ -146,23 +145,23 @@ func (s *Session) Start(ctx context.Context, minter RepoMinter, opts StartOption
|
||||
s.State.TransferBytes = capBytes
|
||||
s.State.SizeSlug = size.Slug
|
||||
}
|
||||
log.Info().Int("droplets", len(droplets)).Int("clients", len(clients)).
|
||||
log.Info().Int("hosts", len(hosts)).Int("clients", len(clients)).
|
||||
Str("obj_size", fmt.Sprintf("%dMiB", opts.PackSizeMiB)).
|
||||
Str("cycle_size", bench.FormatBytes(opts.CycleSize)).
|
||||
Str("cycle_size", resticbench.FormatBytes(opts.CycleSize)).
|
||||
Str("duration", durationLabel(opts.Duration)).
|
||||
Str("cap_per_droplet", bench.FormatBytes(capBytes)).
|
||||
Str("cap_pool", bench.FormatBytes(capBytes*int64(len(droplets)))).
|
||||
Str("cap_per_host", resticbench.FormatBytes(capBytes)).
|
||||
Str("cap_pool", resticbench.FormatBytes(capBytes*int64(len(hosts)))).
|
||||
Msg("launching load (kill previous, then detach the agent)")
|
||||
|
||||
err = eachDroplet(droplets, func(d provider.Host) error {
|
||||
var mine []bench.LoadgenClient
|
||||
err = eachHost(hosts, func(d provider.Host) error {
|
||||
var mine []resticbench.LoadgenClient
|
||||
for _, cl := range clients {
|
||||
if cl.Droplet == d.Name {
|
||||
mine = append(mine, bench.LoadgenClient{Name: cl.Name, Repo: urls[cl.Name], Password: cl.Password})
|
||||
if cl.Host == d.Name {
|
||||
mine = append(mine, resticbench.LoadgenClient{Name: cl.Name, Repo: urls[cl.Name], Password: cl.Password})
|
||||
}
|
||||
}
|
||||
cfg := bench.LoadgenConfig{
|
||||
Op: bench.LoadgenOpLoad,
|
||||
cfg := resticbench.LoadgenConfig{
|
||||
Op: resticbench.LoadgenOpLoad,
|
||||
Clients: mine,
|
||||
Workdir: Workdir,
|
||||
CycleSize: opts.CycleSize,
|
||||
@@ -180,17 +179,17 @@ func (s *Session) Start(ctx context.Context, minter RepoMinter, opts StartOption
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if _, err := s.sshRun(ctx, d.PublicIP, prepScript(), raw); err != nil {
|
||||
if _, err := s.ssh.Run(ctx, d.PublicIP, prepScript(), raw); err != nil {
|
||||
return fmt.Errorf("prepare %s: %w", d.Name, err)
|
||||
}
|
||||
out, err := s.sshRun(ctx, d.PublicIP, launchScript(), nil)
|
||||
out, err := s.ssh.Run(ctx, d.PublicIP, launchScript(), nil)
|
||||
if err != nil {
|
||||
return fmt.Errorf("launch on %s: %w", d.Name, err)
|
||||
}
|
||||
if !strings.Contains(out, "launched") {
|
||||
return fmt.Errorf("launch on %s: unexpected output %q", d.Name, tailStr(out, 200))
|
||||
return fmt.Errorf("launch on %s: unexpected output %q", d.Name, sshx.Tail(out, 200))
|
||||
}
|
||||
log.Info().Str("droplet", d.Name).Str("region", d.Region).Int("clients", len(mine)).Msg("load launched")
|
||||
log.Info().Str("host", d.Name).Str("region", d.Region).Int("clients", len(mine)).Msg("load launched")
|
||||
return nil
|
||||
})
|
||||
if err != nil {
|
||||
@@ -201,20 +200,20 @@ func (s *Session) Start(ctx context.Context, minter RepoMinter, opts StartOption
|
||||
StartedAt: time.Now().UTC(),
|
||||
Label: opts.Label,
|
||||
Params: map[string]string{
|
||||
"droplets": strconv.Itoa(len(droplets)),
|
||||
"clients_per_droplet": strconv.Itoa(opts.ClientsPerDroplet),
|
||||
"obj_size_mib": strconv.Itoa(opts.PackSizeMiB),
|
||||
"cycle_size": bench.FormatBytes(opts.CycleSize),
|
||||
"connections": strconv.Itoa(opts.Connections),
|
||||
"duration": durationLabel(opts.Duration),
|
||||
"cap_per_droplet": bench.FormatBytes(capBytes),
|
||||
"seed": strconv.FormatUint(opts.Seed, 10),
|
||||
"hosts": strconv.Itoa(len(hosts)),
|
||||
"clients_per_host": strconv.Itoa(opts.ClientsPerHost),
|
||||
"obj_size_mib": strconv.Itoa(opts.PackSizeMiB),
|
||||
"cycle_size": resticbench.FormatBytes(opts.CycleSize),
|
||||
"connections": strconv.Itoa(opts.Connections),
|
||||
"duration": durationLabel(opts.Duration),
|
||||
"cap_per_host": resticbench.FormatBytes(capBytes),
|
||||
"seed": strconv.FormatUint(opts.Seed, 10),
|
||||
},
|
||||
}
|
||||
if err := s.SaveState(); err != nil {
|
||||
return err
|
||||
}
|
||||
log.Info().Msg("all droplets launched; `yuctl tools bench-wide watch` for the live dashboard")
|
||||
log.Info().Msg("all hosts launched; `yuctl tools fleet-bench watch` for the live dashboard")
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -226,17 +225,17 @@ func durationLabel(d time.Duration) string {
|
||||
}
|
||||
|
||||
// ensureClients reconciles the persisted client list against the live fleet:
|
||||
// one repository per (droplet, slot), created on first use and reused across
|
||||
// one repository per (host, slot), created on first use and reused across
|
||||
// restarts (fresh seeds keep reuse dedup-proof).
|
||||
func (s *Session) ensureClients(ctx context.Context, minter RepoMinter, droplets []provider.Host, perDroplet int) ([]Client, error) {
|
||||
func (s *Session) ensureClients(ctx context.Context, minter RepoMinter, hosts []provider.Host, perHost int) ([]Client, error) {
|
||||
existing := map[string]Client{}
|
||||
for _, cl := range s.State.Clients {
|
||||
existing[cl.Name] = cl
|
||||
}
|
||||
var out []Client
|
||||
changed := false
|
||||
for _, d := range droplets {
|
||||
for j := 1; j <= perDroplet; j++ {
|
||||
for _, d := range hosts {
|
||||
for j := 1; j <= perHost; j++ {
|
||||
name := fmt.Sprintf("%s-c%d", d.Name, j)
|
||||
if cl, ok := existing[name]; ok {
|
||||
out = append(out, cl)
|
||||
@@ -247,7 +246,7 @@ func (s *Session) ensureClients(ctx context.Context, minter RepoMinter, droplets
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("create repository %s: %w", repoName, err)
|
||||
}
|
||||
cl := Client{Name: name, Droplet: d.Name, RepoID: repo.ID, RepoName: repoName, Password: randHex(16)}
|
||||
cl := Client{Name: name, Host: d.Name, RepoID: repo.ID, RepoName: repoName, Password: randHex(16)}
|
||||
log.Info().Str("repo", repoName).Str("client", name).Msg("created bench repository (persists after the run)")
|
||||
s.State.Clients = append(s.State.Clients, cl)
|
||||
out = append(out, cl)
|
||||
@@ -262,25 +261,25 @@ func (s *Session) ensureClients(ctx context.Context, minter RepoMinter, droplets
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// Stop kills the load everywhere (droplets stay deployed) and collects each
|
||||
// droplet's final status into a Result.
|
||||
// Stop kills the load everywhere (hosts stay deployed) and collects each
|
||||
// host's final status into a Result.
|
||||
func (s *Session) Stop(ctx context.Context) (*Result, error) {
|
||||
droplets, err := s.Droplets(ctx)
|
||||
hosts, err := s.Hosts(ctx)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if len(droplets) == 0 {
|
||||
log.Info().Msg("no droplets; nothing to stop")
|
||||
if len(hosts) == 0 {
|
||||
log.Info().Msg("no hosts; nothing to stop")
|
||||
return nil, nil
|
||||
}
|
||||
log.Info().Int("droplets", len(droplets)).Msg("stopping load on all droplets")
|
||||
if err := eachDroplet(droplets, func(d provider.Host) error {
|
||||
_, err := s.sshRun(ctx, d.PublicIP, killScript, nil)
|
||||
log.Info().Int("hosts", len(hosts)).Msg("stopping load on all hosts")
|
||||
if err := eachHost(hosts, func(d provider.Host) error {
|
||||
_, err := s.ssh.Run(ctx, d.PublicIP, killScript, nil)
|
||||
return err
|
||||
}); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
res, err := s.collect(ctx, droplets)
|
||||
res, err := s.collect(ctx, hosts)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
@@ -292,7 +291,7 @@ func (s *Session) Stop(ctx context.Context) (*Result, error) {
|
||||
}
|
||||
|
||||
// Result is the fleet-aggregated outcome of a run, written to a local JSON on
|
||||
// stop. Client numbers are restic's post-dedup data_added; droplet wire TX is
|
||||
// stop. Client numbers are restic's post-dedup data_added; host wire TX is
|
||||
// what the transfer allowance actually saw.
|
||||
type Result struct {
|
||||
Label string `json:"label"`
|
||||
@@ -301,23 +300,23 @@ type Result struct {
|
||||
StartedAt time.Time `json:"startedAt,omitempty"`
|
||||
ElapsedSeconds float64 `json:"elapsedSeconds,omitempty"`
|
||||
Params map[string]string `json:"params,omitempty"`
|
||||
Droplets []DropletResult `json:"droplets"`
|
||||
Hosts []HostResult `json:"droplets"`
|
||||
TotalUploaded int64 `json:"totalUploaded"`
|
||||
TotalWireTx int64 `json:"totalWireTx"`
|
||||
TotalCycles int `json:"totalCycles"`
|
||||
TotalErrors int `json:"totalErrors"`
|
||||
}
|
||||
|
||||
// DropletResult is one droplet's final tallies.
|
||||
type DropletResult struct {
|
||||
Name string `json:"name"`
|
||||
Region string `json:"region"`
|
||||
State string `json:"state,omitempty"`
|
||||
WireTxBytes int64 `json:"wireTxBytes"`
|
||||
Clients []bench.LoadgenClientStatus `json:"clients,omitempty"`
|
||||
// HostResult is one host's final tallies.
|
||||
type HostResult struct {
|
||||
Name string `json:"name"`
|
||||
Region string `json:"region"`
|
||||
State string `json:"state,omitempty"`
|
||||
WireTxBytes int64 `json:"wireTxBytes"`
|
||||
Clients []resticbench.LoadgenClientStatus `json:"clients,omitempty"`
|
||||
}
|
||||
|
||||
func (s *Session) collect(ctx context.Context, droplets []provider.Host) (*Result, error) {
|
||||
func (s *Session) collect(ctx context.Context, hosts []provider.Host) (*Result, error) {
|
||||
res := &Result{Partition: s.Partition, Created: time.Now().UTC(), Label: "run"}
|
||||
if s.State.Run != nil {
|
||||
res.Label = s.State.Run.Label
|
||||
@@ -326,27 +325,27 @@ func (s *Session) collect(ctx context.Context, droplets []provider.Host) (*Resul
|
||||
res.Params = s.State.Run.Params
|
||||
}
|
||||
var mu sync.Mutex
|
||||
err := eachDroplet(droplets, func(d provider.Host) error {
|
||||
out, err := s.sshRun(ctx, d.PublicIP, "cat "+statusPath+" 2>/dev/null || true", nil)
|
||||
err := eachHost(hosts, func(d provider.Host) error {
|
||||
out, err := s.ssh.Run(ctx, d.PublicIP, "cat "+statusPath+" 2>/dev/null || true", nil)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
dr := DropletResult{Name: d.Name, Region: d.Region}
|
||||
var st bench.LoadgenStatus
|
||||
dr := HostResult{Name: d.Name, Region: d.Region}
|
||||
var st resticbench.LoadgenStatus
|
||||
if json.Unmarshal([]byte(out), &st) == nil {
|
||||
dr.State = st.State
|
||||
dr.WireTxBytes = st.WireTxBytes
|
||||
dr.Clients = st.Clients
|
||||
}
|
||||
mu.Lock()
|
||||
res.Droplets = append(res.Droplets, dr)
|
||||
res.Hosts = append(res.Hosts, dr)
|
||||
mu.Unlock()
|
||||
return nil
|
||||
})
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
for _, dr := range res.Droplets {
|
||||
for _, dr := range res.Hosts {
|
||||
res.TotalWireTx += dr.WireTxBytes
|
||||
for _, cl := range dr.Clients {
|
||||
res.TotalUploaded += cl.Uploaded
|
||||
@@ -357,30 +356,30 @@ func (s *Session) collect(ctx context.Context, droplets []provider.Host) (*Resul
|
||||
return res, nil
|
||||
}
|
||||
|
||||
// DropletStatus is one droplet's live sample.
|
||||
type DropletStatus struct {
|
||||
// HostStatus is one host's live sample.
|
||||
type HostStatus struct {
|
||||
Name, Region, IP string
|
||||
Reachable bool
|
||||
Err string
|
||||
AgentProcs, ResticProcs int
|
||||
TxBps, RxBps float64
|
||||
Status *bench.LoadgenStatus
|
||||
Status *resticbench.LoadgenStatus
|
||||
}
|
||||
|
||||
// StatusReport is what status/watch render.
|
||||
type StatusReport struct {
|
||||
Run *RunInfo
|
||||
TransferCap int64 // per droplet
|
||||
Droplets []DropletStatus
|
||||
TransferCap int64 // per host
|
||||
Hosts []HostStatus
|
||||
}
|
||||
|
||||
// Status samples every droplet in parallel with one ssh round-trip each:
|
||||
// Status samples every host in parallel with one ssh round-trip each:
|
||||
// NIC counters over the window, the agent's status file, and process counts.
|
||||
func (s *Session) Status(ctx context.Context, sampleSeconds int) (*StatusReport, error) {
|
||||
if sampleSeconds <= 0 {
|
||||
sampleSeconds = 5
|
||||
}
|
||||
droplets, err := s.Droplets(ctx)
|
||||
hosts, err := s.Hosts(ctx)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
@@ -392,30 +391,26 @@ func (s *Session) Status(ctx context.Context, sampleSeconds int) (*StatusReport,
|
||||
sampleSeconds, statusPath)
|
||||
|
||||
var mu sync.Mutex
|
||||
_ = eachDroplet(droplets, func(d provider.Host) error {
|
||||
ds := DropletStatus{Name: d.Name, Region: d.Region, IP: d.PublicIP}
|
||||
out, err := s.sshRun(ctx, d.PublicIP, script, nil)
|
||||
_ = eachHost(hosts, func(d provider.Host) error {
|
||||
ds := HostStatus{Name: d.Name, Region: d.Region, IP: d.PublicIP}
|
||||
out, err := s.ssh.Run(ctx, d.PublicIP, script, nil)
|
||||
if err != nil {
|
||||
ds.Err = tailStr(err.Error(), 120)
|
||||
ds.Err = sshx.Tail(err.Error(), 120)
|
||||
} else {
|
||||
ds.Reachable = true
|
||||
parseSample(out, sampleSeconds, &ds)
|
||||
}
|
||||
mu.Lock()
|
||||
report.Droplets = append(report.Droplets, ds)
|
||||
report.Hosts = append(report.Hosts, ds)
|
||||
mu.Unlock()
|
||||
return nil
|
||||
})
|
||||
sortDroplets(report.Droplets)
|
||||
sort.Slice(report.Hosts, func(i, j int) bool { return report.Hosts[i].Name < report.Hosts[j].Name })
|
||||
return report, nil
|
||||
}
|
||||
|
||||
func sortDroplets(ds []DropletStatus) {
|
||||
sort.Slice(ds, func(i, j int) bool { return ds[i].Name < ds[j].Name })
|
||||
}
|
||||
|
||||
// parseSample decodes the three-marker sample script output.
|
||||
func parseSample(out string, sampleSeconds int, ds *DropletStatus) {
|
||||
func parseSample(out string, sampleSeconds int, ds *HostStatus) {
|
||||
part1, rest, ok := strings.Cut(out, "---S1---")
|
||||
if !ok {
|
||||
return
|
||||
@@ -431,7 +426,7 @@ func parseSample(out string, sampleSeconds int, ds *DropletStatus) {
|
||||
ds.TxBps = float64(after.TX-before.TX) * 8 / float64(sampleSeconds)
|
||||
ds.RxBps = float64(after.RX-before.RX) * 8 / float64(sampleSeconds)
|
||||
|
||||
var st bench.LoadgenStatus
|
||||
var st resticbench.LoadgenStatus
|
||||
if json.Unmarshal([]byte(strings.TrimSpace(part3)), &st) == nil && !st.StartedAt.IsZero() {
|
||||
ds.Status = &st
|
||||
}
|
||||
@@ -441,44 +436,44 @@ func parseSample(out string, sampleSeconds int, ds *DropletStatus) {
|
||||
}
|
||||
}
|
||||
|
||||
// Cleanup forgets and prunes every bench-do snapshot, running the agent's
|
||||
// cleanup op synchronously on each droplet (fresh URLs are minted; the
|
||||
// droplets must still exist). Refuses while load is running unless force.
|
||||
// Cleanup forgets and prunes every fleet-bench snapshot, running the agent's
|
||||
// cleanup op synchronously on each host (fresh URLs are minted; the hosts
|
||||
// must still exist). Refuses while load is running unless force.
|
||||
func (s *Session) Cleanup(ctx context.Context, minter RepoMinter, force bool) error {
|
||||
droplets, err := s.Droplets(ctx)
|
||||
hosts, err := s.Hosts(ctx)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if len(droplets) == 0 {
|
||||
return fmt.Errorf("no fleet droplets to clean from; repos can only be pruned while the fleet exists")
|
||||
if len(hosts) == 0 {
|
||||
return fmt.Errorf("no fleet hosts to clean from; repos can only be pruned while the fleet exists")
|
||||
}
|
||||
if len(s.State.Clients) == 0 {
|
||||
log.Info().Msg("no bench repositories recorded; nothing to clean")
|
||||
return nil
|
||||
}
|
||||
if !force {
|
||||
if running, err := s.anyLoadRunning(ctx, droplets); err == nil && running {
|
||||
return fmt.Errorf("load is still running; `yuctl tools bench-wide stop` first (or --force)")
|
||||
if running, err := s.anyLoadRunning(ctx, hosts); err == nil && running {
|
||||
return fmt.Errorf("load is still running; `yuctl tools fleet-bench stop` first (or --force)")
|
||||
}
|
||||
}
|
||||
byDroplet := map[string][]Client{}
|
||||
byHost := map[string][]Client{}
|
||||
for _, cl := range s.State.Clients {
|
||||
byDroplet[cl.Droplet] = append(byDroplet[cl.Droplet], cl)
|
||||
byHost[cl.Host] = append(byHost[cl.Host], cl)
|
||||
}
|
||||
return eachDroplet(droplets, func(d provider.Host) error {
|
||||
mine := byDroplet[d.Name]
|
||||
return eachHost(hosts, func(d provider.Host) error {
|
||||
mine := byHost[d.Name]
|
||||
if len(mine) == 0 {
|
||||
return nil
|
||||
}
|
||||
var lcs []bench.LoadgenClient
|
||||
var lcs []resticbench.LoadgenClient
|
||||
for _, cl := range mine {
|
||||
u, err := minter.RepositoryURL(ctx, cl.RepoID)
|
||||
if err != nil {
|
||||
return fmt.Errorf("mint URL for %s: %w", cl.RepoName, err)
|
||||
}
|
||||
lcs = append(lcs, bench.LoadgenClient{Name: cl.Name, Repo: u, Password: cl.Password})
|
||||
lcs = append(lcs, resticbench.LoadgenClient{Name: cl.Name, Repo: u, Password: cl.Password})
|
||||
}
|
||||
cfg := bench.LoadgenConfig{Op: bench.LoadgenOpCleanup, Clients: lcs, Workdir: Workdir}
|
||||
cfg := resticbench.LoadgenConfig{Op: resticbench.LoadgenOpCleanup, Clients: lcs, Workdir: Workdir}
|
||||
raw, err := json.Marshal(cfg)
|
||||
if err != nil {
|
||||
return err
|
||||
@@ -490,8 +485,7 @@ func (s *Session) Cleanup(ctx context.Context, minter RepoMinter, force bool) er
|
||||
// streamCleanup runs the agent cleanup op over a live ssh session, logging
|
||||
// its event stream.
|
||||
func (s *Session) streamCleanup(ctx context.Context, d provider.Host, cfg []byte) error {
|
||||
cmd := exec.CommandContext(ctx, "ssh",
|
||||
s.sshArgs(s.Provider.SSHUser()+"@"+d.PublicIP, "$HOME/"+bench.RemoteBinDir+"/bench-agent --loadgen")...)
|
||||
cmd := s.ssh.Command(ctx, d.PublicIP, "$HOME/"+resticbench.RemoteBinDir+"/bench-agent --loadgen")
|
||||
cmd.Stdin = strings.NewReader(string(cfg))
|
||||
cmd.Stderr = os.Stderr
|
||||
stdout, err := cmd.StdoutPipe()
|
||||
@@ -502,27 +496,21 @@ func (s *Session) streamCleanup(ctx context.Context, d provider.Host, cfg []byte
|
||||
return err
|
||||
}
|
||||
var fatal string
|
||||
sc := bufio.NewScanner(stdout)
|
||||
sc.Buffer(make([]byte, 1<<20), 8<<20)
|
||||
for sc.Scan() {
|
||||
var ev bench.Event
|
||||
if json.Unmarshal(sc.Bytes(), &ev) != nil {
|
||||
continue
|
||||
}
|
||||
_ = resticbench.ScanEvents(stdout, func(ev resticbench.Event) {
|
||||
switch ev.Type {
|
||||
case "phase_start":
|
||||
log.Info().Str("droplet", d.Name).Msgf("%s: started", ev.Phase)
|
||||
log.Info().Str("host", d.Name).Msgf("%s: started", ev.Phase)
|
||||
case "phase_done":
|
||||
e := log.Info().Str("droplet", d.Name)
|
||||
e := log.Info().Str("host", d.Name)
|
||||
if ev.PhaseResult != nil {
|
||||
e = e.Str("duration", bench.FormatDuration(ev.PhaseResult.Seconds)).
|
||||
e = e.Str("duration", resticbench.FormatDuration(ev.PhaseResult.Seconds)).
|
||||
Int64("snapshots", ev.PhaseResult.Bytes)
|
||||
}
|
||||
e.Msgf("%s: done", ev.Phase)
|
||||
case "fatal":
|
||||
fatal = ev.Message
|
||||
}
|
||||
}
|
||||
})
|
||||
if err := cmd.Wait(); err != nil {
|
||||
if fatal != "" {
|
||||
return fmt.Errorf("cleanup on %s: %s", d.Name, fatal)
|
||||
+43
-122
@@ -1,34 +1,35 @@
|
||||
// Package benchwide deploys and drives a fleet of cloud VMs — across
|
||||
// providers (DigitalOcean, Hetzner, …; see internal/provider) — running real
|
||||
// Package fleetbench deploys and drives a fleet of cloud VMs — across
|
||||
// providers (DigitalOcean, Hetzner, …; see yuctl/provider) — running real
|
||||
// restic clients against michael, the external-user data path over the public
|
||||
// internet. It is the VM-shaped sibling of internal/warp (fleet lifecycle:
|
||||
// internet. It is the VM-shaped sibling of fleet/warp (fleet lifecycle:
|
||||
// deploy/start/status/stop/cleanup/undeploy) built on the bench agent's
|
||||
// loadgen mode: every host runs a detached supervisor looping seeded
|
||||
// generate→backup cycles per client, hard-capped at the host's transfer
|
||||
// allowance so a forgotten run cannot burn into paid overage. Each provider is
|
||||
// a separate, independently-addressable fleet (state keyed by partition ×
|
||||
// provider), so multiple providers can load michael concurrently.
|
||||
package benchwide
|
||||
package fleetbench
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"crypto/rand"
|
||||
"encoding/hex"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
yctx "yuctl/internal/context"
|
||||
"yuctl/internal/provider"
|
||||
"yuctl/ctxstore"
|
||||
"yuctl/fleet"
|
||||
"yuctl/provider"
|
||||
"yuctl/sshx"
|
||||
)
|
||||
|
||||
// Legacy identifiers from the tool's bench-do/bench-wide eras survive below
|
||||
// (Workdir, the tag prefix, repo prefixes, the state dir and JSON tags):
|
||||
// changing any of them would orphan deployed hosts, persisted repo passwords,
|
||||
// or remote state on live fleets.
|
||||
const (
|
||||
// Workdir is the host-side scratch root: datasets, restic caches, the
|
||||
// loadgen config (transient) and status file, and the agent log.
|
||||
@@ -45,11 +46,11 @@ const (
|
||||
)
|
||||
|
||||
// Client is one restic identity: its admin-api repository and password,
|
||||
// pinned to a droplet by name. Passwords are generated once and persisted —
|
||||
// pinned to a host by name. Passwords are generated once and persisted —
|
||||
// without them the repos can never be reopened (cleanup, restarts).
|
||||
type Client struct {
|
||||
Name string `json:"name"` // <droplet>-c<n>
|
||||
Droplet string `json:"droplet"` // droplet name it runs on
|
||||
Name string `json:"name"` // <host>-c<n>
|
||||
Host string `json:"droplet"`
|
||||
RepoID string `json:"repoId"`
|
||||
RepoName string `json:"repoName"`
|
||||
Password string `json:"password"`
|
||||
@@ -80,7 +81,7 @@ type State struct {
|
||||
Run *RunInfo `json:"run,omitempty"`
|
||||
}
|
||||
|
||||
// Session is the resolved handle for one bench-wide command invocation,
|
||||
// Session is the resolved handle for one fleet-bench command invocation,
|
||||
// scoped to one provider × partition fleet.
|
||||
type Session struct {
|
||||
Partition string
|
||||
@@ -89,6 +90,7 @@ type Session struct {
|
||||
|
||||
providerName string // slug used in the fleet tag + local filenames
|
||||
dir string // ${XDG_CONFIG_HOME}/yuctl/bench-wide
|
||||
ssh *sshx.Client
|
||||
}
|
||||
|
||||
// NewSession builds the named provider's client (env/op token) and loads any
|
||||
@@ -98,7 +100,7 @@ func NewSession(ctx context.Context, partition, providerName string) (*Session,
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
base, err := yctx.Dir()
|
||||
base, err := ctxstore.Dir()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
@@ -106,6 +108,16 @@ func NewSession(ctx context.Context, partition, providerName string) (*Session,
|
||||
if err := os.MkdirAll(filepath.Join(s.dir, "cm"), 0o700); err != nil {
|
||||
return nil, fmt.Errorf("create ssh control dir: %w", err)
|
||||
}
|
||||
// Per-fleet known_hosts: provider IPs get recycled across deployments —
|
||||
// the operator's global file would scream host-key-changed.
|
||||
s.ssh = &sshx.Client{
|
||||
User: prov.SSHUser(),
|
||||
IdentityFile: s.PrivateKeyPath(),
|
||||
KnownHostsFile: s.knownHostsPath(),
|
||||
ControlDir: filepath.Join(s.dir, "cm"),
|
||||
ConnectTimeoutSeconds: 10,
|
||||
Retries: 2,
|
||||
}
|
||||
if err := s.loadState(); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
@@ -165,10 +177,10 @@ func (s *Session) SaveState() error {
|
||||
return os.Rename(tmp, p)
|
||||
}
|
||||
|
||||
// clearFleetState drops everything droplet-bound (ssh key material, run
|
||||
// info) but keeps the client records when repos exist: their passwords are
|
||||
// the only way to ever reopen those repos (e.g. `tools bench --repo-id` with
|
||||
// a re-minted URL to prune them later).
|
||||
// clearFleetState drops everything host-bound (ssh key material, run info)
|
||||
// but keeps the client records when repos exist: their passwords are the only
|
||||
// way to ever reopen those repos (e.g. `tools bench --repo-id` with a
|
||||
// re-minted URL to prune them later).
|
||||
func (s *Session) clearFleetState() error {
|
||||
os.Remove(s.PrivateKeyPath())
|
||||
os.Remove(s.knownHostsPath())
|
||||
@@ -183,111 +195,20 @@ func (s *Session) clearFleetState() error {
|
||||
return s.SaveState()
|
||||
}
|
||||
|
||||
// sshArgs are the fleet ssh options: the ephemeral identity, a per-fleet
|
||||
// known_hosts (droplet IPs get recycled across deployments — the global file
|
||||
// would scream host-key-changed), TOFU on first contact, and connection
|
||||
// multiplexing. Multiplexing matters beyond latency: status/watch/start
|
||||
// otherwise open a fresh port-22 TCP flow per droplet per command, a burst
|
||||
// pattern that trips ssh-targeted rate limiting on some paths (observed as
|
||||
// flapping per-IP port-22 SYN drops while ICMP and other ports stay fine).
|
||||
// One persistent master per droplet keeps the flow count flat.
|
||||
func (s *Session) sshArgs(extra ...string) []string {
|
||||
args := []string{
|
||||
"-o", "BatchMode=yes",
|
||||
"-o", "StrictHostKeyChecking=accept-new",
|
||||
"-o", "UserKnownHostsFile=" + s.knownHostsPath(),
|
||||
"-o", "ConnectTimeout=10",
|
||||
"-o", "ServerAliveInterval=30",
|
||||
"-o", "ServerAliveCountMax=8",
|
||||
"-o", "ControlMaster=auto",
|
||||
"-o", "ControlPath=" + filepath.Join(s.dir, "cm", "%C"),
|
||||
"-o", "ControlPersist=300",
|
||||
"-i", s.PrivateKeyPath(),
|
||||
"-o", "IdentitiesOnly=yes",
|
||||
}
|
||||
return append(args, extra...)
|
||||
}
|
||||
|
||||
// sshRun executes script on the droplet, returning combined stdout. stdin may
|
||||
// be nil. The script travels as the ssh command argument (visible in remote
|
||||
// ps — it must never contain secrets); secret payloads go through stdin.
|
||||
// Connection-level failures (ssh exit 255: banner timeouts, resets — common
|
||||
// when tens of sessions open against fresh droplets) are retried; remote
|
||||
// command failures are not, ssh reports those as the command's own exit code.
|
||||
func (s *Session) sshRun(ctx context.Context, ip, script string, stdin []byte) (string, error) {
|
||||
var lastOut string
|
||||
var lastErr error
|
||||
for attempt := 1; attempt <= 3; attempt++ {
|
||||
out, err := s.sshRunOnce(ctx, ip, script, stdin)
|
||||
if err == nil {
|
||||
return out, nil
|
||||
}
|
||||
lastOut, lastErr = out, err
|
||||
var exit *exec.ExitError
|
||||
if ctx.Err() != nil || !errors.As(err, &exit) || exit.ExitCode() != 255 {
|
||||
break
|
||||
}
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return lastOut, lastErr
|
||||
case <-time.After(time.Duration(attempt) * 5 * time.Second):
|
||||
}
|
||||
}
|
||||
return lastOut, lastErr
|
||||
}
|
||||
|
||||
// sshRunOnce is a single ssh attempt with no retry — used by callers that
|
||||
// implement their own retry cadence (waitSSH), where nesting retries would
|
||||
// multiply the delays.
|
||||
func (s *Session) sshRunOnce(ctx context.Context, ip, script string, stdin []byte) (string, error) {
|
||||
cmd := exec.CommandContext(ctx, "ssh", s.sshArgs(s.Provider.SSHUser()+"@"+ip, script)...)
|
||||
if stdin != nil {
|
||||
cmd.Stdin = bytes.NewReader(stdin)
|
||||
}
|
||||
var out, errb bytes.Buffer
|
||||
cmd.Stdout = &out
|
||||
cmd.Stderr = &errb
|
||||
if err := cmd.Run(); err != nil {
|
||||
return out.String(), fmt.Errorf("ssh %s: %w: %s", ip, err, tailStr(errb.String(), 1000))
|
||||
}
|
||||
return out.String(), nil
|
||||
}
|
||||
|
||||
func tailStr(v string, n int) string {
|
||||
v = strings.TrimSpace(v)
|
||||
if len(v) > n {
|
||||
return "…" + v[len(v)-n:]
|
||||
}
|
||||
return v
|
||||
}
|
||||
|
||||
// Droplets returns the live fleet from the provider's tag listing. (Named for
|
||||
// the historical DO fleet; a host is a host on any provider.)
|
||||
func (s *Session) Droplets(ctx context.Context) ([]provider.Host, error) {
|
||||
// Hosts returns the live fleet from the provider's tag listing.
|
||||
func (s *Session) Hosts(ctx context.Context) ([]provider.Host, error) {
|
||||
return s.Provider.List(ctx, s.Tag())
|
||||
}
|
||||
|
||||
// eachDroplet runs fn against every host in parallel — staggered by 150ms each
|
||||
// so a big fleet doesn't fire all its port-22 SYNs in one burst (see sshArgs)
|
||||
// — and returns the first error (all hosts are still attempted).
|
||||
func eachDroplet(hosts []provider.Host, fn func(d provider.Host) error) error {
|
||||
var wg sync.WaitGroup
|
||||
errs := make([]error, len(hosts))
|
||||
for i, d := range hosts {
|
||||
wg.Add(1)
|
||||
go func(i int, d provider.Host) {
|
||||
defer wg.Done()
|
||||
time.Sleep(time.Duration(i) * 150 * time.Millisecond)
|
||||
errs[i] = fn(d)
|
||||
}(i, d)
|
||||
}
|
||||
wg.Wait()
|
||||
for _, err := range errs {
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
return nil
|
||||
// eachHost runs fn against every host in parallel — staggered by 150ms each
|
||||
// so a big fleet doesn't fire all its port-22 SYNs in one burst (see
|
||||
// sshx.Client.ControlDir) — and returns the first error (all hosts are still
|
||||
// attempted).
|
||||
func eachHost(hosts []provider.Host, fn func(d provider.Host) error) error {
|
||||
return fleet.Each(len(hosts), func(i int) error {
|
||||
time.Sleep(time.Duration(i) * 150 * time.Millisecond)
|
||||
return fn(hosts[i])
|
||||
})
|
||||
}
|
||||
|
||||
func randHex(n int) string {
|
||||
@@ -25,6 +25,7 @@ type CleanupOptions struct {
|
||||
Force bool // purge even while load is running
|
||||
Keep bool // keep the finished Job around for inspection
|
||||
Timeout time.Duration
|
||||
JobLogs io.Writer // where the finished Job's logs go (nil = discard)
|
||||
}
|
||||
|
||||
// Cleanup purges warp test buckets via a hostNetwork Job running `mc` inside
|
||||
@@ -130,8 +131,8 @@ echo cleanup done`, mci, target.Endpoint, strings.Join(patterns, "|"), strings.J
|
||||
return false, nil
|
||||
})
|
||||
|
||||
if logs := s.jobLogs(ctx, name); logs != "" {
|
||||
fmt.Println(logs)
|
||||
if logs := s.jobLogs(ctx, name); logs != "" && opts.JobLogs != nil {
|
||||
fmt.Fprintln(opts.JobLogs, logs)
|
||||
}
|
||||
if waitErr != nil {
|
||||
return fmt.Errorf("cleanup job did not complete (kept for inspection): %w", waitErr)
|
||||
@@ -12,7 +12,8 @@ import (
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
|
||||
"yuctl/internal/netdev"
|
||||
"yuctl/fleet"
|
||||
"yuctl/netdev"
|
||||
)
|
||||
|
||||
// StartOptions shape the load. Zero values reproduce the proven per-pod shape
|
||||
@@ -117,52 +118,42 @@ func (s *Session) Start(ctx context.Context, opts StartOptions) error {
|
||||
Str("obj_size", opts.PutObjSize).Int("rgw_endpoints", len(ips)).Bool("nonstop", nonstop).
|
||||
Msg("relaunching load on all pods in parallel (kill previous, then launch)")
|
||||
|
||||
var wg sync.WaitGroup
|
||||
errs := make([]error, len(pods))
|
||||
for i, pod := range pods {
|
||||
if err := fleet.Each(len(pods), func(i int) error {
|
||||
pod := pods[i]
|
||||
put := share(totalPut, len(pods), i)
|
||||
get := share(totalGet, len(pods), i)
|
||||
wg.Add(1)
|
||||
go func(i int, pod RunnerPod, put, get int) {
|
||||
defer wg.Done()
|
||||
putCmd := fmt.Sprintf(`/warp put --host=%s --host-select=roundrobin%s --bucket=%sput-%s `+
|
||||
`--obj.size=%s --duration=%s --concurrent=%d --noclear --no-color >> /tmp/warp-put.log 2>&1`,
|
||||
hostList, tlsFlags, opts.BucketPrefix, pod.Name, opts.PutObjSize, runDuration, put)
|
||||
getCmd := fmt.Sprintf(`/warp get --host=%s --host-select=roundrobin%s --bucket=%sget-%s `+
|
||||
`--objects=%d --obj.size=%s --duration=%s --concurrent=%d --noclear --no-color >> /tmp/warp-get.log 2>&1`,
|
||||
hostList, tlsFlags, opts.BucketPrefix, pod.Name, opts.GetObjects, opts.GetObjSize, runDuration, get)
|
||||
putCmd := fmt.Sprintf(`/warp put --host=%s --host-select=roundrobin%s --bucket=%sput-%s `+
|
||||
`--obj.size=%s --duration=%s --concurrent=%d --noclear --no-color >> /tmp/warp-put.log 2>&1`,
|
||||
hostList, tlsFlags, opts.BucketPrefix, pod.Name, opts.PutObjSize, runDuration, put)
|
||||
getCmd := fmt.Sprintf(`/warp get --host=%s --host-select=roundrobin%s --bucket=%sget-%s `+
|
||||
`--objects=%d --obj.size=%s --duration=%s --concurrent=%d --noclear --no-color >> /tmp/warp-get.log 2>&1`,
|
||||
hostList, tlsFlags, opts.BucketPrefix, pod.Name, opts.GetObjects, opts.GetObjSize, runDuration, get)
|
||||
|
||||
// Kill and launch are separate execs: the launch script's literal
|
||||
// "/warp put ..." text would otherwise match the kill patterns in
|
||||
// the same command line and SIGTERM the script itself.
|
||||
if _, err := s.podExec(ctx, pod.Name, killScript); err != nil {
|
||||
if s.podGone(ctx, pod.Name) {
|
||||
log.Warn().Str("pod", pod.Name).Msg("pod terminated mid-start; skipping it")
|
||||
return
|
||||
}
|
||||
errs[i] = fmt.Errorf("stop previous load on %s: %w", pod.Name, err)
|
||||
return
|
||||
// Kill and launch are separate execs: the launch script's literal
|
||||
// "/warp put ..." text would otherwise match the kill patterns in
|
||||
// the same command line and SIGTERM the script itself.
|
||||
if _, err := s.podExec(ctx, pod.Name, killScript); err != nil {
|
||||
if s.podGone(ctx, pod.Name) {
|
||||
log.Warn().Str("pod", pod.Name).Msg("pod terminated mid-start; skipping it")
|
||||
return nil
|
||||
}
|
||||
var script strings.Builder
|
||||
if put > 0 {
|
||||
script.WriteString(launchLine("put", putCmd, nonstop))
|
||||
}
|
||||
if get > 0 {
|
||||
script.WriteString(launchLine("get", getCmd, nonstop))
|
||||
}
|
||||
script.WriteString("true\n")
|
||||
if _, err := s.podExec(ctx, pod.Name, script.String()); err != nil {
|
||||
errs[i] = fmt.Errorf("start load on %s: %w", pod.Name, err)
|
||||
return
|
||||
}
|
||||
log.Info().Str("pod", pod.Name).Str("node", pod.Node).Int("put", put).Int("get", get).Msg("load launched")
|
||||
}(i, pod, put, get)
|
||||
}
|
||||
wg.Wait()
|
||||
for _, err := range errs {
|
||||
if err != nil {
|
||||
return err
|
||||
return fmt.Errorf("stop previous load on %s: %w", pod.Name, err)
|
||||
}
|
||||
var script strings.Builder
|
||||
if put > 0 {
|
||||
script.WriteString(launchLine("put", putCmd, nonstop))
|
||||
}
|
||||
if get > 0 {
|
||||
script.WriteString(launchLine("get", getCmd, nonstop))
|
||||
}
|
||||
script.WriteString("true\n")
|
||||
if _, err := s.podExec(ctx, pod.Name, script.String()); err != nil {
|
||||
return fmt.Errorf("start load on %s: %w", pod.Name, err)
|
||||
}
|
||||
log.Info().Str("pod", pod.Name).Str("node", pod.Node).Int("put", put).Int("get", get).Msg("load launched")
|
||||
return nil
|
||||
}); err != nil {
|
||||
return err
|
||||
}
|
||||
log.Info().Msg("all pods launched; `yuctl tools warp watch` for the live dashboard")
|
||||
|
||||
@@ -223,22 +214,13 @@ func (s *Session) Stop(ctx context.Context) error {
|
||||
return nil
|
||||
}
|
||||
log.Info().Int("pods", len(pods)).Msg("killing warp processes on all pods in parallel")
|
||||
var wg sync.WaitGroup
|
||||
errs := make([]error, len(pods))
|
||||
for i, pod := range pods {
|
||||
wg.Add(1)
|
||||
go func(i int, pod RunnerPod) {
|
||||
defer wg.Done()
|
||||
if _, err := s.podExec(ctx, pod.Name, killScript); err != nil {
|
||||
errs[i] = fmt.Errorf("stop load on %s: %w", pod.Name, err)
|
||||
}
|
||||
}(i, pod)
|
||||
}
|
||||
wg.Wait()
|
||||
for _, err := range errs {
|
||||
if err != nil {
|
||||
return err
|
||||
if err := fleet.Each(len(pods), func(i int) error {
|
||||
if _, err := s.podExec(ctx, pods[i].Name, killScript); err != nil {
|
||||
return fmt.Errorf("stop load on %s: %w", pods[i].Name, err)
|
||||
}
|
||||
return nil
|
||||
}); err != nil {
|
||||
return err
|
||||
}
|
||||
log.Info().Msg("verifying nothing survived")
|
||||
if running, err := s.anyLoadRunning(ctx); err == nil && running {
|
||||
@@ -254,27 +236,19 @@ func (s *Session) anyLoadRunning(ctx context.Context) (bool, error) {
|
||||
if err != nil {
|
||||
return false, err
|
||||
}
|
||||
var wg sync.WaitGroup
|
||||
counts := make([]int, len(pods))
|
||||
errs := make([]error, len(pods))
|
||||
for i, pod := range pods {
|
||||
wg.Add(1)
|
||||
go func(i int, pod RunnerPod) {
|
||||
defer wg.Done()
|
||||
out, err := s.podExec(ctx, pod.Name, `pgrep -f '/warp ([p]ut|[g]et)' 2>/dev/null | wc -l`)
|
||||
if err != nil {
|
||||
errs[i] = err
|
||||
return
|
||||
}
|
||||
counts[i], _ = strconv.Atoi(strings.TrimSpace(out))
|
||||
}(i, pod)
|
||||
}
|
||||
wg.Wait()
|
||||
for i, err := range errs {
|
||||
if err := fleet.Each(len(pods), func(i int) error {
|
||||
out, err := s.podExec(ctx, pods[i].Name, `pgrep -f '/warp ([p]ut|[g]et)' 2>/dev/null | wc -l`)
|
||||
if err != nil {
|
||||
return false, err
|
||||
return err
|
||||
}
|
||||
if counts[i] > 0 {
|
||||
counts[i], _ = strconv.Atoi(strings.TrimSpace(out))
|
||||
return nil
|
||||
}); err != nil {
|
||||
return false, err
|
||||
}
|
||||
for _, n := range counts {
|
||||
if n > 0 {
|
||||
return true, nil
|
||||
}
|
||||
}
|
||||
@@ -320,14 +294,22 @@ func (s *Session) Status(ctx context.Context, sampleSeconds int) (*StatusReport,
|
||||
report.Config = cm.Data
|
||||
}
|
||||
|
||||
var mu sync.Mutex
|
||||
var wg sync.WaitGroup
|
||||
var nodePods []RunnerPod
|
||||
seen := map[string]bool{}
|
||||
for _, pod := range pods {
|
||||
if !seen[pod.Node] {
|
||||
seen[pod.Node] = true
|
||||
nodePods = append(nodePods, pod)
|
||||
}
|
||||
}
|
||||
|
||||
// One fan-out over pod-status probes and per-node NIC samples together, so
|
||||
// the fast probes overlap the sampleSeconds-long NIC windows.
|
||||
var mu sync.Mutex
|
||||
report.Pods = make([]PodStatus, len(pods))
|
||||
for i, pod := range pods {
|
||||
wg.Add(1)
|
||||
go func(i int, pod RunnerPod) {
|
||||
defer wg.Done()
|
||||
_ = fleet.Each(len(pods)+len(nodePods), func(i int) error {
|
||||
if i < len(pods) {
|
||||
pod := pods[i]
|
||||
ps := PodStatus{Name: pod.Name, Node: pod.Node}
|
||||
out, err := s.podExec(ctx, pod.Name,
|
||||
`echo "$(pgrep -f '/warp [p]ut' 2>/dev/null | wc -l) `+
|
||||
@@ -348,32 +330,20 @@ func (s *Session) Status(ctx context.Context, sampleSeconds int) (*StatusReport,
|
||||
ps.LastPut = strings.TrimSpace(lines[1])
|
||||
}
|
||||
}
|
||||
mu.Lock()
|
||||
report.Pods[i] = ps
|
||||
mu.Unlock()
|
||||
}(i, pod)
|
||||
}
|
||||
|
||||
seen := map[string]bool{}
|
||||
for _, pod := range pods {
|
||||
if seen[pod.Node] {
|
||||
continue
|
||||
return nil
|
||||
}
|
||||
seen[pod.Node] = true
|
||||
wg.Add(1)
|
||||
go func(pod RunnerPod) {
|
||||
defer wg.Done()
|
||||
nt, err := s.sampleNode(ctx, pod, sampleSeconds)
|
||||
if err != nil {
|
||||
log.Warn().Err(err).Str("node", pod.Node).Msg("throughput sample failed")
|
||||
return
|
||||
}
|
||||
mu.Lock()
|
||||
report.Nodes = append(report.Nodes, *nt)
|
||||
mu.Unlock()
|
||||
}(pod)
|
||||
}
|
||||
wg.Wait()
|
||||
pod := nodePods[i-len(pods)]
|
||||
nt, err := s.sampleNode(ctx, pod, sampleSeconds)
|
||||
if err != nil {
|
||||
log.Warn().Err(err).Str("node", pod.Node).Msg("throughput sample failed")
|
||||
return nil
|
||||
}
|
||||
mu.Lock()
|
||||
report.Nodes = append(report.Nodes, *nt)
|
||||
mu.Unlock()
|
||||
return nil
|
||||
})
|
||||
return report, nil
|
||||
}
|
||||
|
||||
@@ -31,8 +31,8 @@ import (
|
||||
"k8s.io/client-go/tools/clientcmd"
|
||||
"k8s.io/client-go/tools/remotecommand"
|
||||
|
||||
"yuctl/internal/op"
|
||||
"yuctl/internal/state"
|
||||
"yuctl/op"
|
||||
"yuctl/state"
|
||||
)
|
||||
|
||||
// fieldManager identifies this tool's server-side applies.
|
||||
@@ -1,169 +0,0 @@
|
||||
package cli
|
||||
|
||||
import (
|
||||
"context"
|
||||
"crypto/tls"
|
||||
"fmt"
|
||||
"net/http"
|
||||
"net/url"
|
||||
"os"
|
||||
"os/exec"
|
||||
"runtime"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/spf13/cobra"
|
||||
|
||||
"yuctl/internal/adminapi"
|
||||
yctx "yuctl/internal/context"
|
||||
"yuctl/internal/discovery"
|
||||
)
|
||||
|
||||
// adminFlags is the shared flag set for commands that talk to the admin-api.
|
||||
type adminFlags struct {
|
||||
adminURL string
|
||||
insecure bool
|
||||
reauth bool
|
||||
noBrowser bool
|
||||
}
|
||||
|
||||
func (f *adminFlags) register(c *cobra.Command) {
|
||||
c.Flags().StringVar(&f.adminURL, "admin-url", "", "admin-api base URL (default: derived from discovery, or $YUCTL_ADMIN_API_URL)")
|
||||
c.Flags().BoolVar(&f.insecure, "insecure-skip-tls-verify", false, "skip TLS verification")
|
||||
c.Flags().BoolVar(&f.reauth, "reauth", false, "force a fresh browser login")
|
||||
c.Flags().BoolVar(&f.noBrowser, "no-browser", false, "do not auto-open the browser; just print the login URL")
|
||||
}
|
||||
|
||||
func (f *adminFlags) httpClient() *http.Client {
|
||||
hc := &http.Client{Timeout: 30 * time.Second}
|
||||
if f.insecure {
|
||||
tr := http.DefaultTransport.(*http.Transport).Clone()
|
||||
tr.TLSClientConfig = &tls.Config{InsecureSkipVerify: true}
|
||||
hc.Transport = tr
|
||||
}
|
||||
return hc
|
||||
}
|
||||
|
||||
// deriveAdminURL builds the admin-api origin for the partition's primary
|
||||
// region from its k8s discovery payload: the Talos API endpoint lives at
|
||||
// kube.<cluster>.<region>.<provider>.yucca.futo.network, and the admin-api is
|
||||
// published on the same NetBird overlay zone as admin.<...> (the
|
||||
// YUCCA_ADMIN_HOST cluster-setting follows the same convention).
|
||||
func deriveAdminURL(topo *discovery.Topology, partition, region string) (string, error) {
|
||||
k8s := topo.Kubernetes(partition, region)
|
||||
if k8s == nil || k8s.APIEndpoint == "" {
|
||||
return "", fmt.Errorf("no kubernetes discovery payload for %s@%s; pass --admin-url or set YUCTL_ADMIN_API_URL", partition, region)
|
||||
}
|
||||
u, err := url.Parse(k8s.APIEndpoint)
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("parse api_endpoint %q: %w", k8s.APIEndpoint, err)
|
||||
}
|
||||
host, ok := strings.CutPrefix(u.Hostname(), "kube.")
|
||||
if !ok {
|
||||
return "", fmt.Errorf("api_endpoint host %q does not start with kube.; pass --admin-url or set YUCTL_ADMIN_API_URL", u.Hostname())
|
||||
}
|
||||
return "https://admin." + host, nil
|
||||
}
|
||||
|
||||
// resolveAdminURL applies the override chain: --admin-url > $YUCTL_ADMIN_API_URL
|
||||
// > derived from the primary region's discovery.
|
||||
func (f *adminFlags) resolveAdminURL(topo *discovery.Topology, cc *yctx.Context) (string, error) {
|
||||
if f.adminURL != "" {
|
||||
return f.adminURL, nil
|
||||
}
|
||||
if env := os.Getenv("YUCTL_ADMIN_API_URL"); env != "" {
|
||||
return env, nil
|
||||
}
|
||||
primary := topo.PrimaryRegion(cc.Partition)
|
||||
if primary == "" {
|
||||
return "", fmt.Errorf("no primary region found for partition %q (no stack with discovery.role==\"primary\")", cc.Partition)
|
||||
}
|
||||
return deriveAdminURL(topo, cc.Partition, primary)
|
||||
}
|
||||
|
||||
// adminLogin returns an authenticated admin-api client, reusing the cached
|
||||
// per-partition session when valid and running the browser loopback flow
|
||||
// otherwise.
|
||||
func (f *adminFlags) adminLogin(ctx context.Context, cmd *cobra.Command, cc *yctx.Context, topo *discovery.Topology) (*adminapi.Client, *adminapi.Token, error) {
|
||||
adminURL, err := f.resolveAdminURL(topo, cc)
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
hc := f.httpClient()
|
||||
|
||||
token, err := adminapi.LoadToken(cc.Partition)
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
if f.reauth || !token.Valid() {
|
||||
var openFn func(string) error
|
||||
if !f.noBrowser {
|
||||
openFn = openBrowser
|
||||
}
|
||||
fresh, err := adminapi.BrowserLogin(ctx, hc, adminURL, openFn, func(loginURL string) {
|
||||
out := cmd.ErrOrStderr()
|
||||
fmt.Fprintln(out, "Complete the login in your browser:")
|
||||
fmt.Fprintf(out, " %s\n", loginURL)
|
||||
})
|
||||
if err != nil {
|
||||
return nil, nil, fmt.Errorf("browser login: %w", err)
|
||||
}
|
||||
token = *fresh
|
||||
if err := adminapi.SaveToken(cc.Partition, token); err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
}
|
||||
|
||||
return adminapi.NewClient(adminURL, token, hc), &token, nil
|
||||
}
|
||||
|
||||
// openBrowser opens url in the OS default browser, best-effort.
|
||||
func openBrowser(url string) error {
|
||||
switch runtime.GOOS {
|
||||
case "darwin":
|
||||
return exec.Command("open", url).Start()
|
||||
default:
|
||||
return exec.Command("xdg-open", url).Start()
|
||||
}
|
||||
}
|
||||
|
||||
func newLoginCmd() *cobra.Command {
|
||||
flags := &adminFlags{}
|
||||
c := &cobra.Command{
|
||||
Use: "login",
|
||||
Short: "Log in to the partition's yucca-admin-api via the browser",
|
||||
Long: "Authenticate against the selected partition's admin-api (primary region):\n" +
|
||||
"opens the admin-api CLI login in your browser, receives a one-time code on a\n" +
|
||||
"127.0.0.1 listener, exchanges it for a 24h session JWT, and caches it at\n" +
|
||||
"${XDG_CONFIG_HOME:-~/.config}/yuctl/admin-token-<partition>.json.",
|
||||
Args: cobra.NoArgs,
|
||||
RunE: func(cmd *cobra.Command, _ []string) error {
|
||||
ctx := cmd.Context()
|
||||
cc, err := requireContext()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
topo, err := resolveTopology(ctx)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
client, token, err := flags.adminLogin(ctx, cmd, cc, topo)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
// Round-trip the session so "login" only succeeds when the API
|
||||
// actually accepts the token.
|
||||
sub, err := client.GetAuth(ctx)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
fmt.Fprintf(cmd.OutOrStdout(), "logged in to %s as %s (expires %s)\n",
|
||||
cc.Partition, sub, token.Expiry.Local().Format(time.RFC3339))
|
||||
return nil
|
||||
},
|
||||
}
|
||||
flags.register(c)
|
||||
return c
|
||||
}
|
||||
@@ -1,227 +0,0 @@
|
||||
package cli
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
"strings"
|
||||
"text/tabwriter"
|
||||
"time"
|
||||
|
||||
"yuctl/internal/bench"
|
||||
"yuctl/internal/benchwide"
|
||||
)
|
||||
|
||||
// benchDoView renders StatusReports as a styled dashboard; watch mode feeds
|
||||
// it successive samples and it keeps a TX history for the sparkline. Styling
|
||||
// and the meter/sparkline primitives are shared with the warp view.
|
||||
type benchDoView struct {
|
||||
label string
|
||||
history []float64 // combined TX bps per sample
|
||||
}
|
||||
|
||||
func (v *benchDoView) push(combined float64) {
|
||||
v.history = append(v.history, combined)
|
||||
if len(v.history) > 60 {
|
||||
v.history = v.history[len(v.history)-60:]
|
||||
}
|
||||
}
|
||||
|
||||
func (v *benchDoView) render(r *benchwide.StatusReport, sampledAt time.Time, sampleSec int, watching bool) string {
|
||||
var b strings.Builder
|
||||
|
||||
b.WriteString(warpBadge.Render("BENCH-WIDE") + " " + warpTitle.Render(v.label) + "\n")
|
||||
|
||||
if len(r.Droplets) == 0 {
|
||||
b.WriteString(warpWarnSt.Render("no droplets deployed") +
|
||||
warpDimSt.Render(" — yuctl tools bench-wide deploy") + "\n")
|
||||
return warpFrame.Render(strings.TrimRight(b.String(), "\n"))
|
||||
}
|
||||
|
||||
if r.Run != nil {
|
||||
since := time.Since(r.Run.StartedAt).Round(time.Second).String() + " ago"
|
||||
p := r.Run.Params
|
||||
b.WriteString(fmt.Sprintf("%s %s · %s objects · %s/cycle · %s clients/droplet · %s\n",
|
||||
warpOKSt.Render("● "+r.Run.Label),
|
||||
warpDimSt.Render("started "+since),
|
||||
p["obj_size_mib"]+"MiB", p["cycle_size"], p["clients_per_droplet"], p["duration"]))
|
||||
b.WriteString(warpDimSt.Render("cap "+p["cap_per_droplet"]+" per droplet") + "\n")
|
||||
} else {
|
||||
b.WriteString(warpWarnSt.Render("○ no active run recorded") +
|
||||
warpDimSt.Render(" — load stopped or never started") + "\n")
|
||||
}
|
||||
b.WriteString("\n")
|
||||
|
||||
// Droplets: TX bar scaled to the busiest droplet, transfer budget bar
|
||||
// scaled to the cap.
|
||||
var maxBps float64
|
||||
nameW := 7
|
||||
for _, d := range r.Droplets {
|
||||
maxBps = max(maxBps, d.TxBps)
|
||||
nameW = max(nameW, len(d.Name))
|
||||
}
|
||||
var wire, capTotal int64
|
||||
var txBps float64
|
||||
for _, d := range r.Droplets {
|
||||
capBytes := r.TransferCap
|
||||
var used int64
|
||||
state := "-"
|
||||
if d.Status != nil {
|
||||
used = d.Status.WireTxBytes
|
||||
if d.Status.CapBytes > 0 {
|
||||
capBytes = d.Status.CapBytes
|
||||
}
|
||||
state = d.Status.State
|
||||
}
|
||||
budget := " "
|
||||
if capBytes > 0 {
|
||||
pct := float64(used) / float64(capBytes) * 100
|
||||
st := warpOKSt
|
||||
switch {
|
||||
case pct >= 90:
|
||||
st = warpErrSt
|
||||
case pct >= 60:
|
||||
st = warpWarnSt
|
||||
}
|
||||
budget = fmt.Sprintf("%s %s", st.Render(meter(float64(used), float64(capBytes), 10)), st.Render(fmt.Sprintf("%3.0f%%", pct)))
|
||||
}
|
||||
line := fmt.Sprintf("%-*s %s %s %s %s %s %s",
|
||||
nameW, d.Name,
|
||||
warpDimSt.Render(fmt.Sprintf("%-5s", d.Region)),
|
||||
warpTxSt.Render(meter(d.TxBps, maxBps, 14)),
|
||||
padGbps(d.TxBps),
|
||||
budget,
|
||||
warpDimSt.Render(fmt.Sprintf("%9s", bench.FormatBytes(used))),
|
||||
stateCell(d, state))
|
||||
b.WriteString(line + "\n")
|
||||
txBps += d.TxBps
|
||||
wire += used
|
||||
capTotal += capBytes
|
||||
}
|
||||
b.WriteString(fmt.Sprintf("%s %s %s · %s\n",
|
||||
warpDimSt.Render(fmt.Sprintf("%-*s", nameW+6, "aggregate")),
|
||||
warpTxSt.Render("TX"), padGbps(txBps),
|
||||
warpTotalSt.Render(fmt.Sprintf("pool %s / %s", bench.FormatBytes(wire), bench.FormatBytes(capTotal)))))
|
||||
if len(v.history) > 1 {
|
||||
b.WriteString(warpDimSt.Render("history ") + warpTotalSt.Render(sparkline(v.history, 40)) + "\n")
|
||||
}
|
||||
b.WriteString("\n")
|
||||
|
||||
// Clients: loop progress per restic identity.
|
||||
cnameW := 6
|
||||
for _, d := range r.Droplets {
|
||||
if d.Status == nil {
|
||||
continue
|
||||
}
|
||||
for _, c := range d.Status.Clients {
|
||||
cnameW = max(cnameW, len(shortClient(c.Name)))
|
||||
}
|
||||
}
|
||||
b.WriteString(warpDimSt.Render(fmt.Sprintf("%-*s %-9s %6s %10s %6s %s", cnameW, "CLIENT", "PHASE", "CYCLES", "UPLOADED", "ERR", "LAST ERROR")) + "\n")
|
||||
for _, d := range r.Droplets {
|
||||
if d.Status == nil {
|
||||
if d.Err != "" {
|
||||
b.WriteString(fmt.Sprintf("%-*s %s\n", cnameW, shortClient(d.Name), warpErrSt.Render(d.Err)))
|
||||
}
|
||||
continue
|
||||
}
|
||||
for _, c := range d.Status.Clients {
|
||||
lastErr := ""
|
||||
if c.LastError != "" {
|
||||
lastErr = warpErrSt.Render(truncate(c.LastError, 48))
|
||||
}
|
||||
b.WriteString(fmt.Sprintf("%-*s %-9s %6d %10s %s %s\n",
|
||||
cnameW, shortClient(c.Name), phaseCell(c.Phase), c.Cycles,
|
||||
bench.FormatBytes(c.Uploaded+c.CurrentBytes), errCell6(c.Errors), lastErr))
|
||||
}
|
||||
}
|
||||
|
||||
footer := fmt.Sprintf("sampled %ds window at %s", sampleSec, sampledAt.Format("15:04:05"))
|
||||
if watching {
|
||||
footer += " · ctrl-c to quit"
|
||||
}
|
||||
b.WriteString("\n" + warpDimSt.Render(footer))
|
||||
|
||||
return warpFrame.Render(b.String())
|
||||
}
|
||||
|
||||
// shortClient drops the shared fleet prefix so tables stay narrow.
|
||||
func shortClient(name string) string {
|
||||
return strings.TrimPrefix(name, "yucca-bench-")
|
||||
}
|
||||
|
||||
func phaseCell(phase string) string {
|
||||
switch phase {
|
||||
case "backup":
|
||||
return warpOKSt.Render(fmt.Sprintf("%-9s", phase))
|
||||
case "generate":
|
||||
return warpTxSt.Render(fmt.Sprintf("%-9s", phase))
|
||||
default:
|
||||
return warpDimSt.Render(fmt.Sprintf("%-9s", phase))
|
||||
}
|
||||
}
|
||||
|
||||
func errCell6(n int) string {
|
||||
s := fmt.Sprintf("%6d", n)
|
||||
if n > 0 {
|
||||
return warpErrSt.Render(s)
|
||||
}
|
||||
return warpDimSt.Render(s)
|
||||
}
|
||||
|
||||
// stateCell colors the droplet's agent state, flagging a dead agent while a
|
||||
// run is recorded.
|
||||
func stateCell(d benchwide.DropletStatus, state string) string {
|
||||
switch {
|
||||
case !d.Reachable:
|
||||
return warpErrSt.Render("unreachable")
|
||||
case state == "running" && d.AgentProcs > 0:
|
||||
return warpOKSt.Render("running")
|
||||
case state == "capped":
|
||||
return warpWarnSt.Render("capped")
|
||||
case state == "done":
|
||||
return warpDimSt.Render("done")
|
||||
case d.AgentProcs == 0 && state == "running":
|
||||
return warpErrSt.Render("agent dead")
|
||||
default:
|
||||
return warpDimSt.Render(state)
|
||||
}
|
||||
}
|
||||
|
||||
func saveBenchDoResult(path string, r *benchwide.Result) error {
|
||||
b, err := json.MarshalIndent(r, "", " ")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
return os.WriteFile(path, append(b, '\n'), 0o644)
|
||||
}
|
||||
|
||||
func renderBenchDoResult(w io.Writer, r *benchwide.Result) {
|
||||
fmt.Fprintf(w, "\n%s partition=%s elapsed=%s\n", r.Label, r.Partition, bench.FormatDuration(r.ElapsedSeconds))
|
||||
tw := tabwriter.NewWriter(w, 2, 4, 2, ' ', 0)
|
||||
fmt.Fprintln(tw, "DROPLET\tREGION\tSTATE\tWIRE TX\tCLIENT\tCYCLES\tUPLOADED\tERRORS")
|
||||
for _, d := range r.Droplets {
|
||||
if len(d.Clients) == 0 {
|
||||
fmt.Fprintf(tw, "%s\t%s\t%s\t%s\t-\t-\t-\t-\n",
|
||||
shortClient(d.Name), d.Region, d.State, bench.FormatBytes(d.WireTxBytes))
|
||||
continue
|
||||
}
|
||||
for i, c := range d.Clients {
|
||||
name, region, state, wireTx := shortClient(d.Name), d.Region, d.State, bench.FormatBytes(d.WireTxBytes)
|
||||
if i > 0 {
|
||||
name, region, state, wireTx = "", "", "", ""
|
||||
}
|
||||
fmt.Fprintf(tw, "%s\t%s\t%s\t%s\t%s\t%d\t%s\t%d\n",
|
||||
name, region, state, wireTx, "c"+strings.TrimPrefix(c.Name, d.Name+"-c"), c.Cycles,
|
||||
bench.FormatBytes(c.Uploaded), c.Errors)
|
||||
}
|
||||
}
|
||||
tw.Flush()
|
||||
fmt.Fprintf(w, "totals: uploaded=%s wire-tx=%s cycles=%d errors=%d",
|
||||
bench.FormatBytes(r.TotalUploaded), bench.FormatBytes(r.TotalWireTx), r.TotalCycles, r.TotalErrors)
|
||||
if r.ElapsedSeconds > 0 && r.TotalUploaded > 0 {
|
||||
fmt.Fprintf(w, " avg=%s", bench.FormatBPS(float64(r.TotalUploaded)/r.ElapsedSeconds))
|
||||
}
|
||||
fmt.Fprintln(w)
|
||||
}
|
||||
@@ -1,506 +0,0 @@
|
||||
package cli
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"net/url"
|
||||
"os"
|
||||
"strconv"
|
||||
"strings"
|
||||
"text/tabwriter"
|
||||
|
||||
"github.com/spf13/cobra"
|
||||
|
||||
"yuctl/internal/adminapi"
|
||||
)
|
||||
|
||||
// newUsersCmd builds the `users` subtree.
|
||||
func newUsersCmd() *cobra.Command {
|
||||
cmd := &cobra.Command{
|
||||
Use: "users",
|
||||
Short: "User administration via yucca-admin-api",
|
||||
}
|
||||
cmd.AddCommand(newUsersListCmd())
|
||||
cmd.AddCommand(newUsersAllowlistCmd())
|
||||
cmd.AddCommand(newUsersViewDashboardCmd())
|
||||
cmd.AddCommand(newUsersFeaturesCmd())
|
||||
cmd.AddCommand(newUsersConnectionsCmd())
|
||||
return cmd
|
||||
}
|
||||
|
||||
// newUsersFeaturesCmd builds `users features`: per-user feature-flag overrides.
|
||||
func newUsersFeaturesCmd() *cobra.Command {
|
||||
cmd := &cobra.Command{
|
||||
Use: "features",
|
||||
Short: "Per-user feature-flag overrides",
|
||||
}
|
||||
cmd.AddCommand(newUsersFeaturesListCmd())
|
||||
cmd.AddCommand(newUsersFeaturesSetCmd())
|
||||
cmd.AddCommand(newUsersFeaturesClearCmd())
|
||||
return cmd
|
||||
}
|
||||
|
||||
func newUsersFeaturesListCmd() *cobra.Command {
|
||||
flags := &adminFlags{}
|
||||
c := &cobra.Command{
|
||||
Use: "list <email>",
|
||||
Short: "Show a user's resolved feature flags and overrides",
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
ctx := cmd.Context()
|
||||
client, _, err := flags.allowlistClient(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
userID, err := resolveUserID(ctx, client, args[0])
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
features, err := client.GetUserFeatures(ctx, userID)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
overridden := map[string]adminapi.FeatureOverride{}
|
||||
for _, o := range features.Overrides {
|
||||
overridden[o.Flag] = o
|
||||
}
|
||||
|
||||
w := tabwriter.NewWriter(cmd.OutOrStdout(), 0, 2, 2, ' ', 0)
|
||||
fmt.Fprintln(w, "FLAG\tVALUE\tSOURCE\tSET BY\tREASON")
|
||||
for flag, value := range features.Features {
|
||||
if o, ok := overridden[flag]; ok {
|
||||
reason := ""
|
||||
if o.Reason != nil {
|
||||
reason = *o.Reason
|
||||
}
|
||||
fmt.Fprintf(w, "%s\t%t\toverride\t%s\t%s\n", flag, value, o.SetBy, reason)
|
||||
} else {
|
||||
fmt.Fprintf(w, "%s\t%t\tdefault\t\t\n", flag, value)
|
||||
}
|
||||
}
|
||||
w.Flush()
|
||||
return nil
|
||||
},
|
||||
}
|
||||
flags.register(c)
|
||||
return c
|
||||
}
|
||||
|
||||
func newUsersFeaturesSetCmd() *cobra.Command {
|
||||
flags := &adminFlags{}
|
||||
var reason string
|
||||
c := &cobra.Command{
|
||||
Use: "set <email> <flag> on|off",
|
||||
Short: "Set a per-user feature-flag override",
|
||||
Args: cobra.ExactArgs(3),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
var value bool
|
||||
switch args[2] {
|
||||
case "on", "true":
|
||||
value = true
|
||||
case "off", "false":
|
||||
value = false
|
||||
default:
|
||||
return fmt.Errorf("value must be on|off, got %q", args[2])
|
||||
}
|
||||
|
||||
ctx := cmd.Context()
|
||||
client, _, err := flags.allowlistClient(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
userID, err := resolveUserID(ctx, client, args[0])
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if _, err := client.SetUserFeature(ctx, userID, args[1], value, reason); err != nil {
|
||||
return err
|
||||
}
|
||||
fmt.Fprintf(cmd.ErrOrStderr(), "%s set to %t for %s\n", args[1], value, args[0])
|
||||
return nil
|
||||
},
|
||||
}
|
||||
c.Flags().StringVar(&reason, "reason", "", "audit note stored on the override")
|
||||
flags.register(c)
|
||||
return c
|
||||
}
|
||||
|
||||
func newUsersFeaturesClearCmd() *cobra.Command {
|
||||
flags := &adminFlags{}
|
||||
c := &cobra.Command{
|
||||
Use: "clear <email> <flag>",
|
||||
Short: "Clear an override (revert to the registry default)",
|
||||
Args: cobra.ExactArgs(2),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
ctx := cmd.Context()
|
||||
client, _, err := flags.allowlistClient(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
userID, err := resolveUserID(ctx, client, args[0])
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := client.ClearUserFeature(ctx, userID, args[1]); err != nil {
|
||||
return err
|
||||
}
|
||||
fmt.Fprintf(cmd.ErrOrStderr(), "%s override cleared for %s\n", args[1], args[0])
|
||||
return nil
|
||||
},
|
||||
}
|
||||
flags.register(c)
|
||||
return c
|
||||
}
|
||||
|
||||
// newUsersConnectionsCmd builds `users connections`.
|
||||
func newUsersConnectionsCmd() *cobra.Command {
|
||||
cmd := &cobra.Command{
|
||||
Use: "connections",
|
||||
Short: "A user's connection instances (immich/restic)",
|
||||
}
|
||||
|
||||
flags := &adminFlags{}
|
||||
list := &cobra.Command{
|
||||
Use: "list <email>",
|
||||
Short: "List a user's connections",
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
ctx := cmd.Context()
|
||||
client, _, err := flags.allowlistClient(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
userID, err := resolveUserID(ctx, client, args[0])
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
connections, err := client.GetUserConnections(ctx, userID)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
w := tabwriter.NewWriter(cmd.OutOrStdout(), 0, 2, 2, ' ', 0)
|
||||
fmt.Fprintln(w, "ID\tTYPE\tNAME\tCREATED\tLAST SEEN")
|
||||
for _, connection := range connections {
|
||||
lastSeen := ""
|
||||
if connection.LastSeenAt != nil {
|
||||
lastSeen = *connection.LastSeenAt
|
||||
}
|
||||
fmt.Fprintf(w, "%s\t%s\t%s\t%s\t%s\n", connection.ID, connection.Type, connection.Name, connection.CreatedAt, lastSeen)
|
||||
}
|
||||
w.Flush()
|
||||
return nil
|
||||
},
|
||||
}
|
||||
flags.register(list)
|
||||
cmd.AddCommand(list)
|
||||
return cmd
|
||||
}
|
||||
|
||||
const defaultGrafanaURL = "https://grafana.futostatus.com"
|
||||
|
||||
func newUsersViewDashboardCmd() *cobra.Command {
|
||||
flags := &adminFlags{}
|
||||
var userID, email, grafanaURL string
|
||||
var noOpen bool
|
||||
c := &cobra.Command{
|
||||
Use: "view-dashboard",
|
||||
Short: "Open the per-user Grafana dashboard for a user",
|
||||
Long: "Build the Grafana per-user drill-down URL (dashboard uid yucca-per-user) for\n" +
|
||||
"a user and open it in the browser. --id builds the URL without contacting the\n" +
|
||||
"admin-api; --email resolves the user via the partition's admin-api first.",
|
||||
Args: cobra.NoArgs,
|
||||
RunE: func(cmd *cobra.Command, _ []string) error {
|
||||
id := userID
|
||||
if email != "" {
|
||||
client, partition, err := flags.allowlistClient(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
users, err := client.ListUsers(cmd.Context(), 0)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
id = ""
|
||||
for _, u := range users {
|
||||
if strings.EqualFold(u.Email, email) {
|
||||
id = u.ID
|
||||
break
|
||||
}
|
||||
}
|
||||
if id == "" {
|
||||
return fmt.Errorf("no user with email %q in partition %s", email, partition)
|
||||
}
|
||||
}
|
||||
|
||||
base := grafanaURL
|
||||
if base == "" {
|
||||
base = os.Getenv("YUCTL_GRAFANA_URL")
|
||||
}
|
||||
if base == "" {
|
||||
base = defaultGrafanaURL
|
||||
}
|
||||
dashboardURL := strings.TrimRight(base, "/") + "/d/yucca-per-user?var-user=" + url.QueryEscape(id)
|
||||
|
||||
fmt.Fprintln(cmd.OutOrStdout(), dashboardURL)
|
||||
if noOpen {
|
||||
return nil
|
||||
}
|
||||
if err := openBrowser(dashboardURL); err != nil {
|
||||
return fmt.Errorf("open browser: %w", err)
|
||||
}
|
||||
return nil
|
||||
},
|
||||
}
|
||||
c.Flags().StringVar(&userID, "id", "", "user id (uuid)")
|
||||
c.Flags().StringVar(&email, "email", "", "user email; resolved to an id via the admin-api")
|
||||
c.Flags().StringVar(&grafanaURL, "grafana-url", "", "Grafana base URL (default: $YUCTL_GRAFANA_URL or "+defaultGrafanaURL+")")
|
||||
c.Flags().BoolVar(&noOpen, "no-open", false, "print the dashboard URL instead of opening the browser")
|
||||
c.MarkFlagsOneRequired("id", "email")
|
||||
c.MarkFlagsMutuallyExclusive("id", "email")
|
||||
flags.register(c)
|
||||
return c
|
||||
}
|
||||
|
||||
func newUsersListCmd() *cobra.Command {
|
||||
flags := &adminFlags{}
|
||||
var limitFlag string
|
||||
c := &cobra.Command{
|
||||
Use: "list",
|
||||
Short: "List users in the partition's primary region",
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
ctx := cmd.Context()
|
||||
cc, err := requireContext()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
limit, err := adminapi.ParseLimit(limitFlag)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
topo, err := resolveTopology(ctx)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
client, _, err := flags.adminLogin(ctx, cmd, cc, topo)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
users, err := client.ListUsers(ctx, limit)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
w := tabwriter.NewWriter(cmd.OutOrStdout(), 0, 2, 2, ' ', 0)
|
||||
fmt.Fprintln(w, "ID\tSUB\tNAME\tEMAIL\tDISABLED")
|
||||
for _, u := range users {
|
||||
fmt.Fprintf(w, "%s\t%s\t%s\t%s\t%t\n", u.ID, u.Sub, u.Name, u.Email, u.Disabled)
|
||||
}
|
||||
w.Flush()
|
||||
fmt.Fprintf(cmd.ErrOrStderr(), "\n%d user(s) in partition %s\n", len(users), cc.Partition)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
c.Flags().StringVar(&limitFlag, "limit", "", "page size for the admin-api (default: server default)")
|
||||
flags.register(c)
|
||||
return c
|
||||
}
|
||||
|
||||
// newUsersAllowlistCmd builds the `users allowlist` subtree (beta email
|
||||
// allowlist + invites).
|
||||
func newUsersAllowlistCmd() *cobra.Command {
|
||||
cmd := &cobra.Command{
|
||||
Use: "allowlist",
|
||||
Short: "Manage the beta email allowlist and invites",
|
||||
}
|
||||
cmd.AddCommand(newAllowlistListCmd())
|
||||
cmd.AddCommand(newAllowlistAddCmd())
|
||||
cmd.AddCommand(newAllowlistRemoveCmd())
|
||||
cmd.AddCommand(newAllowlistInviteCmd())
|
||||
cmd.AddCommand(newAllowlistInviteBatchCmd())
|
||||
return cmd
|
||||
}
|
||||
|
||||
func (f *adminFlags) allowlistClient(cmd *cobra.Command) (*adminapi.Client, string, error) {
|
||||
ctx := cmd.Context()
|
||||
cc, err := requireContext()
|
||||
if err != nil {
|
||||
return nil, "", err
|
||||
}
|
||||
topo, err := resolveTopology(ctx)
|
||||
if err != nil {
|
||||
return nil, "", err
|
||||
}
|
||||
client, _, err := f.adminLogin(ctx, cmd, cc, topo)
|
||||
if err != nil {
|
||||
return nil, "", err
|
||||
}
|
||||
return client, cc.Partition, nil
|
||||
}
|
||||
|
||||
func printAllowlistEntries(cmd *cobra.Command, entries []adminapi.AllowlistEntry) {
|
||||
w := tabwriter.NewWriter(cmd.OutOrStdout(), 0, 2, 2, ' ', 0)
|
||||
fmt.Fprintln(w, "EMAIL\tCODE\tINVITED\tUSED\tUSED AT\tCREATED")
|
||||
for _, e := range entries {
|
||||
usedAt := ""
|
||||
if e.InviteUsedAt != nil {
|
||||
usedAt = *e.InviteUsedAt
|
||||
}
|
||||
fmt.Fprintf(w, "%s\t%s\t%t\t%t\t%s\t%s\n", e.Email, e.InviteCode, e.Invited, e.InviteUsed, usedAt, e.CreatedAt)
|
||||
}
|
||||
w.Flush()
|
||||
}
|
||||
|
||||
func newAllowlistListCmd() *cobra.Command {
|
||||
flags := &adminFlags{}
|
||||
var limitFlag string
|
||||
c := &cobra.Command{
|
||||
Use: "list",
|
||||
Short: "List allowlist entries",
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
limit, err := adminapi.ParseLimit(limitFlag)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
client, partition, err := flags.allowlistClient(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
entries, err := client.ListAllowlist(cmd.Context(), limit)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
printAllowlistEntries(cmd, entries)
|
||||
fmt.Fprintf(cmd.ErrOrStderr(), "\n%d entries in partition %s\n", len(entries), partition)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
c.Flags().StringVar(&limitFlag, "limit", "", "page size for the admin-api (default: server default)")
|
||||
flags.register(c)
|
||||
return c
|
||||
}
|
||||
|
||||
func newAllowlistAddCmd() *cobra.Command {
|
||||
flags := &adminFlags{}
|
||||
var staged bool
|
||||
c := &cobra.Command{
|
||||
Use: "add <email>",
|
||||
Short: "Allow an email to sign up (--staged to waitlist it instead)",
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
client, _, err := flags.allowlistClient(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
entry, err := client.AddAllowlistEntry(cmd.Context(), args[0], staged)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
printAllowlistEntries(cmd, []adminapi.AllowlistEntry{*entry})
|
||||
return nil
|
||||
},
|
||||
}
|
||||
c.Flags().BoolVar(&staged, "staged", false, "stage the email without allowing login yet")
|
||||
flags.register(c)
|
||||
return c
|
||||
}
|
||||
|
||||
func newAllowlistRemoveCmd() *cobra.Command {
|
||||
flags := &adminFlags{}
|
||||
c := &cobra.Command{
|
||||
Use: "remove <email>",
|
||||
Short: "Remove an email from the allowlist",
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
client, _, err := flags.allowlistClient(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := client.RemoveAllowlistEntry(cmd.Context(), args[0]); err != nil {
|
||||
return err
|
||||
}
|
||||
fmt.Fprintf(cmd.ErrOrStderr(), "removed %s\n", args[0])
|
||||
return nil
|
||||
},
|
||||
}
|
||||
flags.register(c)
|
||||
return c
|
||||
}
|
||||
|
||||
func newAllowlistInviteCmd() *cobra.Command {
|
||||
flags := &adminFlags{}
|
||||
c := &cobra.Command{
|
||||
Use: "invite <email>[,<email>...]",
|
||||
Short: "Invite emails: allow them to sign up, creating entries as needed",
|
||||
Args: cobra.MinimumNArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
var emails []string
|
||||
for _, arg := range args {
|
||||
for _, email := range strings.Split(arg, ",") {
|
||||
if email = strings.TrimSpace(email); email != "" {
|
||||
emails = append(emails, email)
|
||||
}
|
||||
}
|
||||
}
|
||||
client, _, err := flags.allowlistClient(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
entries, err := client.InviteEmails(cmd.Context(), emails)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
printAllowlistEntries(cmd, entries)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
flags.register(c)
|
||||
return c
|
||||
}
|
||||
|
||||
func newAllowlistInviteBatchCmd() *cobra.Command {
|
||||
flags := &adminFlags{}
|
||||
c := &cobra.Command{
|
||||
Use: "invite-batch <count>",
|
||||
Short: "Invite the oldest <count> staged (waitlisted) emails",
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
count, err := strconv.Atoi(args[0])
|
||||
if err != nil || count < 1 {
|
||||
return fmt.Errorf("count must be a positive integer")
|
||||
}
|
||||
client, _, err := flags.allowlistClient(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
entries, err := client.InviteBatch(cmd.Context(), count)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
printAllowlistEntries(cmd, entries)
|
||||
fmt.Fprintf(cmd.ErrOrStderr(), "\ninvited %d entries\n", len(entries))
|
||||
return nil
|
||||
},
|
||||
}
|
||||
flags.register(c)
|
||||
return c
|
||||
}
|
||||
|
||||
// resolveUserID turns an --user email into a user id via the admin-api.
|
||||
func resolveUserID(ctx context.Context, client *adminapi.Client, email string) (string, error) {
|
||||
users, err := client.ListUsers(ctx, 0)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
for _, u := range users {
|
||||
if strings.EqualFold(u.Email, email) {
|
||||
return u.ID, nil
|
||||
}
|
||||
}
|
||||
return "", fmt.Errorf("no user with email %q", email)
|
||||
}
|
||||
@@ -1,194 +0,0 @@
|
||||
package cli
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/charmbracelet/lipgloss"
|
||||
|
||||
"yuctl/internal/warp"
|
||||
)
|
||||
|
||||
// warpView renders StatusReports as a styled dashboard; watch mode feeds it
|
||||
// successive samples and it keeps a throughput history for the sparkline.
|
||||
type warpView struct {
|
||||
label string
|
||||
history []float64 // combined bps per sample
|
||||
}
|
||||
|
||||
var (
|
||||
warpAccent = lipgloss.AdaptiveColor{Light: "#6C50FF", Dark: "#9D7CFF"}
|
||||
warpOK = lipgloss.AdaptiveColor{Light: "#12A150", Dark: "#2ECC71"}
|
||||
warpErr = lipgloss.AdaptiveColor{Light: "#D0021B", Dark: "#FF5F56"}
|
||||
warpWarn = lipgloss.AdaptiveColor{Light: "#B8860B", Dark: "#F5C542"}
|
||||
warpDim = lipgloss.AdaptiveColor{Light: "244", Dark: "241"}
|
||||
warpTx = lipgloss.AdaptiveColor{Light: "#0087AF", Dark: "#33D1E0"}
|
||||
warpRx = lipgloss.AdaptiveColor{Light: "#AF5F00", Dark: "#F5A623"}
|
||||
|
||||
warpBadge = lipgloss.NewStyle().Bold(true).Foreground(lipgloss.Color("#FFFFFF")).Background(warpAccent).Padding(0, 1)
|
||||
warpTitle = lipgloss.NewStyle().Bold(true)
|
||||
warpDimSt = lipgloss.NewStyle().Foreground(warpDim)
|
||||
warpOKSt = lipgloss.NewStyle().Foreground(warpOK)
|
||||
warpErrSt = lipgloss.NewStyle().Bold(true).Foreground(warpErr)
|
||||
warpWarnSt = lipgloss.NewStyle().Foreground(warpWarn)
|
||||
warpTxSt = lipgloss.NewStyle().Foreground(warpTx)
|
||||
warpRxSt = lipgloss.NewStyle().Foreground(warpRx)
|
||||
warpTotalSt = lipgloss.NewStyle().Bold(true).Foreground(warpAccent)
|
||||
warpFrame = lipgloss.NewStyle().Border(lipgloss.RoundedBorder()).BorderForeground(warpAccent).Padding(0, 1)
|
||||
)
|
||||
|
||||
func (v *warpView) push(combined float64) {
|
||||
v.history = append(v.history, combined)
|
||||
if len(v.history) > 60 {
|
||||
v.history = v.history[len(v.history)-60:]
|
||||
}
|
||||
}
|
||||
|
||||
func (v *warpView) render(r *warp.StatusReport, sampledAt time.Time, sampleSec int, watching bool) string {
|
||||
var b strings.Builder
|
||||
|
||||
b.WriteString(warpBadge.Render("WARP") + " " + warpTitle.Render(v.label) + "\n")
|
||||
|
||||
if len(r.Pods) == 0 {
|
||||
b.WriteString(warpWarnSt.Render("no runner pods deployed") +
|
||||
warpDimSt.Render(" — yuctl tools warp deploy") + "\n")
|
||||
return warpFrame.Render(strings.TrimRight(b.String(), "\n"))
|
||||
}
|
||||
|
||||
if r.Config != nil {
|
||||
since := r.Config["started_at"]
|
||||
if t, err := time.Parse(time.RFC3339, since); err == nil {
|
||||
since = time.Since(t).Round(time.Second).String() + " ago"
|
||||
}
|
||||
b.WriteString(fmt.Sprintf("%s %s · %s PUT / %s GET · %s/%s objects · %s RGWs\n",
|
||||
warpOKSt.Render("● "+r.Config["mode"]),
|
||||
warpDimSt.Render("started "+since),
|
||||
r.Config["put_streams"], r.Config["get_streams"],
|
||||
r.Config["put_obj_size"], r.Config["get_obj_size"],
|
||||
r.Config["rgw_endpoints"]))
|
||||
b.WriteString(warpDimSt.Render("endpoint "+r.Config["endpoint"]) + "\n")
|
||||
} else {
|
||||
b.WriteString(warpWarnSt.Render("○ no active run recorded") +
|
||||
warpDimSt.Render(" — load stopped or never started") + "\n")
|
||||
}
|
||||
b.WriteString("\n")
|
||||
|
||||
// Nodes: TX/RX bars scaled to the busiest direction in this frame.
|
||||
var maxBps float64
|
||||
for _, n := range r.Nodes {
|
||||
maxBps = max(maxBps, max(n.TxBps, n.RxBps))
|
||||
}
|
||||
nodeW, ifaceW := 4, 5
|
||||
for _, n := range r.Nodes {
|
||||
nodeW = max(nodeW, len(n.Node))
|
||||
ifaceW = max(ifaceW, len(n.Iface))
|
||||
}
|
||||
var tx, rx float64
|
||||
for _, n := range r.Nodes {
|
||||
b.WriteString(fmt.Sprintf("%-*s %s %s %s %s %s\n",
|
||||
nodeW, n.Node,
|
||||
warpDimSt.Render(fmt.Sprintf("%-*s", ifaceW, n.Iface)),
|
||||
warpTxSt.Render(meter(n.TxBps, maxBps, 14)),
|
||||
padGbps(n.TxBps),
|
||||
warpRxSt.Render(meter(n.RxBps, maxBps, 14)),
|
||||
padGbps(n.RxBps)))
|
||||
tx += n.TxBps
|
||||
rx += n.RxBps
|
||||
}
|
||||
if len(r.Nodes) > 0 {
|
||||
b.WriteString(fmt.Sprintf("%s %s %s · %s %s · %s\n",
|
||||
warpDimSt.Render(fmt.Sprintf("%-*s", nodeW+ifaceW-2, "aggregate")),
|
||||
warpTxSt.Render("TX"), padGbps(tx),
|
||||
warpRxSt.Render("RX"), padGbps(rx),
|
||||
warpTotalSt.Render("combined "+fmtGbps(tx+rx))))
|
||||
if len(v.history) > 1 {
|
||||
b.WriteString(warpDimSt.Render("history ") + warpTotalSt.Render(sparkline(v.history, 40)) + "\n")
|
||||
}
|
||||
b.WriteString("\n")
|
||||
}
|
||||
|
||||
// Pods: process liveness and log error counts.
|
||||
podW := 3
|
||||
for _, p := range r.Pods {
|
||||
podW = max(podW, len(p.Name))
|
||||
}
|
||||
b.WriteString(warpDimSt.Render(fmt.Sprintf("%-*s %-*s %6s %6s %10s %10s", podW, "POD", nodeW, "NODE", "PUT", "GET", "ERR(PUT)", "ERR(GET)")) + "\n")
|
||||
for _, p := range r.Pods {
|
||||
b.WriteString(fmt.Sprintf("%-*s %-*s %s %s %s %s\n",
|
||||
podW, p.Name, nodeW, p.Node,
|
||||
procCell(p.PutProcs, r.Config != nil),
|
||||
procCell(p.GetProcs, r.Config != nil),
|
||||
errCell(p.PutErrors), errCell(p.GetErrors)))
|
||||
}
|
||||
|
||||
footer := fmt.Sprintf("sampled %ds window at %s", sampleSec, sampledAt.Format("15:04:05"))
|
||||
if watching {
|
||||
footer += " · ctrl-c to quit"
|
||||
}
|
||||
b.WriteString("\n" + warpDimSt.Render(footer))
|
||||
|
||||
return warpFrame.Render(b.String())
|
||||
}
|
||||
|
||||
// procCell colors a warp process count: green when alive, red when a run is
|
||||
// recorded but nothing is running, dim otherwise.
|
||||
func procCell(n int, runRecorded bool) string {
|
||||
s := fmt.Sprintf("%6d", n)
|
||||
switch {
|
||||
case n > 0:
|
||||
return warpOKSt.Render(s)
|
||||
case runRecorded:
|
||||
return warpErrSt.Render(s)
|
||||
default:
|
||||
return warpDimSt.Render(s)
|
||||
}
|
||||
}
|
||||
|
||||
func errCell(n int) string {
|
||||
s := fmt.Sprintf("%10d", n)
|
||||
if n > 0 {
|
||||
return warpErrSt.Render(s)
|
||||
}
|
||||
return warpDimSt.Render(s)
|
||||
}
|
||||
|
||||
func fmtGbps(bps float64) string {
|
||||
return fmt.Sprintf("%.2f Gbps", bps/1e9)
|
||||
}
|
||||
|
||||
func padGbps(bps float64) string {
|
||||
return fmt.Sprintf("%11s", fmtGbps(bps))
|
||||
}
|
||||
|
||||
// meter renders a horizontal bar of width w, filled to value/scale.
|
||||
func meter(value, scale float64, w int) string {
|
||||
filled := 0
|
||||
if scale > 0 {
|
||||
filled = int(value / scale * float64(w))
|
||||
}
|
||||
filled = min(max(filled, 0), w)
|
||||
return strings.Repeat("█", filled) + strings.Repeat("░", w-filled)
|
||||
}
|
||||
|
||||
// sparkline renders the last len(vals) samples with block glyphs, scaled to
|
||||
// the window maximum, downsampled to at most w points.
|
||||
func sparkline(vals []float64, w int) string {
|
||||
if len(vals) > w {
|
||||
vals = vals[len(vals)-w:]
|
||||
}
|
||||
var maxV float64
|
||||
for _, v := range vals {
|
||||
maxV = max(maxV, v)
|
||||
}
|
||||
if maxV == 0 {
|
||||
return strings.Repeat("▁", len(vals))
|
||||
}
|
||||
glyphs := []rune("▁▂▃▄▅▆▇█")
|
||||
var b strings.Builder
|
||||
for _, v := range vals {
|
||||
i := int(v / maxV * float64(len(glyphs)-1))
|
||||
b.WriteRune(glyphs[min(max(i, 0), len(glyphs)-1)])
|
||||
}
|
||||
return b.String()
|
||||
}
|
||||
@@ -1,6 +1,6 @@
|
||||
// Command yuctl is the yucca operations CLI. It resolves the
|
||||
// partition/region/ceph topology from Terraform "discovery" outputs (read
|
||||
// directly from S3 state) and drives day-2 operations. See internal/cli for the
|
||||
// directly from S3 state) and drives day-2 operations. See yuctl/cli for the
|
||||
// command tree and packages/yuctl/README.md for usage.
|
||||
package main
|
||||
|
||||
@@ -12,7 +12,7 @@ import (
|
||||
|
||||
"github.com/rs/zerolog/log"
|
||||
|
||||
"yuctl/internal/cli"
|
||||
"yuctl/cli"
|
||||
)
|
||||
|
||||
func main() {
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
// Package netdev parses /proc/net/dev counter snapshots. Shared by everything
|
||||
// that measures NIC throughput from that file — the warp status sampler and
|
||||
// the bench-do agent/orchestrator — so virtual-interface filtering stays in
|
||||
// the fleet-bench agent/orchestrator — so virtual-interface filtering stays in
|
||||
// one place.
|
||||
package netdev
|
||||
|
||||
@@ -5,10 +5,10 @@ import (
|
||||
"strconv"
|
||||
"time"
|
||||
|
||||
"yuctl/internal/do"
|
||||
"yuctl/do"
|
||||
)
|
||||
|
||||
// doProvider adapts the DigitalOcean client (internal/do) to Provider. It owns
|
||||
// doProvider adapts the DigitalOcean client (yuctl/do) to Provider. It owns
|
||||
// no logic of its own — just the Host/Size translation and the int⇄string id
|
||||
// bridging Provider's opaque ids require.
|
||||
type doProvider struct{ c *do.Client }
|
||||
@@ -14,19 +14,19 @@ import (
|
||||
|
||||
"github.com/rs/zerolog/log"
|
||||
|
||||
"yuctl/internal/op"
|
||||
"yuctl/op"
|
||||
)
|
||||
|
||||
// hetznerTokenRef is the 1Password item holding the Hetzner Cloud API token;
|
||||
// $HCLOUD_TOKEN skips op entirely, $YUCTL_HCLOUD_TOKEN_REF overrides the ref.
|
||||
const hetznerTokenRef = "op://yucca_tf_prod/HCLOUD_API_TOKEN/password"
|
||||
|
||||
// fleetLabel is the Hetzner label key carrying the bench-wide fleet tag; the
|
||||
// fleetLabel is the Hetzner label key carrying the fleet-bench fleet tag; the
|
||||
// tag is the value, and list/destroy select on it.
|
||||
const fleetLabel = "yuctl-fleet"
|
||||
|
||||
// hetznerProvider talks to the Hetzner Cloud API (https://api.hetzner.cloud/v1)
|
||||
// over plain net/http — the surface bench-wide needs is small enough that a
|
||||
// over plain net/http — the surface fleet-bench needs is small enough that a
|
||||
// full SDK dependency isn't worth it.
|
||||
type hetznerProvider struct {
|
||||
token string
|
||||
@@ -1,4 +1,4 @@
|
||||
// Package provider abstracts the cloud that hosts a bench-wide fleet so the
|
||||
// Package provider abstracts the cloud that hosts a fleet-bench fleet so the
|
||||
// fleet lifecycle (deploy/start/status/stop/cleanup/undeploy) is identical
|
||||
// across providers. Each provider is pure VM plumbing — create/list/destroy
|
||||
// tagged hosts, upload an ephemeral ssh key, resolve a size's price and
|
||||
@@ -1,4 +1,4 @@
|
||||
package bench
|
||||
package resticbench
|
||||
|
||||
import (
|
||||
"context"
|
||||
+1
-1
@@ -1,4 +1,4 @@
|
||||
package bench
|
||||
package resticbench
|
||||
|
||||
import (
|
||||
"crypto/sha256"
|
||||
@@ -1,4 +1,4 @@
|
||||
package bench
|
||||
package resticbench
|
||||
|
||||
// Phase names, in execution order within a cell.
|
||||
const (
|
||||
@@ -1,4 +1,4 @@
|
||||
package bench
|
||||
package resticbench
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
//go:build !embedagent
|
||||
|
||||
package bench
|
||||
package resticbench
|
||||
|
||||
import "errors"
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
//go:build embedagent
|
||||
|
||||
package bench
|
||||
package resticbench
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
@@ -1,4 +1,10 @@
|
||||
package bench
|
||||
package resticbench
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"encoding/json"
|
||||
"io"
|
||||
)
|
||||
|
||||
// Event is one line of the agent→orchestrator stream (JSON per line on the
|
||||
// agent's stdout).
|
||||
@@ -13,3 +19,17 @@ type Event struct {
|
||||
PhaseResult *PhaseResult `json:"phaseResult,omitempty"`
|
||||
Result *RunResult `json:"result,omitempty"`
|
||||
}
|
||||
|
||||
// Non-JSON lines are skipped: agent stderr never lands here, but a remote
|
||||
// shell may still chirp.
|
||||
func ScanEvents(r io.Reader, handle func(Event)) error {
|
||||
sc := bufio.NewScanner(r)
|
||||
sc.Buffer(make([]byte, 1<<20), 8<<20)
|
||||
for sc.Scan() {
|
||||
var ev Event
|
||||
if json.Unmarshal(sc.Bytes(), &ev) == nil {
|
||||
handle(ev)
|
||||
}
|
||||
}
|
||||
return sc.Err()
|
||||
}
|
||||
@@ -1,4 +1,4 @@
|
||||
package bench
|
||||
package resticbench
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
@@ -1,4 +1,4 @@
|
||||
package bench
|
||||
package resticbench
|
||||
|
||||
import (
|
||||
"context"
|
||||
@@ -11,10 +11,10 @@ import (
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"yuctl/internal/netdev"
|
||||
"yuctl/netdev"
|
||||
)
|
||||
|
||||
// Loadgen is the bench-do agent mode: a supervisor that runs one continuous
|
||||
// Loadgen is the fleet-bench agent mode: a supervisor that runs one continuous
|
||||
// restic backup loop per client until the duration elapses, the droplet's
|
||||
// transfer cap is hit, or it is killed. Unlike the bench mode it is built to
|
||||
// run detached (nohup) on a droplet: progress goes to a status file the
|
||||
@@ -100,7 +100,7 @@ func (c *LoadgenConfig) defaults() error {
|
||||
}
|
||||
|
||||
// LoadgenStatus is the droplet-local progress file, rewritten atomically every
|
||||
// few seconds and read (plus a live NIC sample) by `bench-do status`.
|
||||
// few seconds and read (plus a live NIC sample) by `fleet-bench status`.
|
||||
type LoadgenStatus struct {
|
||||
StartedAt time.Time `json:"startedAt"`
|
||||
UpdatedAt time.Time `json:"updatedAt"`
|
||||
@@ -344,7 +344,7 @@ func clientLoop(ctx context.Context, cfg LoadgenConfig, idx int, cl LoadgenClien
|
||||
}
|
||||
}
|
||||
|
||||
// RunLoadgenCleanup forgets and prunes every bench-do snapshot in each
|
||||
// RunLoadgenCleanup forgets and prunes every fleet-bench snapshot in each
|
||||
// client's repository, streaming bench Events (it runs synchronously over the
|
||||
// orchestrator's ssh session, unlike the detached load mode).
|
||||
func RunLoadgenCleanup(ctx context.Context, cfg LoadgenConfig, emit func(Event)) error {
|
||||
+1
-1
@@ -1,4 +1,4 @@
|
||||
package bench
|
||||
package resticbench
|
||||
|
||||
import "testing"
|
||||
|
||||
+28
-67
@@ -1,20 +1,20 @@
|
||||
package bench
|
||||
package resticbench
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"bytes"
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
"os/exec"
|
||||
|
||||
"github.com/rs/zerolog/log"
|
||||
|
||||
"yuctl/sshx"
|
||||
)
|
||||
|
||||
// RemoteBinDir is where the agent and restic land on a remote host, relative
|
||||
// to $HOME. Shared with bench-do, which pushes the same binaries to droplets.
|
||||
// to $HOME. Shared with fleet-bench, which pushes the same binaries to its
|
||||
// hosts.
|
||||
const RemoteBinDir = ".cache/yuctl-bench/bin"
|
||||
|
||||
const remoteDir = RemoteBinDir
|
||||
@@ -25,7 +25,19 @@ type RunOpts struct {
|
||||
SSHIdentity string // ssh private key file ("" = ssh defaults/agent)
|
||||
AgentBin string // local linux/amd64 bench-agent; "" = use the embedded one
|
||||
Config Config
|
||||
Out string // local results path ("" = don't save)
|
||||
Out string // local results path ("" = don't save)
|
||||
Summary io.Writer // summary destination (nil = stdout)
|
||||
}
|
||||
|
||||
func (o *RunOpts) ssh() *sshx.Client {
|
||||
return &sshx.Client{IdentityFile: o.SSHIdentity, Retries: 2}
|
||||
}
|
||||
|
||||
func (o *RunOpts) summary() io.Writer {
|
||||
if o.Summary != nil {
|
||||
return o.Summary
|
||||
}
|
||||
return os.Stdout
|
||||
}
|
||||
|
||||
// Run pushes the agent + pinned restic to the host, streams the run, and
|
||||
@@ -41,18 +53,19 @@ func Run(ctx context.Context, opts RunOpts) (*RunResult, error) {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
ssh := opts.ssh()
|
||||
log.Info().Str("host", opts.Host).Msg("pushing agent + restic " + ResticVersion)
|
||||
if err := push(ctx, opts.SSHIdentity, opts.Host, agentBin, resticBin); err != nil {
|
||||
if err := ssh.Push(ctx, opts.Host, remoteDir, map[string]string{agentBin: "bench-agent", resticBin: "restic"}); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
log.Info().Str("host", opts.Host).Ints("connections", opts.Config.Connections).
|
||||
Str("size", FormatBytes(opts.Config.Size)).Msg("starting remote benchmark")
|
||||
result, err := drive(ctx, opts.SSHIdentity, opts.Host, opts.Config)
|
||||
result, err := drive(ctx, ssh, opts.Host, opts.Config)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return finish(result, opts.Out)
|
||||
return finish(result, opts.Out, opts.summary())
|
||||
}
|
||||
|
||||
// RunHere executes the benchmark on the local machine — no ssh, the agent
|
||||
@@ -74,50 +87,20 @@ func RunHere(ctx context.Context, opts RunOpts) (*RunResult, error) {
|
||||
if sink.result == nil {
|
||||
return nil, fmt.Errorf("benchmark finished without a result")
|
||||
}
|
||||
return finish(sink.result, opts.Out)
|
||||
return finish(sink.result, opts.Out, opts.summary())
|
||||
}
|
||||
|
||||
func finish(result *RunResult, out string) (*RunResult, error) {
|
||||
func finish(result *RunResult, out string, summary io.Writer) (*RunResult, error) {
|
||||
if out != "" {
|
||||
if err := SaveResult(out, result); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
log.Info().Str("path", out).Msg("results saved")
|
||||
}
|
||||
RenderSummary(os.Stdout, result)
|
||||
RenderSummary(summary, result)
|
||||
return result, nil
|
||||
}
|
||||
|
||||
// sshOpts builds the common ssh/scp options. accept-new (TOFU) keeps first
|
||||
// contact with a discovery-resolved IP from failing BatchMode.
|
||||
func sshOpts(identity string) []string {
|
||||
args := []string{
|
||||
"-o", "BatchMode=yes",
|
||||
"-o", "StrictHostKeyChecking=accept-new",
|
||||
"-o", "ServerAliveInterval=30",
|
||||
"-o", "ServerAliveCountMax=8",
|
||||
}
|
||||
if identity != "" {
|
||||
args = append(args, "-i", identity, "-o", "IdentitiesOnly=yes")
|
||||
}
|
||||
return args
|
||||
}
|
||||
|
||||
func sshBase(identity, host string) []string {
|
||||
return append(sshOpts(identity), host)
|
||||
}
|
||||
|
||||
func run(ctx context.Context, name string, args ...string) error {
|
||||
cmd := exec.CommandContext(ctx, name, args...)
|
||||
var out bytes.Buffer
|
||||
cmd.Stdout = &out
|
||||
cmd.Stderr = &out
|
||||
if err := cmd.Run(); err != nil {
|
||||
return fmt.Errorf("%s %v: %w: %s", name, args, err, tail(out.String(), 1000))
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// AgentBinary resolves the local linux/amd64 agent to push: an explicit path,
|
||||
// or the embedded one materialized into a temp file. The returned cleanup
|
||||
// removes the materialized copy.
|
||||
@@ -144,24 +127,10 @@ func AgentBinary(explicit string) (string, func(), error) {
|
||||
return path, func() { os.RemoveAll(dir) }, nil
|
||||
}
|
||||
|
||||
func push(ctx context.Context, identity, host, agentBin, resticBin string) error {
|
||||
if err := run(ctx, "ssh", append(sshBase(identity, host), "mkdir -p "+remoteDir)...); err != nil {
|
||||
return err
|
||||
}
|
||||
scp := append([]string{"-q"}, sshOpts(identity)...)
|
||||
if err := run(ctx, "scp", append(scp, agentBin, host+":"+remoteDir+"/bench-agent")...); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := run(ctx, "scp", append(scp, resticBin, host+":"+remoteDir+"/restic")...); err != nil {
|
||||
return err
|
||||
}
|
||||
return run(ctx, "ssh", append(sshBase(identity, host), "chmod +x "+remoteDir+"/bench-agent "+remoteDir+"/restic")...)
|
||||
}
|
||||
|
||||
// drive runs the remote agent, feeding Config over stdin and consuming the
|
||||
// event stream from stdout. The agent's stderr passes straight through.
|
||||
func drive(ctx context.Context, identity, host string, cfg Config) (*RunResult, error) {
|
||||
cmd := exec.CommandContext(ctx, "ssh", append(sshBase(identity, host), remoteDir+"/bench-agent")...)
|
||||
func drive(ctx context.Context, ssh *sshx.Client, host string, cfg Config) (*RunResult, error) {
|
||||
cmd := ssh.Command(ctx, host, remoteDir+"/bench-agent")
|
||||
cmd.Stderr = os.Stderr
|
||||
|
||||
stdin, err := cmd.StdinPipe()
|
||||
@@ -182,15 +151,7 @@ func drive(ctx context.Context, identity, host string, cfg Config) (*RunResult,
|
||||
}()
|
||||
|
||||
sink := &eventSink{}
|
||||
sc := bufio.NewScanner(stdout)
|
||||
sc.Buffer(make([]byte, 1<<20), 8<<20)
|
||||
for sc.Scan() {
|
||||
var ev Event
|
||||
if err := json.Unmarshal(sc.Bytes(), &ev); err != nil {
|
||||
continue
|
||||
}
|
||||
sink.handle(ev)
|
||||
}
|
||||
_ = ScanEvents(stdout, sink.handle)
|
||||
waitErr := cmd.Wait()
|
||||
if sink.fatal != "" {
|
||||
return nil, fmt.Errorf("agent: %s", sink.fatal)
|
||||
@@ -1,4 +1,4 @@
|
||||
package bench
|
||||
package resticbench
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
@@ -1,4 +1,4 @@
|
||||
package bench
|
||||
package resticbench
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
@@ -2,7 +2,7 @@
|
||||
// orchestrator on the dev machine pushes an agent (yucca-bench itself, built
|
||||
// for linux) plus a pinned restic binary to a management host, runs
|
||||
// write/incremental/restore phases there, and reports results locally.
|
||||
package bench
|
||||
package resticbench
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
@@ -0,0 +1,158 @@
|
||||
// Package sshx is yuctl's single ssh/scp layer, shelling out to the system
|
||||
// OpenSSH (agent support, ssh_config, ControlMaster — a Go ssh library would
|
||||
// reimplement all three badly). Scripts travel as the ssh command argument —
|
||||
// visible in remote ps, so they must never contain secrets; secret payloads
|
||||
// go through stdin.
|
||||
package sshx
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"sort"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
)
|
||||
|
||||
// Client carries one target family's connection policy. The zero value is a
|
||||
// plain BatchMode/TOFU client using the operator's ssh defaults and agent.
|
||||
type Client struct {
|
||||
// User is prepended to hosts that don't already embed one ("" = ssh
|
||||
// config).
|
||||
User string
|
||||
|
||||
// IdentityFile pins a private key (with IdentitiesOnly); "" = defaults.
|
||||
IdentityFile string
|
||||
|
||||
// KnownHostsFile overrides the known-hosts file. Fleets with recycled
|
||||
// provider IPs need their own — the operator's global file would scream
|
||||
// host-key-changed.
|
||||
KnownHostsFile string
|
||||
|
||||
// ControlDir enables connection multiplexing with masters persisted
|
||||
// under it. Multiplexing matters beyond latency: fan-out commands
|
||||
// otherwise open a fresh port-22 TCP flow per host per command, a burst
|
||||
// pattern that trips ssh-targeted rate limiting on some paths (observed
|
||||
// as flapping per-IP port-22 SYN drops while ICMP and other ports stay
|
||||
// fine). One persistent master per host keeps the flow count flat.
|
||||
ControlDir string
|
||||
|
||||
// ConnectTimeoutSeconds bounds connection establishment (0 = ssh default).
|
||||
ConnectTimeoutSeconds int
|
||||
|
||||
// Retries is how many times Run re-attempts after a connection-level
|
||||
// failure (ssh exit 255: banner timeouts, resets — common when tens of
|
||||
// sessions open against fresh hosts). Remote command failures are never
|
||||
// retried; ssh reports those as the command's own exit code.
|
||||
Retries int
|
||||
}
|
||||
|
||||
func (c *Client) dest(host string) string {
|
||||
if c.User != "" && !strings.Contains(host, "@") {
|
||||
return c.User + "@" + host
|
||||
}
|
||||
return host
|
||||
}
|
||||
|
||||
// options are the -o/-i arguments shared by ssh and scp. accept-new (TOFU)
|
||||
// keeps first contact with a freshly resolved IP from failing BatchMode.
|
||||
func (c *Client) options() []string {
|
||||
args := []string{
|
||||
"-o", "BatchMode=yes",
|
||||
"-o", "StrictHostKeyChecking=accept-new",
|
||||
"-o", "ServerAliveInterval=30",
|
||||
"-o", "ServerAliveCountMax=8",
|
||||
}
|
||||
if c.KnownHostsFile != "" {
|
||||
args = append(args, "-o", "UserKnownHostsFile="+c.KnownHostsFile)
|
||||
}
|
||||
if c.ConnectTimeoutSeconds > 0 {
|
||||
args = append(args, "-o", "ConnectTimeout="+strconv.Itoa(c.ConnectTimeoutSeconds))
|
||||
}
|
||||
if c.ControlDir != "" {
|
||||
args = append(args,
|
||||
"-o", "ControlMaster=auto",
|
||||
"-o", "ControlPath="+filepath.Join(c.ControlDir, "%C"),
|
||||
"-o", "ControlPersist=300")
|
||||
}
|
||||
if c.IdentityFile != "" {
|
||||
args = append(args, "-i", c.IdentityFile, "-o", "IdentitiesOnly=yes")
|
||||
}
|
||||
return args
|
||||
}
|
||||
|
||||
func (c *Client) Run(ctx context.Context, host, script string, stdin []byte) (string, error) {
|
||||
var lastOut string
|
||||
var lastErr error
|
||||
for attempt := 1; attempt <= c.Retries+1; attempt++ {
|
||||
out, err := c.RunOnce(ctx, host, script, stdin)
|
||||
if err == nil {
|
||||
return out, nil
|
||||
}
|
||||
lastOut, lastErr = out, err
|
||||
var exit *exec.ExitError
|
||||
if ctx.Err() != nil || !errors.As(err, &exit) || exit.ExitCode() != 255 {
|
||||
break
|
||||
}
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return lastOut, lastErr
|
||||
case <-time.After(time.Duration(attempt) * 5 * time.Second):
|
||||
}
|
||||
}
|
||||
return lastOut, lastErr
|
||||
}
|
||||
|
||||
// RunOnce is a single ssh attempt with no retry — for callers with their own
|
||||
// retry cadence (readiness polling), where nesting retries multiplies delays.
|
||||
func (c *Client) RunOnce(ctx context.Context, host, script string, stdin []byte) (string, error) {
|
||||
cmd := exec.CommandContext(ctx, "ssh", append(c.options(), c.dest(host), script)...)
|
||||
if stdin != nil {
|
||||
cmd.Stdin = bytes.NewReader(stdin)
|
||||
}
|
||||
var out, errb bytes.Buffer
|
||||
cmd.Stdout = &out
|
||||
cmd.Stderr = &errb
|
||||
if err := cmd.Run(); err != nil {
|
||||
return out.String(), fmt.Errorf("ssh %s: %w: %s", host, err, Tail(errb.String(), 1000))
|
||||
}
|
||||
return out.String(), nil
|
||||
}
|
||||
|
||||
// Command is for long-lived streaming sessions; the caller owns the pipes and
|
||||
// lifecycle.
|
||||
func (c *Client) Command(ctx context.Context, host, remote string) *exec.Cmd {
|
||||
return exec.CommandContext(ctx, "ssh", append(c.options(), c.dest(host), remote)...)
|
||||
}
|
||||
|
||||
// Push copies local files (keys) into remoteDir on host under the given
|
||||
// remote names (values), and marks them executable.
|
||||
func (c *Client) Push(ctx context.Context, host, remoteDir string, files map[string]string) error {
|
||||
if _, err := c.Run(ctx, host, "mkdir -p "+remoteDir, nil); err != nil {
|
||||
return err
|
||||
}
|
||||
names := make([]string, 0, len(files))
|
||||
for local, name := range files {
|
||||
cmd := exec.CommandContext(ctx, "scp",
|
||||
append(append([]string{"-q"}, c.options()...), local, c.dest(host)+":"+remoteDir+"/"+name)...)
|
||||
if out, err := cmd.CombinedOutput(); err != nil {
|
||||
return fmt.Errorf("scp %s to %s: %w: %s", local, host, err, Tail(string(out), 500))
|
||||
}
|
||||
names = append(names, remoteDir+"/"+name)
|
||||
}
|
||||
sort.Strings(names)
|
||||
_, err := c.Run(ctx, host, "chmod +x "+strings.Join(names, " "), nil)
|
||||
return err
|
||||
}
|
||||
|
||||
func Tail(v string, n int) string {
|
||||
v = strings.TrimSpace(v)
|
||||
if len(v) > n {
|
||||
return "…" + v[len(v)-n:]
|
||||
}
|
||||
return v
|
||||
}
|
||||
@@ -1,7 +1,7 @@
|
||||
// Package k8s wraps Talos node operations (talosctl) for a region's K8s cluster,
|
||||
// Package talos wraps Talos node operations (talosctl) for a region's K8s cluster,
|
||||
// resolving the talosconfig from 1Password to a 0600 temp file and driving
|
||||
// `talosctl upgrade` against the control-plane node IPs from discovery.
|
||||
package k8s
|
||||
package talos
|
||||
|
||||
import (
|
||||
"context"
|
||||
@@ -11,8 +11,8 @@ import (
|
||||
|
||||
"github.com/rs/zerolog"
|
||||
|
||||
"yuctl/internal/op"
|
||||
"yuctl/internal/state"
|
||||
"yuctl/op"
|
||||
"yuctl/state"
|
||||
)
|
||||
|
||||
// UpgradeOptions controls a talos upgrade run.
|
||||
@@ -28,7 +28,7 @@ type UpgradeOptions struct {
|
||||
// TalosUpgrade resolves the talosconfig referenced by the kubernetes payload and
|
||||
// runs `talosctl upgrade` against each control-plane node. The temp talosconfig
|
||||
// is removed on return. With DryRun the commands are only logged.
|
||||
func TalosUpgrade(ctx context.Context, k state.Kubernetes, opts UpgradeOptions, logger zerolog.Logger) error {
|
||||
func Upgrade(ctx context.Context, k state.Kubernetes, opts UpgradeOptions, logger zerolog.Logger) error {
|
||||
nodes := opts.Nodes
|
||||
if len(nodes) == 0 {
|
||||
nodes = k.CPNodeIPs
|
||||
@@ -0,0 +1,20 @@
|
||||
// Package ui is the terminal presentation layer. Commands write through an
|
||||
// IOStreams instead of os.Stdout/os.Stderr so tests can capture output; Out
|
||||
// is for the command's payload (parseable, redirectable), Err for progress
|
||||
// and human-only chatter.
|
||||
package ui
|
||||
|
||||
import (
|
||||
"io"
|
||||
"os"
|
||||
)
|
||||
|
||||
type IOStreams struct {
|
||||
In io.Reader
|
||||
Out io.Writer
|
||||
Err io.Writer
|
||||
}
|
||||
|
||||
func System() *IOStreams {
|
||||
return &IOStreams{In: os.Stdin, Out: os.Stdout, Err: os.Stderr}
|
||||
}
|
||||
@@ -0,0 +1,26 @@
|
||||
package ui
|
||||
|
||||
import "github.com/charmbracelet/lipgloss"
|
||||
|
||||
var (
|
||||
accent = lipgloss.AdaptiveColor{Light: "#6C50FF", Dark: "#9D7CFF"}
|
||||
good = lipgloss.AdaptiveColor{Light: "#12A150", Dark: "#2ECC71"}
|
||||
bad = lipgloss.AdaptiveColor{Light: "#D0021B", Dark: "#FF5F56"}
|
||||
warn = lipgloss.AdaptiveColor{Light: "#B8860B", Dark: "#F5C542"}
|
||||
muted = lipgloss.AdaptiveColor{Light: "244", Dark: "241"}
|
||||
tx = lipgloss.AdaptiveColor{Light: "#0087AF", Dark: "#33D1E0"}
|
||||
rx = lipgloss.AdaptiveColor{Light: "#AF5F00", Dark: "#F5A623"}
|
||||
)
|
||||
|
||||
var (
|
||||
Badge = lipgloss.NewStyle().Bold(true).Foreground(lipgloss.Color("#FFFFFF")).Background(accent).Padding(0, 1)
|
||||
Title = lipgloss.NewStyle().Bold(true)
|
||||
Muted = lipgloss.NewStyle().Foreground(muted)
|
||||
OK = lipgloss.NewStyle().Foreground(good)
|
||||
Bad = lipgloss.NewStyle().Bold(true).Foreground(bad)
|
||||
Warn = lipgloss.NewStyle().Foreground(warn)
|
||||
TX = lipgloss.NewStyle().Foreground(tx)
|
||||
RX = lipgloss.NewStyle().Foreground(rx)
|
||||
Total = lipgloss.NewStyle().Bold(true).Foreground(accent)
|
||||
Frame = lipgloss.NewStyle().Border(lipgloss.RoundedBorder()).BorderForeground(accent).Padding(0, 1)
|
||||
)
|
||||
@@ -0,0 +1,61 @@
|
||||
package ui
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// Meter renders a horizontal bar of width w, filled to value/scale.
|
||||
func Meter(value, scale float64, w int) string {
|
||||
filled := 0
|
||||
if scale > 0 {
|
||||
filled = int(value / scale * float64(w))
|
||||
}
|
||||
filled = min(max(filled, 0), w)
|
||||
return strings.Repeat("█", filled) + strings.Repeat("░", w-filled)
|
||||
}
|
||||
|
||||
// Sparkline renders the last len(vals) samples with block glyphs, scaled to
|
||||
// the window maximum, downsampled to at most w points.
|
||||
func Sparkline(vals []float64, w int) string {
|
||||
if len(vals) > w {
|
||||
vals = vals[len(vals)-w:]
|
||||
}
|
||||
var maxV float64
|
||||
for _, v := range vals {
|
||||
maxV = max(maxV, v)
|
||||
}
|
||||
if maxV == 0 {
|
||||
return strings.Repeat("▁", len(vals))
|
||||
}
|
||||
glyphs := []rune("▁▂▃▄▅▆▇█")
|
||||
var b strings.Builder
|
||||
for _, v := range vals {
|
||||
i := int(v / maxV * float64(len(glyphs)-1))
|
||||
b.WriteRune(glyphs[min(max(i, 0), len(glyphs)-1)])
|
||||
}
|
||||
return b.String()
|
||||
}
|
||||
|
||||
func FmtGbps(bps float64) string {
|
||||
return fmt.Sprintf("%.2f Gbps", bps/1e9)
|
||||
}
|
||||
|
||||
func PadGbps(bps float64) string {
|
||||
return fmt.Sprintf("%11s", FmtGbps(bps))
|
||||
}
|
||||
|
||||
func ErrCell(n, width int) string {
|
||||
s := fmt.Sprintf("%*d", width, n)
|
||||
if n > 0 {
|
||||
return Bad.Render(s)
|
||||
}
|
||||
return Muted.Render(s)
|
||||
}
|
||||
|
||||
func Truncate(s string, n int) string {
|
||||
if len(s) <= n {
|
||||
return s
|
||||
}
|
||||
return s[:n] + "…"
|
||||
}
|
||||
@@ -0,0 +1,18 @@
|
||||
package ui
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestSparklineAndMeter(t *testing.T) {
|
||||
if got := Sparkline([]float64{0, 1, 2, 4}, 10); len([]rune(got)) != 4 {
|
||||
t.Errorf("sparkline length: %q", got)
|
||||
}
|
||||
if got := Meter(50, 100, 10); !strings.HasPrefix(got, "█████░") {
|
||||
t.Errorf("meter: %q", got)
|
||||
}
|
||||
if got := Meter(0, 0, 4); got != "░░░░" {
|
||||
t.Errorf("empty meter: %q", got)
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user