npm audit --audit-level=high now exits 0. The advisories were in dev-only dependencies (braces via jest 29).
198 lines
6.3 KiB
Python
198 lines
6.3 KiB
Python
# flake8: noqa: E501 (demo data and canned output contain long literal lines)
|
|
import json
|
|
import time
|
|
import urllib.request
|
|
|
|
B = "http://127.0.0.1:8400"
|
|
H = {
|
|
"Remote-User": "alex.rivera",
|
|
"Remote-Name": "Alex Rivera",
|
|
"Remote-Email": "alex@example.com",
|
|
"Remote-Groups": "admin,employee",
|
|
"Content-Type": "application/json",
|
|
}
|
|
|
|
|
|
def call(method, path, body=None):
|
|
req = urllib.request.Request(
|
|
B + path, method=method, headers=H, data=json.dumps(body).encode() if body is not None else None
|
|
)
|
|
try:
|
|
with urllib.request.urlopen(req, timeout=30) as r:
|
|
return json.loads(r.read() or b"{}")
|
|
except urllib.error.HTTPError as e:
|
|
return {"error": e.code, "body": e.read().decode()[:200]}
|
|
|
|
|
|
workers = {w["name"]: w["id"] for w in call("GET", "/api/workers")}
|
|
print(workers)
|
|
wf = {}
|
|
wf["health"] = call(
|
|
"POST",
|
|
"/api/workflows",
|
|
{
|
|
"name": "Node health check",
|
|
"description": "Uptime, root filesystem and memory on every online worker",
|
|
"definition": {
|
|
"steps": [
|
|
{"id": "uptime", "name": "Uptime", "type": "execute", "command": "uptime", "targets": ["all"]},
|
|
{
|
|
"id": "disk",
|
|
"name": "Root filesystem usage",
|
|
"type": "execute",
|
|
"command": "df -h / | tail -1",
|
|
"targets": ["all"],
|
|
},
|
|
{"id": "mem", "name": "Memory", "type": "execute", "command": "free -h | head -2", "targets": ["all"]},
|
|
]
|
|
},
|
|
},
|
|
)["id"]
|
|
wf["restart"] = call(
|
|
"POST",
|
|
"/api/workflows",
|
|
{
|
|
"name": "Rolling restart (with approval)",
|
|
"description": "Checks state, asks a human to approve, restarts one node, verifies",
|
|
"definition": {
|
|
"steps": [
|
|
{
|
|
"id": "check",
|
|
"name": "Check cluster state",
|
|
"type": "execute",
|
|
"command": 'echo "all 3 nodes healthy"; date',
|
|
"targets": ["all"],
|
|
},
|
|
{
|
|
"id": "ask",
|
|
"name": "Approve restart on node-01?",
|
|
"type": "prompt",
|
|
"message": "All nodes are healthy. Restart the service on node-01?",
|
|
"options": ["Proceed", "Abort"],
|
|
"key": "approval",
|
|
"routes": {"Abort": "end"},
|
|
},
|
|
{
|
|
"id": "restart",
|
|
"name": "Restart service on node-01",
|
|
"type": "execute",
|
|
"command": 'echo "stopping service"; sleep 1; echo "starting service"; echo "service active"',
|
|
"targets": ["node-01"],
|
|
},
|
|
{"id": "settle", "name": "Let it settle", "type": "wait", "duration": 2},
|
|
{
|
|
"id": "verify",
|
|
"name": "Verify",
|
|
"type": "execute",
|
|
"command": 'echo "verification passed on $(hostname)"',
|
|
"targets": ["node-01"],
|
|
},
|
|
]
|
|
},
|
|
},
|
|
)["id"]
|
|
wf["route"] = call(
|
|
"POST",
|
|
"/api/workflows",
|
|
{
|
|
"name": "Capacity check with routing",
|
|
"description": "Parses KEY=VALUE output and routes on the result",
|
|
"definition": {
|
|
"steps": [
|
|
{
|
|
"id": "probe",
|
|
"name": "Probe capacity",
|
|
"type": "execute",
|
|
"command": 'echo "FREE_PCT=12"; echo "POOL=app-nvme"',
|
|
"targets": ["node-02"],
|
|
},
|
|
{"id": "parse", "name": "Parse results", "type": "parse"},
|
|
{
|
|
"id": "decide",
|
|
"name": "Decide",
|
|
"type": "route",
|
|
"conditions": [
|
|
{"if": "parseInt(state.free_pct) < 20", "goto": "warn", "label": "Low capacity"},
|
|
{"default": True, "goto": "end"},
|
|
],
|
|
},
|
|
{
|
|
"id": "warn",
|
|
"name": "Raise warning",
|
|
"type": "execute",
|
|
"command": 'echo "WARNING: pool is below 20% free"; exit 0',
|
|
"targets": ["node-02"],
|
|
},
|
|
]
|
|
},
|
|
},
|
|
)["id"]
|
|
wf["broken"] = call(
|
|
"POST",
|
|
"/api/workflows",
|
|
{
|
|
"name": "Check backup mount",
|
|
"description": "Fails when the backup mount is missing",
|
|
"definition": {
|
|
"steps": [
|
|
{
|
|
"id": "mnt",
|
|
"name": "Backup mount present?",
|
|
"type": "execute",
|
|
"command": "ls /mnt/does-not-exist-backup",
|
|
"targets": ["node-03"],
|
|
}
|
|
]
|
|
},
|
|
},
|
|
)["id"]
|
|
print(wf)
|
|
|
|
|
|
def run(k, params=None):
|
|
r = call("POST", "/api/executions", {"workflow_id": wf[k], "params": params or {}})
|
|
time.sleep(1.2)
|
|
return r
|
|
|
|
|
|
# quick commands first (history)
|
|
for name, cmd in (
|
|
("node-01", "uname -a"),
|
|
("node-02", "df -h /"),
|
|
("node-03", "free -h"),
|
|
("node-01", "uptime"),
|
|
("node-02", "lscpu | head -8"),
|
|
):
|
|
call("POST", f"/api/workers/{workers[name]}/command", {"command": cmd})
|
|
time.sleep(1.2)
|
|
for k in ("health", "health", "route", "broken"):
|
|
print(k, run(k).get("id"))
|
|
time.sleep(3)
|
|
r = run("restart")
|
|
time.sleep(2.5)
|
|
print("restart#1", r.get("id"))
|
|
print(call("POST", f"/api/executions/{r['id']}/respond", {"response": "Proceed"}))
|
|
time.sleep(6)
|
|
run("health")
|
|
time.sleep(3)
|
|
r2 = run("restart")
|
|
print("restart waiting", r2.get("id")) # left waiting for approval on purpose
|
|
for name, cmd, t, v in (
|
|
("Disk usage report", "df -h /", "interval", "60"),
|
|
("Memory snapshot", "free -h | head -2", "hourly", "1"),
|
|
("Uptime log", "uptime", "interval", "120"),
|
|
):
|
|
print(
|
|
call(
|
|
"POST",
|
|
"/api/scheduled-commands",
|
|
{
|
|
"name": name,
|
|
"command": cmd,
|
|
"worker_ids": [workers["node-01"], workers["node-02"]],
|
|
"schedule_type": t,
|
|
"schedule_value": v,
|
|
},
|
|
)
|
|
)
|