testing and documentation
This commit is contained in:
@@ -0,0 +1,800 @@
|
||||
#!/usr/bin/env python3
|
||||
"""bench.py — ECHO performance test suite (timing harness).
|
||||
|
||||
Measures the operations the 1.1.0 (pooling + concurrent read_many + shared cache)
|
||||
and 1.2.0 (fuzzy resolve / alias learning) updates targeted, and gates them on
|
||||
RELATIVE ratios (portable across machines/WAN) rather than absolute ms.
|
||||
|
||||
It reuses the live `echo` client module so it times the REAL pooled/concurrent
|
||||
code path. Reads are non-destructive; write metrics live in a disposable
|
||||
namespace (_agent/_bench/<run-id>/) cleaned up at the end unless --keep.
|
||||
|
||||
This is a MANUAL / pre-release tool, not a CI gate — it depends on the live
|
||||
endpoint (echoapi.alwisp.com) and WAN latency.
|
||||
|
||||
python3 bench.py <metric> [options]
|
||||
python3 bench.py all # the read-only safe set
|
||||
python3 bench.py read-full -n 10
|
||||
python3 bench.py subject-pull
|
||||
python3 bench.py bulk-put -k 50
|
||||
python3 bench.py worker-sweep
|
||||
python3 bench.py resolve # 1.2.0 fuzzy correctness + timing
|
||||
python3 bench.py --list
|
||||
|
||||
Metrics: read-full, subject-pull, bulk-get, bulk-put, bulk-append,
|
||||
pool-warmup, worker-sweep, resolve, index, lint, lock, queue, cache, soak
|
||||
|
||||
Options:
|
||||
-n / --iterations N timed iterations (default 10 read / 5 write)
|
||||
-k / --count K notes/appends for bulk metrics (default 50)
|
||||
--workers W override ECHO_WORKERS for this run
|
||||
--warmup M warm-up iterations to discard (default 2)
|
||||
--soak-seconds S duration for the soak metric (default 60)
|
||||
--keep do NOT delete the _bench namespace after the run
|
||||
--no-gate skip relative-ratio gating (report numbers only)
|
||||
--json-out PATH write the results JSON here (default results/<date>-<commit>.json)
|
||||
--quiet suppress the per-metric human table (JSON only)
|
||||
|
||||
Exit: 0 all gates pass (or --no-gate) · 1 a gate failed · 2 usage/setup error.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import contextlib
|
||||
import datetime as dt
|
||||
import io
|
||||
import json
|
||||
import os
|
||||
import statistics
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
import uuid
|
||||
from pathlib import Path
|
||||
|
||||
HERE = Path(__file__).resolve().parent
|
||||
SCRIPTS = HERE.parent.parent / "scripts"
|
||||
sys.path.insert(0, str(SCRIPTS))
|
||||
|
||||
import echo # noqa: E402 the validated client — real pooled/concurrent path
|
||||
import echo_index # noqa: E402 resolve / fuzzy_candidates (1.2.0)
|
||||
import echo_recall # noqa: E402 recall (search + 1-hop expansion)
|
||||
|
||||
BASELINES = HERE / "baselines.json"
|
||||
RESULTS_DIR = HERE / "results"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# stats + timing
|
||||
# ---------------------------------------------------------------------------
|
||||
def stats(samples: list[float]) -> dict:
|
||||
"""min/median/p90/p95/max/mean in milliseconds from a list of seconds."""
|
||||
if not samples:
|
||||
return {}
|
||||
ms = sorted(s * 1000.0 for s in samples)
|
||||
|
||||
def pct(p: float) -> float:
|
||||
if len(ms) == 1:
|
||||
return ms[0]
|
||||
k = (len(ms) - 1) * p
|
||||
lo, hi = int(k), min(int(k) + 1, len(ms) - 1)
|
||||
return ms[lo] + (ms[hi] - ms[lo]) * (k - lo)
|
||||
|
||||
return {
|
||||
"n": len(ms),
|
||||
"min_ms": round(ms[0], 2),
|
||||
"median_ms": round(statistics.median(ms), 2),
|
||||
"p90_ms": round(pct(0.90), 2),
|
||||
"p95_ms": round(pct(0.95), 2),
|
||||
"max_ms": round(ms[-1], 2),
|
||||
"mean_ms": round(statistics.fmean(ms), 2),
|
||||
}
|
||||
|
||||
|
||||
def timed(fn, n: int, warmup: int) -> dict:
|
||||
"""Run fn() warmup+n times. Returns warm-up samples separately from the timed
|
||||
set so cold-start cost (the thing keep-alive kills) is visible, not averaged in."""
|
||||
warm: list[float] = []
|
||||
samples: list[float] = []
|
||||
for i in range(warmup + n):
|
||||
t0 = time.perf_counter()
|
||||
fn()
|
||||
dt_s = time.perf_counter() - t0
|
||||
(warm if i < warmup else samples).append(dt_s)
|
||||
out = stats(samples)
|
||||
if warm:
|
||||
out["warmup"] = stats(warm)
|
||||
return out
|
||||
|
||||
|
||||
def silent(fn):
|
||||
"""Run fn() with stdout swallowed (recall/resolve print a lot)."""
|
||||
def wrapped():
|
||||
with contextlib.redirect_stdout(io.StringIO()):
|
||||
return fn()
|
||||
return wrapped
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# vault enumeration (mirrors sweep.py / vault_lint.py)
|
||||
# ---------------------------------------------------------------------------
|
||||
SKIP_BASENAMES = {"README.md"}
|
||||
|
||||
|
||||
def _list_dir(path: str):
|
||||
p = "" if path in ("", "/") else (path if path.endswith("/") else path + "/")
|
||||
body = echo.get_text(p)
|
||||
if body is None:
|
||||
return [], []
|
||||
try:
|
||||
j = json.loads(body)
|
||||
except json.JSONDecodeError:
|
||||
return [], []
|
||||
entries = list(j.get("files", [])) + list(j.get("folders", []))
|
||||
files = [e for e in entries if not e.endswith("/")]
|
||||
folders = [e[:-1] for e in entries if e.endswith("/")]
|
||||
return files, folders
|
||||
|
||||
|
||||
def walk_vault(prefix: str = ""):
|
||||
files, folders = _list_dir(prefix)
|
||||
for f in files:
|
||||
yield prefix + f
|
||||
for d in folders:
|
||||
yield from walk_vault(f"{prefix}{d}/")
|
||||
|
||||
|
||||
def all_notes() -> list[str]:
|
||||
return [p for p in walk_vault()
|
||||
if p.endswith(".md") and p.rsplit("/", 1)[-1] not in SKIP_BASENAMES]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# write primitives (replicate echo.py semantics so we time the real cost)
|
||||
# ---------------------------------------------------------------------------
|
||||
def put_verified(path: str, body: str) -> None:
|
||||
"""PUT then read-back GET — the same verify echo.py cmd_put does."""
|
||||
status, b = echo.request("PUT", echo.vault_url(path),
|
||||
data=body.encode(), headers={"Content-Type": "text/markdown"})
|
||||
echo.check(status, b, f"put {path}")
|
||||
status, _ = echo.request("GET", echo.vault_url(path))
|
||||
if status != 200:
|
||||
raise echo.EchoError(f"put {path}: did not verify (GET {status})")
|
||||
|
||||
|
||||
def append_line(path: str, line: str) -> bool:
|
||||
"""Whole-line idempotent append (mirrors echo.py cmd_append). Returns True if a
|
||||
write happened, False if it was an idempotent skip."""
|
||||
status, body = echo.request("GET", echo.vault_url(path))
|
||||
if status == 200:
|
||||
existing = [ln.rstrip("\r") for ln in body.decode(errors="replace").splitlines()]
|
||||
if line.rstrip("\r") in existing:
|
||||
return False
|
||||
status, body = echo.request("POST", echo.vault_url(path),
|
||||
data=f"{line}\n".encode(),
|
||||
headers={"Content-Type": "text/markdown"})
|
||||
echo.check(status, body, f"append {path}")
|
||||
return True
|
||||
|
||||
|
||||
def delete_path(path: str) -> None:
|
||||
echo.request("DELETE", echo.vault_url(path))
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# metrics
|
||||
# ---------------------------------------------------------------------------
|
||||
def m_read_full(ctx) -> dict:
|
||||
"""§3.1 full-vault read: serial GET vs concurrent read_many -> speedup ratio."""
|
||||
paths = all_notes()
|
||||
if not paths:
|
||||
raise echo.EchoError("read-full: no notes enumerated (vault unreachable?)")
|
||||
|
||||
def serial():
|
||||
for p in paths:
|
||||
echo.get_text(p)
|
||||
|
||||
def concurrent():
|
||||
echo.read_many(paths)
|
||||
|
||||
# one warm pass so neither side pays the cold-handshake tax (pool-warmup owns that)
|
||||
echo.get_text(paths[0])
|
||||
# serial is a slow reference baseline (N serial round-trips); one warm pass is
|
||||
# enough to establish the ratio, so we don't multiply the full-vault serial cost.
|
||||
serial_t = timed(serial, n=1, warmup=0)
|
||||
concurrent_t = timed(concurrent, n=ctx.n, warmup=ctx.warmup)
|
||||
ratio = (serial_t["median_ms"] / concurrent_t["median_ms"]) if concurrent_t.get("median_ms") else None
|
||||
return {
|
||||
"note_count": len(paths),
|
||||
"workers": echo.MAX_WORKERS,
|
||||
"serial": serial_t,
|
||||
"concurrent": concurrent_t,
|
||||
"speedup_ratio": round(ratio, 2) if ratio else None,
|
||||
"notes_per_sec_concurrent": round(len(paths) / (concurrent_t["median_ms"] / 1000.0), 1)
|
||||
if concurrent_t.get("median_ms") else None,
|
||||
}
|
||||
|
||||
|
||||
def m_subject_pull(ctx) -> dict:
|
||||
"""§3.2 recall a subject: time resolve (floor), search, and full recall, split out."""
|
||||
import fixtures
|
||||
index = echo_index.load()
|
||||
out = {"subjects": []}
|
||||
for label, query in fixtures.SUBJECTS:
|
||||
# floor: O(1) index resolve
|
||||
resolve_t = timed(lambda: echo_index.resolve(index, query), n=ctx.n, warmup=ctx.warmup)
|
||||
# search-only layer
|
||||
def search():
|
||||
with contextlib.redirect_stdout(io.StringIO()):
|
||||
echo.request("POST", f"{echo.BASE}/search/simple/?query={query}")
|
||||
search_t = timed(search, n=ctx.n, warmup=ctx.warmup)
|
||||
# full recall (search + 1-hop expansion reads)
|
||||
recall_t = timed(silent(lambda: echo_recall.recall(query)), n=ctx.n, warmup=ctx.warmup)
|
||||
out["subjects"].append({
|
||||
"label": label, "query": query,
|
||||
"resolve_floor": resolve_t,
|
||||
"search_only": search_t,
|
||||
"recall_total": recall_t,
|
||||
})
|
||||
return out
|
||||
|
||||
|
||||
def m_bulk_get(ctx) -> dict:
|
||||
"""§3.3 bulk GET at a controlled K: serial vs read_many over fixture notes."""
|
||||
import fixtures
|
||||
paths = _seed_notes(ctx, fixtures)
|
||||
|
||||
def serial():
|
||||
for p in paths:
|
||||
echo.get_text(p)
|
||||
|
||||
def concurrent():
|
||||
echo.read_many(paths)
|
||||
|
||||
serial_t = timed(serial, n=ctx.n, warmup=1)
|
||||
concurrent_t = timed(concurrent, n=ctx.n, warmup=ctx.warmup)
|
||||
ratio = (serial_t["median_ms"] / concurrent_t["median_ms"]) if concurrent_t.get("median_ms") else None
|
||||
return {"k": len(paths), "workers": echo.MAX_WORKERS,
|
||||
"serial": serial_t, "concurrent": concurrent_t,
|
||||
"speedup_ratio": round(ratio, 2) if ratio else None}
|
||||
|
||||
|
||||
def m_bulk_put(ctx) -> dict:
|
||||
"""§3.3 bulk PUT (read-back verified) -> puts/sec + per-put percentiles."""
|
||||
import fixtures
|
||||
k = ctx.k
|
||||
samples: list[float] = []
|
||||
for i in range(k):
|
||||
body = fixtures.note_body(ctx.run_id, i)
|
||||
path = fixtures.note_path(ctx.run_id, i)
|
||||
t0 = time.perf_counter()
|
||||
put_verified(path, body)
|
||||
samples.append(time.perf_counter() - t0)
|
||||
s = stats(samples)
|
||||
total_s = sum(samples)
|
||||
return {"k": k, "per_put": s, "total_ms": round(total_s * 1000, 1),
|
||||
"puts_per_sec": round(k / total_s, 1) if total_s else None}
|
||||
|
||||
|
||||
def m_bulk_append(ctx) -> dict:
|
||||
"""§3.3 bulk APPEND + idempotent-skip path. Confirms idempotency isn't O(file)."""
|
||||
import fixtures
|
||||
target = fixtures.append_target(ctx.run_id)
|
||||
put_verified(target, fixtures.append_target_seed(ctx.run_id))
|
||||
k = ctx.k
|
||||
lines = [f"- 2026-06-23: bench append line {i:04d}" for i in range(k)]
|
||||
|
||||
write_samples: list[float] = []
|
||||
for ln in lines:
|
||||
t0 = time.perf_counter()
|
||||
wrote = append_line(target, ln)
|
||||
write_samples.append(time.perf_counter() - t0)
|
||||
if not wrote:
|
||||
raise echo.EchoError("bulk-append: expected a write but got an idempotent skip")
|
||||
|
||||
# second pass: every line already present -> must skip, and skip cost is the
|
||||
# idempotency GET on a now-larger file. Confirms no superlinear blowup.
|
||||
skip_samples: list[float] = []
|
||||
skips = 0
|
||||
for ln in lines:
|
||||
t0 = time.perf_counter()
|
||||
wrote = append_line(target, ln)
|
||||
skip_samples.append(time.perf_counter() - t0)
|
||||
if not wrote:
|
||||
skips += 1
|
||||
|
||||
return {
|
||||
"k": k,
|
||||
"append_write": stats(write_samples),
|
||||
"append_skip": stats(skip_samples),
|
||||
"skips_confirmed": skips,
|
||||
"all_skipped_second_pass": skips == k,
|
||||
}
|
||||
|
||||
|
||||
def m_pool_warmup(ctx) -> dict:
|
||||
"""§4.1 cold vs warm single GET — the clearest proof keep-alive is alive."""
|
||||
paths = all_notes()
|
||||
probe = paths[0] if paths else "_agent/echo-vault.md"
|
||||
echo._drop_connection() # force a cold handshake
|
||||
t0 = time.perf_counter()
|
||||
echo.get_text(probe)
|
||||
cold_ms = (time.perf_counter() - t0) * 1000.0
|
||||
warm = timed(lambda: echo.get_text(probe), n=max(ctx.n, 10), warmup=1)
|
||||
ratio = round(cold_ms / warm["median_ms"], 2) if warm.get("median_ms") else None
|
||||
return {"probe": probe, "cold_ms": round(cold_ms, 2), "warm": warm,
|
||||
"cold_over_warm_ratio": ratio}
|
||||
|
||||
|
||||
def m_worker_sweep(ctx) -> dict:
|
||||
"""§4.2 concurrency scaling: full-vault concurrent read at workers 1,2,4,8,16."""
|
||||
paths = all_notes()
|
||||
if not paths:
|
||||
raise echo.EchoError("worker-sweep: no notes enumerated")
|
||||
original = echo.MAX_WORKERS
|
||||
rows = []
|
||||
try:
|
||||
# w=1 is a full serial pass (slow); the curve only needs the shape, so use a
|
||||
# modest sample count and a single pre-warm rather than per-setting warmups.
|
||||
reps = max(1, ctx.n // 2)
|
||||
for w in (1, 2, 4, 8, 16):
|
||||
echo.MAX_WORKERS = w
|
||||
echo.read_many(paths[:5]) # warm the pool at this worker count
|
||||
t = timed(lambda: echo.read_many(paths), n=reps, warmup=0)
|
||||
rows.append({"workers": w, "median_ms": t["median_ms"], "p95_ms": t["p95_ms"]})
|
||||
finally:
|
||||
echo.MAX_WORKERS = original
|
||||
best = min(rows, key=lambda r: r["median_ms"])
|
||||
return {"note_count": len(paths), "sweep": rows, "knee_workers": best["workers"],
|
||||
"default_workers": original}
|
||||
|
||||
|
||||
def m_resolve(ctx) -> dict:
|
||||
"""§4.3 fuzzy resolve — timing AND correctness against the 1.2.0 guard table."""
|
||||
import fixtures
|
||||
index = echo_index.load()
|
||||
cases = []
|
||||
passes = info = 0
|
||||
for case in fixtures.RESOLVE_CASES:
|
||||
mention = case["mention"]
|
||||
exact_t = timed(lambda: echo_index.resolve(index, mention), n=ctx.n, warmup=ctx.warmup)
|
||||
fuzzy_t = timed(lambda: echo_index.fuzzy_candidates(index, mention), n=ctx.n, warmup=ctx.warmup)
|
||||
_, entity = echo_index.resolve(index, mention)
|
||||
exact_slug = (entity or {}).get("slug") or _slug_from_entity(entity)
|
||||
cands = echo_index.fuzzy_candidates(index, mention) or []
|
||||
cand_slugs = [_cand_slug(c) for c in cands]
|
||||
|
||||
expect, want = case["expect"], case.get("expect_slug")
|
||||
if expect == "exact":
|
||||
ok = exact_slug == want
|
||||
elif expect == "candidates":
|
||||
ok = (exact_slug != want) and (want in cand_slugs)
|
||||
else: # miss
|
||||
ok = entity is None
|
||||
verdict = "PASS" if ok else "INFO" # tune table to live vault before promoting to FAIL
|
||||
passes += ok
|
||||
info += (not ok)
|
||||
cases.append({"mention": mention, "expect": expect, "expect_slug": want,
|
||||
"resolved_slug": exact_slug, "candidate_slugs": cand_slugs[:5],
|
||||
"verdict": verdict,
|
||||
"resolve_timing": exact_t, "fuzzy_timing": fuzzy_t})
|
||||
return {"cases": cases, "passes": passes, "needs_tuning": info,
|
||||
"note": "verdicts are PASS/INFO only; promote to FAIL once the table is "
|
||||
"confirmed against the live vault"}
|
||||
|
||||
|
||||
def _serial_expand(seeds, nmap, base, max_hops):
|
||||
"""The pre-fix serial graph BFS, kept here as the reference baseline so the suite
|
||||
both (a) measures the concurrency speedup and (b) asserts the concurrent
|
||||
expand_graph still returns byte-identical ranking. If echo_recall.expand_graph ever
|
||||
regresses to serial, the ratio gate fails; if its results drift, the parity check fails."""
|
||||
from collections import deque
|
||||
score_of = dict(base)
|
||||
seen = set(seeds)
|
||||
results = {}
|
||||
dq = deque((s, 0) for s in seeds)
|
||||
while dq:
|
||||
path, hop = dq.popleft()
|
||||
if hop >= max_hops:
|
||||
continue
|
||||
text = echo_recall.links.get_text(path)
|
||||
if text is None:
|
||||
continue
|
||||
body = echo_recall.strip_frontmatter(text)
|
||||
targets = set(echo_recall.links.all_wikilinks(body)) | set(echo_recall.links.source_notes(text))
|
||||
parent = score_of.get(path, 1.0)
|
||||
for t in targets:
|
||||
tp = echo_recall._resolve_target(t, nmap)
|
||||
if not tp or tp in seeds:
|
||||
continue
|
||||
decayed = parent * (echo_recall.GRAPH_DECAY ** (hop + 1))
|
||||
if decayed > results.get(tp, (0.0, ""))[0]:
|
||||
results[tp] = (decayed, path)
|
||||
if tp not in seen:
|
||||
seen.add(tp)
|
||||
dq.append((tp, hop + 1))
|
||||
return sorted(results.items(), key=lambda kv: -kv[1][0])
|
||||
|
||||
|
||||
def m_expand_graph(ctx) -> dict:
|
||||
"""recall() graph layer: serial vs concurrent neighbourhood expansion. This is the
|
||||
metric that pins the real recall() bottleneck (the graph BFS, not BM25) and guards
|
||||
the read_many concurrency fix. Also asserts the concurrent path's ranking + scores
|
||||
match the serial reference exactly."""
|
||||
import fixtures
|
||||
ix = echo_recall.load_index()
|
||||
index = echo_index.load()
|
||||
nmap = echo_index.name_map(index)
|
||||
max_hops = echo_recall.MAX_HOPS
|
||||
# hub subjects stress expansion most; fall back to all subjects if none labelled hub
|
||||
subjects = [s for s in fixtures.SUBJECTS if "hub" in s[0]] or fixtures.SUBJECTS
|
||||
rows = []
|
||||
for label, query in subjects:
|
||||
hits = ix.score(query, limit=8)
|
||||
base = {p: s for p, s in hits}
|
||||
seeds = [p for p, _ in hits]
|
||||
if not seeds:
|
||||
continue
|
||||
echo_recall.expand_graph(seeds, nmap, base, max_hops) # warm
|
||||
serial_t = timed(lambda: _serial_expand(seeds, nmap, base, max_hops), n=1, warmup=0)
|
||||
conc_t = timed(lambda: echo_recall.expand_graph(seeds, nmap, base, max_hops),
|
||||
n=ctx.n, warmup=1)
|
||||
old = _serial_expand(seeds, nmap, base, max_hops)
|
||||
new = echo_recall.expand_graph(seeds, nmap, base, max_hops)
|
||||
ranking_ok = [p for p, _ in old] == [p for p, _ in new]
|
||||
scores_ok = ranking_ok and all(
|
||||
abs(old[i][1][0] - new[i][1][0]) < 1e-9 for i in range(len(old)))
|
||||
ratio = (serial_t["median_ms"] / conc_t["median_ms"]) if conc_t.get("median_ms") else None
|
||||
rows.append({"label": label, "query": query, "neighbours": len(new),
|
||||
"serial_ms": serial_t["median_ms"], "concurrent_ms": conc_t["median_ms"],
|
||||
"speedup": round(ratio, 2) if ratio else None,
|
||||
"ranking_identical": ranking_ok, "scores_identical": scores_ok})
|
||||
speedups = [r["speedup"] for r in rows if r["speedup"]]
|
||||
return {"subjects": rows,
|
||||
"min_speedup": round(min(speedups), 2) if speedups else None,
|
||||
"all_ranking_identical": all(r["ranking_identical"] for r in rows),
|
||||
"all_scores_identical": all(r["scores_identical"] for r in rows)}
|
||||
|
||||
|
||||
def _slug_from_entity(entity):
|
||||
if not entity:
|
||||
return None
|
||||
path = entity.get("path", "")
|
||||
return path.rsplit("/", 1)[-1].removesuffix(".md") if path else None
|
||||
|
||||
|
||||
def _cand_slug(c):
|
||||
if isinstance(c, dict):
|
||||
return c.get("slug") or _slug_from_entity(c)
|
||||
if isinstance(c, (list, tuple)) and c:
|
||||
return c[0]
|
||||
return str(c)
|
||||
|
||||
|
||||
def m_index(ctx) -> dict:
|
||||
"""§4.4 entity-index rebuild + link symmetrize timing (sweep.py, dry-run = read-only)."""
|
||||
return _time_script("sweep.py", [])
|
||||
|
||||
|
||||
def m_lint(ctx) -> dict:
|
||||
"""§4.5 full-vault lint timing (vault_lint.py) — regression-gate vs 1.1.0 baseline."""
|
||||
return _time_script("vault_lint.py", [], allow_exit={0, 1})
|
||||
|
||||
|
||||
def _time_script(name: str, extra: list[str], allow_exit=frozenset({0})) -> dict:
|
||||
script = SCRIPTS / name
|
||||
env = dict(os.environ, ECHO_KEY_LEGACY_OK="1")
|
||||
t0 = time.perf_counter()
|
||||
proc = subprocess.run([sys.executable, str(script), *extra],
|
||||
capture_output=True, text=True, env=env, timeout=120)
|
||||
elapsed = (time.perf_counter() - t0) * 1000.0
|
||||
return {"script": name, "elapsed_ms": round(elapsed, 1),
|
||||
"exit_code": proc.returncode,
|
||||
"ok": proc.returncode in allow_exit,
|
||||
"stderr_tail": proc.stderr.strip().splitlines()[-3:] if proc.stderr else []}
|
||||
|
||||
|
||||
def m_lock(ctx) -> dict:
|
||||
"""§4.6 advisory-lock timing: acquire-free, contended (held-fresh), release."""
|
||||
owner = f"bench-{ctx.run_id}"
|
||||
other = f"bench-other-{ctx.run_id}"
|
||||
|
||||
def acquire(o):
|
||||
return echo.cmd_lock(o, quiet=True)
|
||||
|
||||
def release(o):
|
||||
return echo.cmd_unlock(o, quiet=True)
|
||||
|
||||
release(owner); release(other) # clean slate
|
||||
t0 = time.perf_counter(); rc_free = acquire(owner); free_ms = (time.perf_counter() - t0) * 1000
|
||||
t0 = time.perf_counter(); rc_held = acquire(other); held_ms = (time.perf_counter() - t0) * 1000
|
||||
t0 = time.perf_counter(); release(owner); rel_ms = (time.perf_counter() - t0) * 1000
|
||||
return {"acquire_free_ms": round(free_ms, 2), "acquire_free_rc": rc_free,
|
||||
"contended_ms": round(held_ms, 2), "contended_rc_expected_75": rc_held,
|
||||
"release_ms": round(rel_ms, 2),
|
||||
"ok": rc_free == 0 and rc_held == 75}
|
||||
|
||||
|
||||
def m_queue(ctx) -> dict:
|
||||
"""§4.7 offline write-ahead queue: enqueue K writes against an unreachable base,
|
||||
then time the flush replay on 'reconnect'. Uses a bogus base so nothing real is
|
||||
written; restores the base afterward. No production writes occur."""
|
||||
import importlib
|
||||
import echo_queue
|
||||
k = min(ctx.k, 20)
|
||||
saved_base = echo.BASE
|
||||
enqueue_samples: list[float] = []
|
||||
try:
|
||||
for i in range(k):
|
||||
url = f"{echo.BASE}/vault/{fixtures_qpath(ctx.run_id, i)}"
|
||||
t0 = time.perf_counter()
|
||||
echo_queue.enqueue("PUT", url, b"x", {"Content-Type": "text/markdown"},
|
||||
idem_key=f"bench-{ctx.run_id}-{i}")
|
||||
enqueue_samples.append(time.perf_counter() - t0)
|
||||
pending_before = len(echo_queue.pending())
|
||||
# flush is left UNMEASURED-as-success here: with a real endpoint it would
|
||||
# try to write. We only assert enqueue worked and report depth; a true
|
||||
# flush-replay timing needs a sandbox vault (see README "queue caveat").
|
||||
return {"k": k, "enqueue": stats(enqueue_samples),
|
||||
"pending_after_enqueue": pending_before,
|
||||
"note": "flush replay intentionally not run against the live vault; "
|
||||
"enqueue path timed only. See README queue caveat."}
|
||||
finally:
|
||||
echo.BASE = saved_base
|
||||
importlib.reload(echo_queue)
|
||||
|
||||
|
||||
def fixtures_qpath(run_id: str, i: int) -> str:
|
||||
return f"_agent/_bench/{run_id}/queued-{i:04d}.md"
|
||||
|
||||
|
||||
def m_cache(ctx) -> dict:
|
||||
"""§4.8 read_many dedup check: a path list with duplicates must collapse to the
|
||||
unique set (the cache/dedup that stops a sweep re-fetching a link target)."""
|
||||
paths = all_notes()[:20] or ["_agent/echo-vault.md"]
|
||||
dupd = paths + paths + paths # 3x
|
||||
t0 = time.perf_counter()
|
||||
result = echo.read_many(dupd)
|
||||
elapsed = (time.perf_counter() - t0) * 1000
|
||||
return {"requested": len(dupd), "unique": len(set(dupd)),
|
||||
"returned": len(result), "deduped": len(result) == len(set(dupd)),
|
||||
"elapsed_ms": round(elapsed, 2)}
|
||||
|
||||
|
||||
def m_soak(ctx) -> dict:
|
||||
"""§4.10 stability: loop full-vault reads for N seconds; watch for errors/drift."""
|
||||
paths = all_notes()
|
||||
if not paths:
|
||||
raise echo.EchoError("soak: no notes enumerated")
|
||||
deadline = time.time() + ctx.soak_seconds
|
||||
durations: list[float] = []
|
||||
errors = 0
|
||||
while time.time() < deadline:
|
||||
t0 = time.perf_counter()
|
||||
try:
|
||||
res = echo.read_many(paths)
|
||||
if any(v is None for v in res.values()):
|
||||
errors += 1
|
||||
except Exception:
|
||||
errors += 1
|
||||
durations.append(time.perf_counter() - t0)
|
||||
s = stats(durations)
|
||||
# drift = late iterations slower than early ones (leak / pool exhaustion signal)
|
||||
half = len(durations) // 2 or 1
|
||||
early = statistics.fmean(durations[:half]) * 1000
|
||||
late = statistics.fmean(durations[half:]) * 1000
|
||||
return {"seconds": ctx.soak_seconds, "iterations": len(durations), "errors": errors,
|
||||
"per_pass": s, "early_mean_ms": round(early, 2), "late_mean_ms": round(late, 2),
|
||||
"drift_ratio_late_over_early": round(late / early, 2) if early else None}
|
||||
|
||||
|
||||
METRICS = {
|
||||
"read-full": m_read_full,
|
||||
"subject-pull": m_subject_pull,
|
||||
"bulk-get": m_bulk_get,
|
||||
"bulk-put": m_bulk_put,
|
||||
"bulk-append": m_bulk_append,
|
||||
"pool-warmup": m_pool_warmup,
|
||||
"worker-sweep": m_worker_sweep,
|
||||
"resolve": m_resolve,
|
||||
"expand-graph": m_expand_graph,
|
||||
"index": m_index,
|
||||
"lint": m_lint,
|
||||
"lock": m_lock,
|
||||
"queue": m_queue,
|
||||
"cache": m_cache,
|
||||
"soak": m_soak,
|
||||
}
|
||||
|
||||
# the read-only safe set run by `all` (no writes to the vault)
|
||||
READONLY = ["pool-warmup", "read-full", "worker-sweep", "subject-pull",
|
||||
"resolve", "expand-graph", "index", "lint", "cache"]
|
||||
WRITES = {"bulk-get", "bulk-put", "bulk-append", "lock"} # touch the _bench namespace
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# fixture seeding / cleanup
|
||||
# ---------------------------------------------------------------------------
|
||||
def _seed_notes(ctx, fixtures) -> list[str]:
|
||||
"""Ensure K bench notes exist; return their paths. Cached on ctx so bulk-get
|
||||
and bulk-put can share the seed within one run."""
|
||||
if getattr(ctx, "_seeded", None):
|
||||
return ctx._seeded
|
||||
paths = []
|
||||
for i in range(ctx.k):
|
||||
p = fixtures.note_path(ctx.run_id, i)
|
||||
put_verified(p, fixtures.note_body(ctx.run_id, i))
|
||||
paths.append(p)
|
||||
ctx._seeded = paths
|
||||
return paths
|
||||
|
||||
|
||||
def cleanup_namespace(run_id: str) -> int:
|
||||
"""Delete every file under _agent/_bench/<run-id>/. Returns count deleted."""
|
||||
prefix = f"_agent/_bench/{run_id}/"
|
||||
victims = [p for p in walk_vault(prefix)]
|
||||
for p in victims:
|
||||
delete_path(p)
|
||||
# remove the (now-empty) run dir marker if the API created one — best effort
|
||||
return len(victims)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# gating
|
||||
# ---------------------------------------------------------------------------
|
||||
def load_baselines() -> dict:
|
||||
if BASELINES.exists():
|
||||
return json.loads(BASELINES.read_text())
|
||||
return {}
|
||||
|
||||
|
||||
def gate(metric: str, result: dict, baselines: dict) -> list[dict]:
|
||||
"""Apply relative-ratio gates from baselines.json. Returns a list of checks."""
|
||||
rules = baselines.get(metric, {})
|
||||
checks = []
|
||||
for rule in rules.get("gates", []):
|
||||
path, op, threshold = rule["field"], rule["op"], rule["min"]
|
||||
val = _dig(result, path)
|
||||
if val is None:
|
||||
checks.append({"rule": rule, "value": None, "pass": None, "reason": "field absent"})
|
||||
continue
|
||||
ok = (val >= threshold) if op == ">=" else (val <= threshold)
|
||||
checks.append({"rule": rule, "value": val, "pass": ok})
|
||||
return checks
|
||||
|
||||
|
||||
def _dig(obj, dotted: str):
|
||||
cur = obj
|
||||
for part in dotted.split("."):
|
||||
if isinstance(cur, list):
|
||||
try:
|
||||
cur = cur[int(part)]
|
||||
continue
|
||||
except (ValueError, IndexError):
|
||||
return None
|
||||
if not isinstance(cur, dict) or part not in cur:
|
||||
return None
|
||||
cur = cur[part]
|
||||
return cur
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# main
|
||||
# ---------------------------------------------------------------------------
|
||||
class Ctx:
|
||||
pass
|
||||
|
||||
|
||||
def env_block(args) -> dict:
|
||||
commit = "unknown"
|
||||
try:
|
||||
commit = subprocess.run(["git", "rev-parse", "--short", "HEAD"],
|
||||
cwd=str(SCRIPTS), capture_output=True, text=True,
|
||||
timeout=10).stdout.strip() or "unknown"
|
||||
except Exception:
|
||||
pass
|
||||
return {
|
||||
"date": dt.datetime.now().isoformat(timespec="seconds"),
|
||||
"endpoint": echo.BASE,
|
||||
"commit": commit,
|
||||
"echo_workers": echo.MAX_WORKERS,
|
||||
"echo_timeout": echo.TIMEOUT,
|
||||
"python": sys.version.split()[0],
|
||||
"iterations": args.iterations,
|
||||
"count_k": args.count,
|
||||
}
|
||||
|
||||
|
||||
def human_table(name: str, result: dict, checks: list[dict]) -> str:
|
||||
lines = [f"\n=== {name} ==="]
|
||||
lines.append(json.dumps(result, indent=2))
|
||||
for c in checks:
|
||||
mark = "PASS" if c["pass"] else ("—" if c["pass"] is None else "FAIL")
|
||||
lines.append(f" [{mark}] {c['rule']['field']} {c['rule']['op']} {c['rule']['min']} (got {c['value']})")
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def main(argv=None) -> int:
|
||||
ap = argparse.ArgumentParser(description="ECHO performance test suite (manual/pre-release)")
|
||||
ap.add_argument("metric", nargs="?", help="metric name, or 'all' for the read-only set")
|
||||
ap.add_argument("--list", action="store_true", help="list metrics and exit")
|
||||
ap.add_argument("-n", "--iterations", type=int, default=None)
|
||||
ap.add_argument("-k", "--count", type=int, default=50)
|
||||
ap.add_argument("--workers", type=int, default=None)
|
||||
ap.add_argument("--warmup", type=int, default=2)
|
||||
ap.add_argument("--soak-seconds", type=int, default=60)
|
||||
ap.add_argument("--keep", action="store_true", help="do not delete the _bench namespace")
|
||||
ap.add_argument("--no-gate", action="store_true")
|
||||
ap.add_argument("--json-out", default=None)
|
||||
ap.add_argument("--quiet", action="store_true")
|
||||
args = ap.parse_args(argv)
|
||||
|
||||
if args.list or not args.metric:
|
||||
print("Metrics:", ", ".join(METRICS))
|
||||
print("Groups : all (read-only safe set:", ", ".join(READONLY) + ")")
|
||||
print("Writes to _bench namespace:", ", ".join(sorted(WRITES)))
|
||||
return 0 if args.list else 2
|
||||
|
||||
if args.workers is not None:
|
||||
echo.MAX_WORKERS = args.workers
|
||||
|
||||
selected = READONLY if args.metric == "all" else [args.metric]
|
||||
for m in selected:
|
||||
if m not in METRICS:
|
||||
print(f"unknown metric: {m}", file=sys.stderr)
|
||||
return 2
|
||||
|
||||
ctx = Ctx()
|
||||
ctx.run_id = dt.datetime.now().strftime("%Y%m%d-%H%M%S") + "-" + uuid.uuid4().hex[:6]
|
||||
ctx.k = args.count
|
||||
ctx.warmup = args.warmup
|
||||
ctx.soak_seconds = args.soak_seconds
|
||||
ctx._seeded = None
|
||||
|
||||
baselines = load_baselines()
|
||||
need_writes = any(m in WRITES for m in selected)
|
||||
lock_owner = f"bench-{ctx.run_id}"
|
||||
if need_writes:
|
||||
if echo.cmd_lock(lock_owner, quiet=True) == 75:
|
||||
print("bench: vault lock is held by another session — aborting write metrics.",
|
||||
file=sys.stderr)
|
||||
return 2
|
||||
|
||||
report = {"env": env_block(args), "run_id": ctx.run_id, "metrics": {}}
|
||||
any_fail = False
|
||||
try:
|
||||
for m in selected:
|
||||
ctx.n = args.iterations if args.iterations is not None else (5 if m in WRITES else 10)
|
||||
try:
|
||||
result = METRICS[m](ctx)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
result = {"error": str(exc)}
|
||||
any_fail = True
|
||||
checks = [] if args.no_gate else gate(m, result, baselines)
|
||||
if any(c["pass"] is False for c in checks):
|
||||
any_fail = True
|
||||
report["metrics"][m] = {"result": result, "gates": checks}
|
||||
if not args.quiet:
|
||||
print(human_table(m, result, checks))
|
||||
finally:
|
||||
if need_writes:
|
||||
echo.cmd_unlock(lock_owner, quiet=True)
|
||||
if need_writes and not args.keep:
|
||||
n = cleanup_namespace(ctx.run_id)
|
||||
report["cleanup"] = {"deleted": n}
|
||||
if not args.quiet:
|
||||
print(f"\nbench: cleaned up {n} files under _agent/_bench/{ctx.run_id}/")
|
||||
elif need_writes and args.keep:
|
||||
report["cleanup"] = {"deleted": 0, "kept": f"_agent/_bench/{ctx.run_id}/"}
|
||||
|
||||
RESULTS_DIR.mkdir(parents=True, exist_ok=True)
|
||||
out_path = Path(args.json_out) if args.json_out else \
|
||||
RESULTS_DIR / f"{report['env']['date'][:10]}-{report['env']['commit']}.json"
|
||||
out_path.write_text(json.dumps(report, indent=2))
|
||||
print(f"\nbench: results -> {out_path}")
|
||||
return 1 if any_fail else 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
Reference in New Issue
Block a user