a7f3044c94
Add skill/SKILL.md (+ skill/evaluate-extension-usage.py, referenced by the skill via ./) so the canonical 'how to use fork/recall/ssh-controlmaster' skill lives next to the extensions it documents — the single source of truth. Motivation: the global AGENTS.md (pi-toolkit) tells every pi session to read ~/.agents/skills/pi-extensions/SKILL.md at session start to fix fork/recall under-utilisation, but that skill previously lived ONLY in the private skillset repo. In any environment without the skillset mounted (e.g. a pi-devbox container started without it) the pointer dangled. Co-locating the skill here gives a public, package-owned source that downstreams can vendor. install.sh is intentionally unchanged: skill deployment on a normal workstation stays the skillset repo's responsibility (no double-deploy).
118 lines
4.3 KiB
Python
Executable File
118 lines
4.3 KiB
Python
Executable File
#!/usr/bin/env python3
|
|
"""Evaluate pi-fork / pi-observational-memory usage from pi session transcripts.
|
|
|
|
Mines pi's session .jsonl transcripts and reports:
|
|
- per-tool call counts (highlighting `fork` and `recall`)
|
|
- per-session fork/recall breakdown
|
|
- obsmem passive activity: compaction events, observations carried,
|
|
relevance-tier distribution, tokensBefore
|
|
|
|
Works on any machine. Point it at one or more session roots; by default it
|
|
scans ~/.pi/agent/sessions (the standard pi location, host or container).
|
|
|
|
Usage:
|
|
./evaluate-extension-usage.py # ~/.pi/agent/sessions
|
|
./evaluate-extension-usage.py /path/to/sessions ... # explicit roots
|
|
./evaluate-extension-usage.py --host HOST /path ... # label a root (for combined host+container runs)
|
|
|
|
For a true host+container picture, run once per machine (or copy each
|
|
machine's ~/.pi/agent/sessions here) and pass all roots together.
|
|
"""
|
|
import json, sys, os, glob, re, collections, argparse
|
|
|
|
TIER_RE = re.compile(r'\[(low|medium|high|critical)\]')
|
|
OBS_LINE_RE = re.compile(r'^\[[0-9a-f]{12}\] ', re.M)
|
|
|
|
|
|
def walk_tools(x, counter):
|
|
if isinstance(x, dict):
|
|
tn = x.get("toolName")
|
|
if tn:
|
|
counter[tn] += 1
|
|
for v in x.values():
|
|
walk_tools(v, counter)
|
|
elif isinstance(x, list):
|
|
for v in x:
|
|
walk_tools(v, counter)
|
|
|
|
|
|
def analyze(roots):
|
|
files = []
|
|
for r in roots:
|
|
if os.path.isfile(r) and r.endswith(".jsonl"):
|
|
files.append(r)
|
|
else:
|
|
files += glob.glob(os.path.join(r, "**", "*.jsonl"), recursive=True)
|
|
files = sorted(set(files))
|
|
|
|
tool_total = collections.Counter()
|
|
per_session = []
|
|
compactions = []
|
|
for f in files:
|
|
tc = collections.Counter()
|
|
with open(f, errors="ignore") as fh:
|
|
for ln in fh:
|
|
ln = ln.strip()
|
|
if not ln:
|
|
continue
|
|
try:
|
|
o = json.loads(ln)
|
|
except Exception:
|
|
continue
|
|
walk_tools(o, tc)
|
|
if o.get("type") == "compaction":
|
|
s = o.get("summary", "") or ""
|
|
compactions.append({
|
|
"file": os.path.basename(f),
|
|
"tokensBefore": o.get("tokensBefore"),
|
|
"observations": len(OBS_LINE_RE.findall(s)),
|
|
"tiers": dict(collections.Counter(TIER_RE.findall(s))),
|
|
})
|
|
tool_total.update(tc)
|
|
per_session.append((os.path.basename(f)[:10], tc.get("fork", 0),
|
|
tc.get("recall", 0), sum(tc.values())))
|
|
return files, tool_total, per_session, compactions
|
|
|
|
|
|
def main():
|
|
ap = argparse.ArgumentParser()
|
|
ap.add_argument("roots", nargs="*",
|
|
default=[os.path.expanduser("~/.pi/agent/sessions")])
|
|
args = ap.parse_args()
|
|
|
|
files, tool_total, per_session, comp = analyze(args.roots)
|
|
if not files:
|
|
print("No .jsonl transcripts found under:", args.roots, file=sys.stderr)
|
|
sys.exit(1)
|
|
|
|
print(f"=== {len(files)} transcripts under {args.roots} ===\n")
|
|
print("Tool call totals:")
|
|
for t, c in tool_total.most_common():
|
|
mark = " <== pi-fork" if t == "fork" else (" <== obsmem recall" if t == "recall" else "")
|
|
print(f" {c:6d} {t}{mark}")
|
|
|
|
fk = tool_total["fork"]; rc = tool_total["recall"]
|
|
fk_sess = sum(1 for p in per_session if p[1])
|
|
rc_sess = sum(1 for p in per_session if p[2])
|
|
print(f"\npi-fork: {fk} calls across {fk_sess} sessions")
|
|
print(f"recall: {rc} calls across {rc_sess} sessions"
|
|
+ (" (!) zero recall over the window — see SKILL.md calibration note" if rc == 0 else ""))
|
|
|
|
if comp:
|
|
tot_obs = sum(c["observations"] for c in comp)
|
|
tb = [c["tokensBefore"] for c in comp if c["tokensBefore"]]
|
|
print(f"\nobsmem passive: {len(comp)} compactions, {tot_obs} observations carried"
|
|
+ (f", avg tokensBefore {sum(tb)//len(tb):,}" if tb else ""))
|
|
agg = collections.Counter()
|
|
for c in comp:
|
|
agg.update(c["tiers"])
|
|
if agg:
|
|
print(" relevance tiers:", dict(agg))
|
|
else:
|
|
print("\nobsmem passive: no compaction events found "
|
|
"(short sessions, or obsmem not active on these transcripts)")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|