Files
claude-plugin-inference-arb…/bin/boundary-classify
T
Oleks a218a31055
ci/woodpecker/push/test Pipeline is pending
Dedup MIN_SHARE_OF_SPEND: boundary-classify imports it from ia_store
bin/boundary-classify defined its own copy of the spec §5.2.7 filing
threshold that only happened to match ia_store.py's copy, which
bin/stability-classify's FR-4.5 stability gate actually reads. Tuning
either constant independently would silently desynchronize the gate
from the classifier it's meant to gate. ia_store.py is now the single
definition; boundary-classify imports it.

MIN_INVOCATIONS/MIN_ATTRIBUTED_TURNS are left as-is: stability-classify
does not import or use them, so their duplication doesn't create the
same fail-silently risk. Centralizing all thresholds via
references/signals-catalog.md remains a separate, larger follow-up per
the issue's "Related / proper fix" note.

Fixes kotkan/claude-plugin-inference-arbitrage#14
2026-07-30 05:30:48 +03:00

216 lines
8.8 KiB
Python
Executable File

#!/usr/bin/env python3
"""Applies rubric §6's confidence table and the FR-4.2 filing gate to candidates
whose determinism-test outcomes and falsifiability triple were judged by the
analyst agent.
The judgments are inference (T2 and T4 are irreducibly semantic). Turning those
judgments into a confidence grade and a file/don't-file verdict is a lookup
table, so it lives here rather than in agent prose — position 2 of the rubric,
applied to the plugin's own machinery (FR-8.1).
"""
import argparse
import json
import sys
from ia_store import MIN_SHARE_OF_SPEND
RUBRIC_VERSION = "1.0.0"
OFFLOADABLE_POSITIONS = {"pure-script", "llm-over-script-digest"}
POSITIONS = OFFLOADABLE_POSITIONS | {"script-with-compiled-judgment", "pure-inference"}
TESTS = ("T1", "T2", "T3", "T4", "T5")
# spec §5.2.7 filing threshold — single definition lives in ia_store.py
# (kotkan/claude-plugin-inference-arbitrage#14); FR-4.5's stability gate
# (bin/stability-classify) imports the same constant, so tuning it here
# can no longer desynchronize the gate.
# Below these, the dynamic pass is too thin to carry the share-of-spend gate on
# its own. Derived from the Phase 3 finding that a skill invoked once inside a
# multi-topic session produces a share that is arithmetically real but not
# evidence of a recurring procedure.
MIN_INVOCATIONS = 3
MIN_ATTRIBUTED_TURNS = 30
def triple_state(f):
"""rubric §5 — which of the three elements are producible."""
examples = f.get("examples") or []
return {
"signature": bool(f.get("signature")),
"examples": len(examples) >= 3 and f.get("edge_case_index") is not None,
"overrule_case": bool(f.get("overrule_case")),
}
def grade(cand):
tests = cand.get("determinism_tests", {})
position = cand.get("position")
if position not in POSITIONS:
raise SystemExit(f"unknown position {position!r} for {cand.get('candidate_id')}")
missing = [t for t in TESTS if t not in tests]
if missing:
raise SystemExit(f"{cand.get('candidate_id')}: missing tests {missing}")
f = cand.get("falsifiability") or {}
triple = triple_state(f)
reasons = []
all_pass = all(tests[t] is True for t in TESTS)
oracle_doubt = tests["T2"] is not True or tests["T4"] is not True
if oracle_doubt:
confidence = "low"
reasons.append("T2 or T4 in doubt — the oracle or the semantic gap is unresolved")
elif not all_pass:
confidence = "medium"
reasons.append(
"T2 and T4 hold but "
+ ", ".join(t for t in TESTS if tests[t] is not True)
+ " fails — decomposable, the cut is not yet located"
)
elif position not in OFFLOADABLE_POSITIONS:
confidence = "medium"
if position == "script-with-compiled-judgment" and not cand.get("staleness_plan"):
reasons.append("position 2 without a staleness plan — a deferred bug, not a saving")
else:
reasons.append(f"position {position} is not a whole-step offload")
elif not all(triple.values()):
confidence = "medium"
reasons.append(
"falsifiability triple incomplete: missing "
+ ", ".join(k for k, v in triple.items() if not v)
)
elif not cand.get("escalation_path"):
confidence = "medium"
reasons.append("no escalation path defined (positioning question P3)")
elif not cand.get("digest_schema") and position == "llm-over-script-digest":
confidence = "medium"
reasons.append("digest schema (P1) not writable — the boundary is wished for, not found")
else:
confidence = "high"
reasons.append("all five determinism tests pass; triple complete; escalation path defined")
if position == "script-with-compiled-judgment" and not cand.get("staleness_plan"):
note = "position 2 without a staleness plan"
if note not in " ".join(reasons):
reasons.append(note)
return confidence, triple, reasons
def strength(m):
if not m:
return "unmeasured"
if m.get("invocations", 0) >= MIN_INVOCATIONS and m.get("attributed_turns", 0) >= MIN_ATTRIBUTED_TURNS:
return "measured"
return "thin"
def classify(doc):
spend = doc.get("audited_spend_weighted_tokens") or 0
out = []
for cand in doc.get("candidates", []):
confidence, triple, reasons = grade(cand)
m = cand.get("measurement") or {}
share = round(m.get("offload_value", 0) / spend, 4) if spend else 0.0
ev = strength(m)
if cand.get("category") == "hook-prose-drift":
verdict = "drift-note"
reasons.append(
"hook/prose drift — a restated configuration contract, not an un-scripted "
"algorithm. The fix is to point at the hook file, not to write a script."
)
elif cand.get("position") == "pure-inference":
verdict = "no-offload"
reasons.append(
"position 4 — correctly done by inference. Reported as a finding, not a question."
)
elif not triple["overrule_case"]:
verdict = "boundary-question"
reasons.append(
"HARD GATE (FR-4.2): no overrule case — downgraded to a boundary question, never filed"
)
elif confidence != "high":
verdict = "boundary-question"
reasons.append(f"confidence {confidence} — reported, not filed")
elif ev == "measured":
# The window carries enough invocations for share-of-spend to mean
# something, so spec §5.2.7's threshold applies as written.
if share < MIN_SHARE_OF_SPEND:
verdict = "boundary-question"
reasons.append(
f"offload_value is {share:.2%} of audited spend, below the "
f"{MIN_SHARE_OF_SPEND:.0%} filing threshold"
)
else:
verdict = "file"
reasons.append(f"measured at {share:.2%} of audited spend over {m['invocations']} invocations")
else:
# Too few invocations for the share to be evidence of anything
# (Phase 3: a skill invoked once inside a multi-topic session
# produces a share that is arithmetically real but not a
# measurement). The correctness case is complete regardless, and
# FR-2.4 requires a plugin with no history to remain auditable, so
# this files on the static case with the value claim marked unproven.
verdict = "file"
reasons.append(
f"filed on the static case — evidence is {ev} "
f"({m.get('invocations', 0)} invocations, {m.get('attributed_turns', 0)} attributed turns), "
"so the value claim is unproven and must be stated as such in the issue body"
)
out.append(
{
"candidate_id": cand.get("candidate_id"),
"skill": cand.get("skill"),
"category": cand.get("category", "usage"),
"position": cand.get("position"),
"determinism_tests": cand.get("determinism_tests"),
"falsifiability_triple": triple,
"boundary_confidence": confidence,
"measurement_strength": ev,
"share_of_audited_spend": share,
"offload_value": m.get("offload_value", 0),
"verdict": verdict,
"reasons": reasons,
}
)
out.sort(key=lambda c: c["offload_value"], reverse=True)
highs = [c for c in out if c["boundary_confidence"] == "high"]
return {
"rubric_version": RUBRIC_VERSION,
"target": doc.get("target"),
"summary": {
"candidates": len(out),
"high_confidence": len(highs),
"to_file": sum(1 for c in out if c["verdict"] == "file"),
"boundary_questions": sum(1 for c in out if c["verdict"] == "boundary-question"),
"no_offload": sum(1 for c in out if c["verdict"] == "no-offload"),
"drift_notes": sum(1 for c in out if c["verdict"] == "drift-note"),
"conclusion": "nothing to offload here" if not highs else f"{len(highs)} high-confidence candidate(s)",
},
"candidates": out,
}
def main():
ap = argparse.ArgumentParser(prog="boundary-classify")
ap.add_argument("judgments", nargs="?", default="-", help="candidate judgments JSON, or - for stdin")
ap.add_argument("--json", action="store_true", help="compact single-line JSON")
args = ap.parse_args()
src = sys.stdin if args.judgments == "-" else open(args.judgments)
with src as fh:
doc = json.load(fh)
result = classify(doc)
json.dump(result, sys.stdout, indent=None if args.json else 2)
sys.stdout.write("\n")
if __name__ == "__main__":
main()