0.2.2: судья отличает правки воркера от грязи, которая была до него #2
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "foreman-supervisor",
|
"name": "foreman-supervisor",
|
||||||
"version": "0.2.1",
|
"version": "0.2.2",
|
||||||
"description": "Semantic supervision for our own Claude agents. Wraps thruwire/foreman (MIT) — a runtime that watches a coding agent with a bounded observation (job, git status, diff, output tails, repository instructions), scores ten independent yes/no questions, and lets a deterministic Python policy decide continue / stop / retry / verify / finish / escalate. This plugin replaces both of upstream's paid dependencies: the worker is `claude -p` (Worker protocol, steer() returns False so the policy degrades to stop/retry, per upstream's own docs/workers.md), and the judge is the free-model gateway (free-worker.sh) instead of TypeSafe Jev, so a supervised run costs no paid tokens. Adds two things measured here rather than assumed: the observation carries commits since the run's base commit (our agents commit, which empties `git diff` exactly when the worker does the right thing), and warning dimensions stay muted until the factory has produced evidence (a small judge escalated a healthy run at 25 seconds, before the worker touched a file). Ships foreman-calibrate.py, which replays thresholds over the runs' own JSONL timelines — the calibration upstream says it has not done. v0.2.0 adds bin/foreman-attach.py: supervision of a background worker someone ELSE already started (worker.sh / `claude agents`), which upstream cannot do because its runtime launches the worker itself. The observation is rebuilt from traces the worker already leaves — its state.json, the commits in its worktree, the tail of its transcript — so nothing is injected into the worker. Measured here on a live worker: warning dimensions read stuck 0.80 / off-track 0.80 at 362 s and 0.05 / 0.10 at 766 s on the same run with no commit in between, which is why warmup and confirmation matter more than threshold tuning. Also records that upstream's policy gates STEER_WORKER on `codex_backend == \"app-server\"` — the name of another backend rather than whether the worker accepts direction — so a steerable third-party worker can only ever be killed; the adapter works around it by printing the steering message for the lead to deliver.",
|
"description": "Semantic supervision for our own Claude agents. Wraps thruwire/foreman (MIT) — a runtime that watches a coding agent with a bounded observation (job, git status, diff, output tails, repository instructions), scores ten independent yes/no questions, and lets a deterministic Python policy decide continue / stop / retry / verify / finish / escalate. This plugin replaces both of upstream's paid dependencies: the worker is `claude -p` (Worker protocol, steer() returns False so the policy degrades to stop/retry, per upstream's own docs/workers.md), and the judge is the free-model gateway (free-worker.sh) instead of TypeSafe Jev, so a supervised run costs no paid tokens. Adds two things measured here rather than assumed: the observation carries commits since the run's base commit (our agents commit, which empties `git diff` exactly when the worker does the right thing), and warning dimensions stay muted until the factory has produced evidence (a small judge escalated a healthy run at 25 seconds, before the worker touched a file). Ships foreman-calibrate.py, which replays thresholds over the runs' own JSONL timelines — the calibration upstream says it has not done. v0.2.0 adds bin/foreman-attach.py: supervision of a background worker someone ELSE already started (worker.sh / `claude agents`), which upstream cannot do because its runtime launches the worker itself. The observation is rebuilt from traces the worker already leaves — its state.json, the commits in its worktree, the tail of its transcript — so nothing is injected into the worker. Measured here on a live worker: warning dimensions read stuck 0.80 / off-track 0.80 at 362 s and 0.05 / 0.10 at 766 s on the same run with no commit in between, which is why warmup and confirmation matter more than threshold tuning. Also records that upstream's policy gates STEER_WORKER on `codex_backend == \"app-server\"` — the name of another backend rather than whether the worker accepts direction — so a steerable third-party worker can only ever be killed; the adapter works around it by printing the steering message for the lead to deliver.",
|
||||||
"author": {
|
"author": {
|
||||||
"name": "oleks",
|
"name": "oleks",
|
||||||
|
|||||||
+39
-6
@@ -321,6 +321,7 @@ class FreeModelForeman:
|
|||||||
self.free_tokens = 0
|
self.free_tokens = 0
|
||||||
self.failures = 0
|
self.failures = 0
|
||||||
self.base_commit: str | None = None
|
self.base_commit: str | None = None
|
||||||
|
self.baseline_dirty: frozenset[str] | None = None
|
||||||
|
|
||||||
async def _git(self, *args: str, limit: int = 2_000) -> str:
|
async def _git(self, *args: str, limit: int = 2_000) -> str:
|
||||||
try:
|
try:
|
||||||
@@ -339,6 +340,30 @@ class FreeModelForeman:
|
|||||||
return ""
|
return ""
|
||||||
return stdout.decode("utf-8", errors="replace")[-limit:]
|
return stdout.decode("utf-8", errors="replace")[-limit:]
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _status_paths(git_status: str) -> frozenset[str]:
|
||||||
|
paths = set()
|
||||||
|
for line in git_status.splitlines():
|
||||||
|
entry = line[3:].strip() if len(line) > 3 else ""
|
||||||
|
if entry:
|
||||||
|
paths.add(entry.split(" -> ")[-1])
|
||||||
|
return frozenset(paths)
|
||||||
|
|
||||||
|
def worker_touched(self, observation: FactoryObservation) -> list[str]:
|
||||||
|
"""Files this RUN changed — not files that were already dirty when it started.
|
||||||
|
|
||||||
|
Measured 2026-09-20 on the investigation dataset: the tree held 19 unrelated
|
||||||
|
uncommitted files (someone else's cache and edits). Counting them as the worker's
|
||||||
|
work both defeated the warm-up rule and made the judge call the run off track at 20
|
||||||
|
seconds, because the evidence it saw had nothing to do with the job.
|
||||||
|
"""
|
||||||
|
|
||||||
|
current = self._status_paths(observation.git_status)
|
||||||
|
if self.baseline_dirty is None:
|
||||||
|
self.baseline_dirty = current
|
||||||
|
return []
|
||||||
|
return sorted(current - self.baseline_dirty)
|
||||||
|
|
||||||
async def work_since_start(self) -> dict[str, str]:
|
async def work_since_start(self) -> dict[str, str]:
|
||||||
"""Upstream only watches the working tree; our workers commit, which empties `git diff`.
|
"""Upstream only watches the working tree; our workers commit, which empties `git diff`.
|
||||||
|
|
||||||
@@ -380,8 +405,12 @@ class FreeModelForeman:
|
|||||||
return ""
|
return ""
|
||||||
|
|
||||||
def build_prompt(
|
def build_prompt(
|
||||||
self, observation: FactoryObservation, committed: dict[str, str] | None = None
|
self,
|
||||||
|
observation: FactoryObservation,
|
||||||
|
committed: dict[str, str] | None = None,
|
||||||
|
touched: list[str] | None = None,
|
||||||
) -> str:
|
) -> str:
|
||||||
|
touched = touched if touched is not None else []
|
||||||
active = [
|
active = [
|
||||||
{
|
{
|
||||||
"worker_id": worker.get("worker_id"),
|
"worker_id": worker.get("worker_id"),
|
||||||
@@ -412,8 +441,9 @@ class FreeModelForeman:
|
|||||||
"active_workers": active,
|
"active_workers": active,
|
||||||
"worker_history": history,
|
"worker_history": history,
|
||||||
"worker_exit_status": observation.worker_exit_status,
|
"worker_exit_status": observation.worker_exit_status,
|
||||||
|
"changed_by_this_run": touched[:40],
|
||||||
|
"pre_existing_dirty_files_NOT_the_workers": sorted(self.baseline_dirty or [])[:20],
|
||||||
"git_status": observation.git_status,
|
"git_status": observation.git_status,
|
||||||
"changed_files": observation.changed_files[:40],
|
|
||||||
"git_diff_uncommitted": observation.git_diff,
|
"git_diff_uncommitted": observation.git_diff,
|
||||||
"work_committed_since_start": committed or {},
|
"work_committed_since_start": committed or {},
|
||||||
"latest_worker_output_tail": observation.latest_worker_output,
|
"latest_worker_output_tail": observation.latest_worker_output,
|
||||||
@@ -474,7 +504,8 @@ class FreeModelForeman:
|
|||||||
|
|
||||||
async def assess(self, observation: FactoryObservation) -> FactoryAssessment:
|
async def assess(self, observation: FactoryObservation) -> FactoryAssessment:
|
||||||
committed = await self.work_since_start()
|
committed = await self.work_since_start()
|
||||||
prompt = self.build_prompt(observation, committed)
|
touched = self.worker_touched(observation)
|
||||||
|
prompt = self.build_prompt(observation, committed, touched)
|
||||||
# A free model refuses more often than a paid endpoint; upstream issue #7 makes the same
|
# A free model refuses more often than a paid endpoint; upstream issue #7 makes the same
|
||||||
# point about transient Jev failures — one blip must not escalate a healthy run.
|
# point about transient Jev failures — one blip must not escalate a healthy run.
|
||||||
last_error: ForemanModelError | None = None
|
last_error: ForemanModelError | None = None
|
||||||
@@ -514,7 +545,9 @@ class FreeModelForeman:
|
|||||||
except ForemanModelError:
|
except ForemanModelError:
|
||||||
self.failures += 1
|
self.failures += 1
|
||||||
raise
|
raise
|
||||||
assessment, muted = self._mute_warnings_without_evidence(assessment, observation, committed)
|
assessment, muted = self._mute_warnings_without_evidence(
|
||||||
|
assessment, observation, committed, touched
|
||||||
|
)
|
||||||
assessment, held = self._require_confirmation(assessment, observation)
|
assessment, held = self._require_confirmation(assessment, observation)
|
||||||
if self.console is not None:
|
if self.console is not None:
|
||||||
suffix = " · warm-up: warnings muted (no evidence yet)" if muted else ""
|
suffix = " · warm-up: warnings muted (no evidence yet)" if muted else ""
|
||||||
@@ -530,6 +563,7 @@ class FreeModelForeman:
|
|||||||
assessment: FactoryAssessment,
|
assessment: FactoryAssessment,
|
||||||
observation: FactoryObservation,
|
observation: FactoryObservation,
|
||||||
committed: dict[str, str],
|
committed: dict[str, str],
|
||||||
|
touched: list[str] | None = None,
|
||||||
) -> tuple[FactoryAssessment, bool]:
|
) -> tuple[FactoryAssessment, bool]:
|
||||||
"""Measured 2026-09-20: a small judge escalated at 25 s, before the worker touched a file.
|
"""Measured 2026-09-20: a small judge escalated at 25 s, before the worker touched a file.
|
||||||
|
|
||||||
@@ -545,8 +579,7 @@ class FreeModelForeman:
|
|||||||
for worker in observation.worker_history
|
for worker in observation.worker_history
|
||||||
)
|
)
|
||||||
has_evidence = bool(
|
has_evidence = bool(
|
||||||
observation.changed_files
|
touched
|
||||||
or observation.git_status.strip()
|
|
||||||
or committed.get("commits_since_start", "").strip()
|
or committed.get("commits_since_start", "").strip()
|
||||||
or finished
|
or finished
|
||||||
)
|
)
|
||||||
|
|||||||
@@ -161,7 +161,8 @@ def test_judge_parses_prose_wrapped_and_nested_answers(tmp_path: Path) -> None:
|
|||||||
|
|
||||||
model._call_free_worker = fake_call # type: ignore[assignment]
|
model._call_free_worker = fake_call # type: ignore[assignment]
|
||||||
model.work_since_start = lambda: _completed({}) # type: ignore[assignment]
|
model.work_since_start = lambda: _completed({}) # type: ignore[assignment]
|
||||||
assessment = asyncio.run(model.assess(observation(changed_files=["a.py"])))
|
model.baseline_dirty = frozenset()
|
||||||
|
assessment = asyncio.run(model.assess(observation(git_status="?? a.py\n")))
|
||||||
assert assessment.implementation_complete == 0.5
|
assert assessment.implementation_complete == 0.5
|
||||||
assert assessment.needs_human == 0.5
|
assert assessment.needs_human == 0.5
|
||||||
assert model.free_tokens == 42
|
assert model.free_tokens == 42
|
||||||
@@ -181,7 +182,8 @@ def test_judge_retries_transient_failures_then_succeeds(tmp_path: Path) -> None:
|
|||||||
|
|
||||||
model._call_free_worker = flaky # type: ignore[assignment]
|
model._call_free_worker = flaky # type: ignore[assignment]
|
||||||
model.work_since_start = lambda: _completed({}) # type: ignore[assignment]
|
model.work_since_start = lambda: _completed({}) # type: ignore[assignment]
|
||||||
assessment = asyncio.run(model.assess(observation(changed_files=["a.py"])))
|
model.baseline_dirty = frozenset()
|
||||||
|
assessment = asyncio.run(model.assess(observation(git_status="?? a.py\n")))
|
||||||
assert calls["n"] == 3
|
assert calls["n"] == 3
|
||||||
assert assessment.worker_stuck == 0.1
|
assert assessment.worker_stuck == 0.1
|
||||||
assert model.failures == 2
|
assert model.failures == 2
|
||||||
@@ -212,19 +214,23 @@ def test_warnings_are_muted_until_the_factory_produced_evidence(tmp_path: Path)
|
|||||||
worker_history=[{"worker_id": "worker-1", "status": "running"}],
|
worker_history=[{"worker_id": "worker-1", "status": "running"}],
|
||||||
),
|
),
|
||||||
{},
|
{},
|
||||||
|
touched=[],
|
||||||
)
|
)
|
||||||
assert applied is True
|
assert applied is True
|
||||||
assert muted.needs_human == 0.1
|
assert muted.needs_human == 0.1
|
||||||
assert muted.worker_stuck == 0.0 and muted.agents_md_drift == 0.0
|
assert muted.worker_stuck == 0.0 and muted.agents_md_drift == 0.0
|
||||||
|
|
||||||
kept, applied_again = model._mute_warnings_without_evidence(
|
kept, applied_again = model._mute_warnings_without_evidence(
|
||||||
loud, observation(elapsed_factory_seconds=20.0, changed_files=["fetcher.py"]), {}
|
loud,
|
||||||
|
observation(elapsed_factory_seconds=20.0, changed_files=["fetcher.py"]),
|
||||||
|
{},
|
||||||
|
touched=["fetcher.py"],
|
||||||
)
|
)
|
||||||
assert applied_again is False
|
assert applied_again is False
|
||||||
assert kept.needs_human == 0.9
|
assert kept.needs_human == 0.9
|
||||||
|
|
||||||
late, applied_late = model._mute_warnings_without_evidence(
|
late, applied_late = model._mute_warnings_without_evidence(
|
||||||
loud, observation(elapsed_factory_seconds=600.0), {}
|
loud, observation(elapsed_factory_seconds=600.0), {}, touched=[]
|
||||||
)
|
)
|
||||||
assert applied_late is False
|
assert applied_late is False
|
||||||
assert late.agents_md_drift == 0.9
|
assert late.agents_md_drift == 0.9
|
||||||
@@ -277,3 +283,36 @@ def test_confirmation_is_off_unless_asked(tmp_path: Path) -> None:
|
|||||||
loud = _assessment(needs_human=0.95)
|
loud = _assessment(needs_human=0.95)
|
||||||
kept, held = model._require_confirmation(loud, observation(previous_assessment=None))
|
kept, held = model._require_confirmation(loud, observation(previous_assessment=None))
|
||||||
assert held == [] and kept.needs_human == 0.95
|
assert held == [] and kept.needs_human == 0.95
|
||||||
|
|
||||||
|
|
||||||
|
# ------------------------------------------------ pre-existing dirt vs worker's work
|
||||||
|
|
||||||
|
|
||||||
|
def test_pre_existing_dirty_files_are_not_the_workers_work(tmp_path: Path) -> None:
|
||||||
|
model = judge(tmp_path)
|
||||||
|
dirty = " M facts/other.md\n?? .cache/x.json\n"
|
||||||
|
|
||||||
|
first = model.worker_touched(observation(git_status=dirty))
|
||||||
|
assert first == [] # the first assessment only records the baseline
|
||||||
|
|
||||||
|
later = model.worker_touched(observation(git_status=dirty + "?? facts/new-fact.md\n"))
|
||||||
|
assert later == ["facts/new-fact.md"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_warmup_ignores_a_tree_that_was_already_dirty(tmp_path: Path) -> None:
|
||||||
|
model = judge(tmp_path, warmup_seconds=120.0)
|
||||||
|
loud = _assessment(needs_human=0.9, work_off_track=0.85)
|
||||||
|
dirty = observation(
|
||||||
|
elapsed_factory_seconds=20.0,
|
||||||
|
git_status=" M someone/else.md\n",
|
||||||
|
changed_files=["someone/else.md"],
|
||||||
|
worker_history=[{"worker_id": "worker-1", "status": "running"}],
|
||||||
|
)
|
||||||
|
|
||||||
|
muted, applied = model._mute_warnings_without_evidence(loud, dirty, {}, touched=[])
|
||||||
|
assert applied is True and muted.work_off_track == 0.0
|
||||||
|
|
||||||
|
acting, applied_now = model._mute_warnings_without_evidence(
|
||||||
|
loud, dirty, {}, touched=["facts/new-fact.md"]
|
||||||
|
)
|
||||||
|
assert applied_now is False and acting.work_off_track == 0.85
|
||||||
|
|||||||
Reference in New Issue
Block a user