Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
11 changes: 11 additions & 0 deletions docs/routing.md
Original file line number Diff line number Diff line change
Expand Up @@ -93,3 +93,14 @@ Configured extensions can add namespaced responsibility implementations and hier
groups. A matching group may expose cached child candidates such as projects, then route into the
responsibilities beneath every matching child or the single `best_match` child. Group bindings and
the complete route trace are persisted with the routing decision. See [Extensions](extensions.md).

## Completion stickiness

When `core.completion` observes the finish thresholds met, it records an evidence
fingerprint (the completion-relevant scores, the worker count, and the verification
state). A later assessment whose scores dip below the thresholds — the noisy idle
reassessments semantic scoring produces — may still finish the run, but only while
that fingerprint still holds: no new workers have run and no recorded score has
regressed beyond a 0.20 tolerance. If the evidence moves on, the authorization is
discarded instead of carried forward, so one transient high score can never
permanently authorize finishing.
74 changes: 74 additions & 0 deletions src/foreman/responsibilities/builtin.py
Original file line number Diff line number Diff line change
Expand Up @@ -229,10 +229,37 @@ def directives(self, state: FactoryState, result: ForemanResult) -> list[Directi
return [directive] if directive is not None else []


@dataclass(frozen=True, slots=True)
class _StickyCompletion:
"""Evidence fingerprint captured when completion thresholds were met.

A later assessment may still finish on the strength of this record, but only
while the underlying evidence still holds: no new workers have run since it
was captured and no completion-relevant score has regressed beyond tolerance.
This keeps the "sticky finish" behavior for noisy idle reassessments without
letting one transient high score permanently authorize finishing after the
evidence changes.
"""

iteration: int
worker_count: int
ready_to_finish: float
requirements_satisfied: float
tests_sufficient: float
needs_verification: float
verification_completed: bool


# Maximum regression of a recorded completion score (or rise of
# needs_verification) before a sticky finish authorization is discarded.
_STICKY_EVIDENCE_TOLERANCE = 0.20


@dataclass(slots=True)
class CompletionResponsibility(_CheckConfiguredResponsibility):
config: FactoryConfig
id: str = COMPLETION
_sticky: _StickyCompletion | None = field(default=None, repr=False)
required_check_ids: ClassVar[frozenset[str]] = frozenset(
{"implementation_complete", "requirements_satisfied", "ready_to_finish"}
)
Expand Down Expand Up @@ -274,6 +301,7 @@ def directives(self, state: FactoryState, result: ForemanResult) -> list[Directi
VERIFICATION, "needs_verification"
) < self.minimum(VERIFICATION, "needs_verification")
if finish_ready and verification_resolved:
self._capture_sticky(state, result)
return [
_directive(
state,
Expand All @@ -283,6 +311,22 @@ def directives(self, state: FactoryState, result: ForemanResult) -> list[Directi
priority=700,
)
]
if (
self._sticky is not None
and verification_resolved
and self._sticky_holds(state, result)
):
return [
_directive(
state,
responsibility_id=self.id,
action=InterventionType.FINISH,
reason="completion thresholds met earlier and supporting evidence still holds",
priority=700,
)
]
# The evidence moved on: never carry a stale finish authorization forward.
self._sticky = None
return [
_directive(
state,
Expand All @@ -293,6 +337,36 @@ def directives(self, state: FactoryState, result: ForemanResult) -> list[Directi
)
]

def _capture_sticky(self, state: FactoryState, result: ForemanResult) -> None:
self._sticky = _StickyCompletion(
iteration=max(1, state.iteration),
worker_count=len(state.workers),
ready_to_finish=result.probability(self.id, "ready_to_finish"),
requirements_satisfied=result.probability(self.id, "requirements_satisfied"),
tests_sufficient=result.probability(VERIFICATION, "tests_sufficient"),
needs_verification=result.probability(VERIFICATION, "needs_verification"),
verification_completed=state.verification_completed,
)

def _sticky_holds(self, state: FactoryState, result: ForemanResult) -> bool:
record = self._sticky
if record is None:
return False
if len(state.workers) != record.worker_count:
return False
if record.verification_completed and not state.verification_completed:
return False
for responsibility_id, check_id, recorded in (
(self.id, "ready_to_finish", record.ready_to_finish),
(self.id, "requirements_satisfied", record.requirements_satisfied),
(VERIFICATION, "tests_sufficient", record.tests_sufficient),
):
current = result.probability(responsibility_id, check_id)
if current < recorded - _STICKY_EVIDENCE_TOLERANCE:
return False
needs_verification = result.probability(VERIFICATION, "needs_verification")
return needs_verification <= record.needs_verification + _STICKY_EVIDENCE_TOLERANCE


@dataclass(slots=True)
class VerificationResponsibility(_CheckConfiguredResponsibility):
Expand Down
160 changes: 160 additions & 0 deletions tests/test_completion_sticky.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,160 @@
"""Sticky completion with safe semantics.

Once completion thresholds are met, later noisy or idle reassessments may still
finish the run — but the sticky authorization is an evidence fingerprint, not a
latch. It stays valid only while no new workers have run and no
completion-relevant score has regressed beyond tolerance. One transient high
score can therefore never permanently authorize finishing after the evidence
changes.
"""

from __future__ import annotations

from foreman.config import FactoryConfig
from foreman.models import (
FactoryState,
ForemanResult,
InterventionType,
WorkerRecord,
WorkerStatus,
WorkerType,
)
from foreman.responsibilities import COMPLETION, CompletionResponsibility, builtin_registry


def _completion() -> CompletionResponsibility:
registry = builtin_registry(FactoryConfig())
responsibility = next(r for r in registry.responsibilities if r.id == COMPLETION)
assert isinstance(responsibility, CompletionResponsibility)
return responsibility


def _result(
*, ready: float, req: float, tests: float, verify: float
) -> ForemanResult:
return ForemanResult(
checks={
"core.completion": {
"implementation_complete": 0.8,
"requirements_satisfied": req,
"ready_to_finish": ready,
},
"core.verification": {
"tests_sufficient": tests,
"needs_verification": verify,
},
}
)


def _state() -> FactoryState:
return FactoryState(run_id="r1", job="job", repository="repo")


def _finish(responsibility: CompletionResponsibility, state: FactoryState) -> None:
directives = responsibility.directives(
state, _result(ready=0.8, req=0.8, tests=0.8, verify=0.1)
)
assert len(directives) == 1
assert directives[0].action is InterventionType.FINISH


def test_finish_fires_when_thresholds_met() -> None:
responsibility = _completion()
directives = responsibility.directives(
_state(), _result(ready=0.8, req=0.8, tests=0.8, verify=0.1)
)

assert len(directives) == 1
assert directives[0].action is InterventionType.FINISH
assert directives[0].reason == "completion thresholds satisfied"


def test_sticky_finish_survives_noisy_idle_reassessment() -> None:
"""Scores dip below thresholds but stay within tolerance of the recorded
evidence: the run still finishes instead of churning new workers."""
responsibility = _completion()
state = _state()
_finish(responsibility, state)

directives = responsibility.directives(
state, _result(ready=0.70, req=0.72, tests=0.70, verify=0.15)
)

assert len(directives) == 1
assert directives[0].action is InterventionType.FINISH
assert "supporting evidence still holds" in directives[0].reason


def test_transient_spike_does_not_permanently_authorize_finish() -> None:
"""One high assessment followed by genuinely regressed evidence: the sticky
authorization is discarded and the run goes back to work, not finish."""
responsibility = _completion()
state = _state()
_finish(responsibility, state)

regressed = _result(ready=0.3, req=0.4, tests=0.5, verify=0.2)
directives = responsibility.directives(state, regressed)
assert len(directives) == 1
assert directives[0].action is InterventionType.START_WORKER

# The stale authorization was cleared, not latched: it stays cleared.
directives = responsibility.directives(state, regressed)
assert directives[0].action is InterventionType.START_WORKER


def test_new_worker_invalidates_sticky() -> None:
"""Fresh worker activity after the thresholds were met is new evidence: the
sticky authorization must not survive it."""
responsibility = _completion()
state = _state()
_finish(responsibility, state)

state.workers.append(
WorkerRecord(
worker_id="w1",
worker_type=WorkerType.CODING,
mission="m",
status=WorkerStatus.COMPLETED,
)
)
directives = responsibility.directives(
state, _result(ready=0.70, req=0.72, tests=0.70, verify=0.15)
)
assert directives[0].action is InterventionType.START_WORKER


def test_sticky_still_requires_verification_resolved() -> None:
"""If verification concerns resurface, the sticky record alone cannot finish
the run."""
responsibility = _completion()
state = _state()
_finish(responsibility, state)

directives = responsibility.directives(
state, _result(ready=0.78, req=0.78, tests=0.78, verify=0.8)
)
assert directives[0].action is InterventionType.START_WORKER


def test_sticky_refreshes_while_thresholds_hold() -> None:
"""Assessments that still meet thresholds record fresh evidence rather than
leaning on the earlier record."""
responsibility = _completion()
state = _state()
_finish(responsibility, state)

directives = responsibility.directives(
state, _result(ready=0.9, req=0.9, tests=0.9, verify=0.05)
)
assert directives[0].action is InterventionType.FINISH
assert directives[0].reason == "completion thresholds satisfied"


def test_no_sticky_without_prior_thresholds() -> None:
responsibility = _completion()
state = _state()
low = _result(ready=0.5, req=0.5, tests=0.5, verify=0.1)

assert responsibility.directives(state, low)[0].action is InterventionType.START_WORKER
assert responsibility.directives(state, low)[0].action is InterventionType.START_WORKER