Coverage for src/keel/workcreation.py: 100%
88 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-02 20:26 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-02 20:26 +0000
1"""Deterministic policy for signal-driven work creation."""
3from __future__ import annotations
5import re
6from dataclasses import dataclass
7from typing import Any
9SCHEMA_VERSION = "keel.work-creation.v1"
10DEFAULT_MIN_OCCURRENCES = 2
11DEFAULT_MIN_CONFIDENCE = 0.6
12DEFAULT_MAX_CREATIONS = 5
13DEFAULT_NEAR_TEXT_SIMILARITY = 0.6
15DECISIONS = (
16 "create",
17 "suppress-transient",
18 "suppress-duplicate",
19 "limit-reached",
20)
22_TOKEN_RE = re.compile(r"[a-z0-9]+")
25@dataclass(frozen=True)
26class WorkDecision:
27 """One deterministic work-creation decision."""
29 candidate_id: str
30 decision: str
31 reason: str
32 title: str
33 duplicate_of: int | None = None
35 def as_dict(self) -> dict[str, Any]:
36 result: dict[str, Any] = {
37 "candidate_id": self.candidate_id,
38 "decision": self.decision,
39 "reason": self.reason,
40 "title": self.title,
41 "creates_issue": self.decision == "create",
42 }
43 if self.duplicate_of is not None:
44 result["duplicate_of"] = self.duplicate_of
45 return result
48def contract_as_dict(*, near_text_similarity: float | None = None) -> dict[str, Any]:
49 """Return the shared work-creation policy contract.
51 ``near_text_similarity`` is the project's resolved
52 ``policy_pack.scan.near_text_similarity``. It has to be passed in rather than
53 defaulted here, because the caller embeds this dict *beside* a `dedupe` block that
54 already honours the knob — hardcoding the default shipped two different thresholds
55 under near-identical keys in one contract, agreeing only as long as the project set
56 the knob to exactly the built-in default (#633).
57 """
58 return {
59 "schema_version": SCHEMA_VERSION,
60 "consumer_neutral": True,
61 "deterministic": True,
62 "stdlib_only": True,
63 "source": "signal-driven commands",
64 "decisions": list(DECISIONS),
65 "transient_filter": {
66 "default_min_occurrences": DEFAULT_MIN_OCCURRENCES,
67 "default_min_confidence": DEFAULT_MIN_CONFIDENCE,
68 "transient_outcome": "suppress-transient",
69 },
70 "dedupe": {
71 "against": "open work",
72 "keys": ["dedupe_key", "normalized_title", "near_text"],
73 "near_text_similarity": _confidence(near_text_similarity, DEFAULT_NEAR_TEXT_SIMILARITY),
74 "duplicate_outcome": "suppress-duplicate",
75 },
76 "cycle_limit": {
77 "default_max_creations": DEFAULT_MAX_CREATIONS,
78 "limit_outcome": "limit-reached",
79 },
80 "consumers": [
81 "regression",
82 "review-all-day",
83 "coverage",
84 "deps-audit",
85 "flake-audit",
86 ],
87 }
90def evaluate_candidates(
91 candidates: list[dict[str, Any]] | tuple[dict[str, Any], ...],
92 existing_work: list[dict[str, Any]] | tuple[dict[str, Any], ...] = (),
93 *,
94 min_occurrences: int = DEFAULT_MIN_OCCURRENCES,
95 min_confidence: float = DEFAULT_MIN_CONFIDENCE,
96 max_creations: int = DEFAULT_MAX_CREATIONS,
97 near_text_similarity: float = DEFAULT_NEAR_TEXT_SIMILARITY,
98) -> dict[str, Any]:
99 """Evaluate candidate signals against transient, dedupe, and cycle-limit policy."""
100 policy = {
101 "min_occurrences": _positive_int(min_occurrences, DEFAULT_MIN_OCCURRENCES),
102 "min_confidence": _confidence(min_confidence, DEFAULT_MIN_CONFIDENCE),
103 "max_creations": _positive_int(max_creations, DEFAULT_MAX_CREATIONS),
104 "near_text_similarity": _confidence(
105 near_text_similarity,
106 DEFAULT_NEAR_TEXT_SIMILARITY,
107 ),
108 }
109 normalized_existing = [
110 _normalize_existing(item)
111 for item in existing_work
112 if isinstance(item, dict) and _is_open(item)
113 ]
114 created = 0
115 decisions: list[WorkDecision] = []
116 for index, raw in enumerate(candidates, start=1):
117 if not isinstance(raw, dict):
118 continue
119 candidate = _normalize_candidate(raw, index)
120 duplicate = _find_duplicate(candidate, normalized_existing, policy["near_text_similarity"])
121 if _is_transient(candidate, policy):
122 decisions.append(
123 WorkDecision(
124 candidate["id"],
125 "suppress-transient",
126 "transient-signal",
127 candidate["title"],
128 )
129 )
130 elif duplicate is not None:
131 decisions.append(
132 WorkDecision(
133 candidate["id"],
134 "suppress-duplicate",
135 "open-work-duplicate",
136 candidate["title"],
137 duplicate_of=duplicate["number"],
138 )
139 )
140 elif created >= policy["max_creations"]:
141 decisions.append(
142 WorkDecision(
143 candidate["id"],
144 "limit-reached",
145 "per-cycle-limit-reached",
146 candidate["title"],
147 )
148 )
149 else:
150 created += 1
151 decisions.append(
152 WorkDecision(
153 candidate["id"],
154 "create",
155 "eligible",
156 candidate["title"],
157 )
158 )
159 normalized_existing.append(_created_as_existing(candidate, created))
160 decision_dicts = [decision.as_dict() for decision in decisions]
161 return {
162 "schema_version": SCHEMA_VERSION,
163 "status": "pass",
164 "policy": policy,
165 "summary": {
166 "candidates": len([item for item in candidates if isinstance(item, dict)]),
167 "create": _count(decision_dicts, "create"),
168 "suppress_transient": _count(decision_dicts, "suppress-transient"),
169 "suppress_duplicate": _count(decision_dicts, "suppress-duplicate"),
170 "limit_reached": _count(decision_dicts, "limit-reached"),
171 },
172 "decisions": decision_dicts,
173 }
176def _normalize_candidate(raw: dict[str, Any], index: int) -> dict[str, Any]:
177 title = _string(raw.get("title")) or f"candidate-{index}"
178 body = _string(raw.get("body"))
179 return {
180 "id": _string(raw.get("id")) or f"candidate-{index}",
181 "title": title,
182 "body": body,
183 "dedupe_key": _string(raw.get("dedupe_key")),
184 "occurrences": _positive_int(raw.get("occurrences"), 1),
185 "confidence": _confidence(raw.get("confidence"), 1.0),
186 "normalized_title": _normalize_text(title),
187 "tokens": _tokens(f"{title} {body}"),
188 }
191def _normalize_existing(raw: dict[str, Any]) -> dict[str, Any]:
192 title = _string(raw.get("title"))
193 body = _string(raw.get("body"))
194 return {
195 "number": raw.get("number") if isinstance(raw.get("number"), int) else None,
196 "title": title,
197 "body": body,
198 "dedupe_key": _string(raw.get("dedupe_key")),
199 "normalized_title": _normalize_text(title),
200 "tokens": _tokens(f"{title} {body}"),
201 }
204def _created_as_existing(candidate: dict[str, Any], created_index: int) -> dict[str, Any]:
205 return {
206 "number": -created_index,
207 "title": candidate["title"],
208 "body": candidate["body"],
209 "dedupe_key": candidate["dedupe_key"],
210 "normalized_title": candidate["normalized_title"],
211 "tokens": candidate["tokens"],
212 }
215def _is_transient(candidate: dict[str, Any], policy: dict[str, Any]) -> bool:
216 return (
217 candidate["occurrences"] < policy["min_occurrences"]
218 or candidate["confidence"] < policy["min_confidence"]
219 )
222def _find_duplicate(
223 candidate: dict[str, Any],
224 existing: list[dict[str, Any]],
225 threshold: float,
226) -> dict[str, Any] | None:
227 for item in existing:
228 if _same_key(candidate, item) or _same_title(candidate, item):
229 return item
230 if _jaccard(candidate["tokens"], item["tokens"]) >= threshold:
231 return item
232 return None
235def _same_key(candidate: dict[str, Any], existing: dict[str, Any]) -> bool:
236 return bool(candidate["dedupe_key"] and candidate["dedupe_key"] == existing["dedupe_key"])
239def _same_title(candidate: dict[str, Any], existing: dict[str, Any]) -> bool:
240 return bool(
241 candidate["normalized_title"]
242 and candidate["normalized_title"] == existing["normalized_title"]
243 )
246def _is_open(raw: dict[str, Any]) -> bool:
247 state = _string(raw.get("state")).lower()
248 # Missing state is treated as open so incomplete GitHub/search fixtures
249 # suppress duplicates conservatively instead of creating duplicate work.
250 return state in {"", "open"}
253def _tokens(text: str) -> set[str]:
254 return set(_TOKEN_RE.findall(_normalize_text(text)))
257def _jaccard(left: set[str], right: set[str]) -> float:
258 if not left or not right:
259 return 0.0
260 return len(left & right) / len(left | right)
263def _normalize_text(text: str) -> str:
264 return " ".join(_TOKEN_RE.findall(text.lower()))
267def _string(value: Any) -> str:
268 return value.strip() if isinstance(value, str) and value.strip() else ""
271def _positive_int(value: Any, default: int) -> int:
272 return value if isinstance(value, int) and value > 0 else default
275def _confidence(value: Any, default: float) -> float:
276 return value if isinstance(value, int | float) and 0 <= value <= 1 else default
279def _count(decisions: list[dict[str, Any]], decision: str) -> int:
280 return sum(item["decision"] == decision for item in decisions)