Coverage for src/keel/workcreation.py: 100%

88 statements  

« prev     ^ index     » next       coverage.py v7.16.2, created at 2026-10-02 20:26 +0000

1"""Deterministic policy for signal-driven work creation.""" 

2 

3from __future__ import annotations 

4 

5import re 

6from dataclasses import dataclass 

7from typing import Any 

8 

9SCHEMA_VERSION = "keel.work-creation.v1" 

10DEFAULT_MIN_OCCURRENCES = 2 

11DEFAULT_MIN_CONFIDENCE = 0.6 

12DEFAULT_MAX_CREATIONS = 5 

13DEFAULT_NEAR_TEXT_SIMILARITY = 0.6 

14 

15DECISIONS = ( 

16 "create", 

17 "suppress-transient", 

18 "suppress-duplicate", 

19 "limit-reached", 

20) 

21 

22_TOKEN_RE = re.compile(r"[a-z0-9]+") 

23 

24 

25@dataclass(frozen=True) 

26class WorkDecision: 

27 """One deterministic work-creation decision.""" 

28 

29 candidate_id: str 

30 decision: str 

31 reason: str 

32 title: str 

33 duplicate_of: int | None = None 

34 

35 def as_dict(self) -> dict[str, Any]: 

36 result: dict[str, Any] = { 

37 "candidate_id": self.candidate_id, 

38 "decision": self.decision, 

39 "reason": self.reason, 

40 "title": self.title, 

41 "creates_issue": self.decision == "create", 

42 } 

43 if self.duplicate_of is not None: 

44 result["duplicate_of"] = self.duplicate_of 

45 return result 

46 

47 

48def contract_as_dict(*, near_text_similarity: float | None = None) -> dict[str, Any]: 

49 """Return the shared work-creation policy contract. 

50 

51 ``near_text_similarity`` is the project's resolved 

52 ``policy_pack.scan.near_text_similarity``. It has to be passed in rather than 

53 defaulted here, because the caller embeds this dict *beside* a `dedupe` block that 

54 already honours the knob — hardcoding the default shipped two different thresholds 

55 under near-identical keys in one contract, agreeing only as long as the project set 

56 the knob to exactly the built-in default (#633). 

57 """ 

58 return { 

59 "schema_version": SCHEMA_VERSION, 

60 "consumer_neutral": True, 

61 "deterministic": True, 

62 "stdlib_only": True, 

63 "source": "signal-driven commands", 

64 "decisions": list(DECISIONS), 

65 "transient_filter": { 

66 "default_min_occurrences": DEFAULT_MIN_OCCURRENCES, 

67 "default_min_confidence": DEFAULT_MIN_CONFIDENCE, 

68 "transient_outcome": "suppress-transient", 

69 }, 

70 "dedupe": { 

71 "against": "open work", 

72 "keys": ["dedupe_key", "normalized_title", "near_text"], 

73 "near_text_similarity": _confidence(near_text_similarity, DEFAULT_NEAR_TEXT_SIMILARITY), 

74 "duplicate_outcome": "suppress-duplicate", 

75 }, 

76 "cycle_limit": { 

77 "default_max_creations": DEFAULT_MAX_CREATIONS, 

78 "limit_outcome": "limit-reached", 

79 }, 

80 "consumers": [ 

81 "regression", 

82 "review-all-day", 

83 "coverage", 

84 "deps-audit", 

85 "flake-audit", 

86 ], 

87 } 

88 

89 

90def evaluate_candidates( 

91 candidates: list[dict[str, Any]] | tuple[dict[str, Any], ...], 

92 existing_work: list[dict[str, Any]] | tuple[dict[str, Any], ...] = (), 

93 *, 

94 min_occurrences: int = DEFAULT_MIN_OCCURRENCES, 

95 min_confidence: float = DEFAULT_MIN_CONFIDENCE, 

96 max_creations: int = DEFAULT_MAX_CREATIONS, 

97 near_text_similarity: float = DEFAULT_NEAR_TEXT_SIMILARITY, 

98) -> dict[str, Any]: 

99 """Evaluate candidate signals against transient, dedupe, and cycle-limit policy.""" 

100 policy = { 

101 "min_occurrences": _positive_int(min_occurrences, DEFAULT_MIN_OCCURRENCES), 

102 "min_confidence": _confidence(min_confidence, DEFAULT_MIN_CONFIDENCE), 

103 "max_creations": _positive_int(max_creations, DEFAULT_MAX_CREATIONS), 

104 "near_text_similarity": _confidence( 

105 near_text_similarity, 

106 DEFAULT_NEAR_TEXT_SIMILARITY, 

107 ), 

108 } 

109 normalized_existing = [ 

110 _normalize_existing(item) 

111 for item in existing_work 

112 if isinstance(item, dict) and _is_open(item) 

113 ] 

114 created = 0 

115 decisions: list[WorkDecision] = [] 

116 for index, raw in enumerate(candidates, start=1): 

117 if not isinstance(raw, dict): 

118 continue 

119 candidate = _normalize_candidate(raw, index) 

120 duplicate = _find_duplicate(candidate, normalized_existing, policy["near_text_similarity"]) 

121 if _is_transient(candidate, policy): 

122 decisions.append( 

123 WorkDecision( 

124 candidate["id"], 

125 "suppress-transient", 

126 "transient-signal", 

127 candidate["title"], 

128 ) 

129 ) 

130 elif duplicate is not None: 

131 decisions.append( 

132 WorkDecision( 

133 candidate["id"], 

134 "suppress-duplicate", 

135 "open-work-duplicate", 

136 candidate["title"], 

137 duplicate_of=duplicate["number"], 

138 ) 

139 ) 

140 elif created >= policy["max_creations"]: 

141 decisions.append( 

142 WorkDecision( 

143 candidate["id"], 

144 "limit-reached", 

145 "per-cycle-limit-reached", 

146 candidate["title"], 

147 ) 

148 ) 

149 else: 

150 created += 1 

151 decisions.append( 

152 WorkDecision( 

153 candidate["id"], 

154 "create", 

155 "eligible", 

156 candidate["title"], 

157 ) 

158 ) 

159 normalized_existing.append(_created_as_existing(candidate, created)) 

160 decision_dicts = [decision.as_dict() for decision in decisions] 

161 return { 

162 "schema_version": SCHEMA_VERSION, 

163 "status": "pass", 

164 "policy": policy, 

165 "summary": { 

166 "candidates": len([item for item in candidates if isinstance(item, dict)]), 

167 "create": _count(decision_dicts, "create"), 

168 "suppress_transient": _count(decision_dicts, "suppress-transient"), 

169 "suppress_duplicate": _count(decision_dicts, "suppress-duplicate"), 

170 "limit_reached": _count(decision_dicts, "limit-reached"), 

171 }, 

172 "decisions": decision_dicts, 

173 } 

174 

175 

176def _normalize_candidate(raw: dict[str, Any], index: int) -> dict[str, Any]: 

177 title = _string(raw.get("title")) or f"candidate-{index}" 

178 body = _string(raw.get("body")) 

179 return { 

180 "id": _string(raw.get("id")) or f"candidate-{index}", 

181 "title": title, 

182 "body": body, 

183 "dedupe_key": _string(raw.get("dedupe_key")), 

184 "occurrences": _positive_int(raw.get("occurrences"), 1), 

185 "confidence": _confidence(raw.get("confidence"), 1.0), 

186 "normalized_title": _normalize_text(title), 

187 "tokens": _tokens(f"{title} {body}"), 

188 } 

189 

190 

191def _normalize_existing(raw: dict[str, Any]) -> dict[str, Any]: 

192 title = _string(raw.get("title")) 

193 body = _string(raw.get("body")) 

194 return { 

195 "number": raw.get("number") if isinstance(raw.get("number"), int) else None, 

196 "title": title, 

197 "body": body, 

198 "dedupe_key": _string(raw.get("dedupe_key")), 

199 "normalized_title": _normalize_text(title), 

200 "tokens": _tokens(f"{title} {body}"), 

201 } 

202 

203 

204def _created_as_existing(candidate: dict[str, Any], created_index: int) -> dict[str, Any]: 

205 return { 

206 "number": -created_index, 

207 "title": candidate["title"], 

208 "body": candidate["body"], 

209 "dedupe_key": candidate["dedupe_key"], 

210 "normalized_title": candidate["normalized_title"], 

211 "tokens": candidate["tokens"], 

212 } 

213 

214 

215def _is_transient(candidate: dict[str, Any], policy: dict[str, Any]) -> bool: 

216 return ( 

217 candidate["occurrences"] < policy["min_occurrences"] 

218 or candidate["confidence"] < policy["min_confidence"] 

219 ) 

220 

221 

222def _find_duplicate( 

223 candidate: dict[str, Any], 

224 existing: list[dict[str, Any]], 

225 threshold: float, 

226) -> dict[str, Any] | None: 

227 for item in existing: 

228 if _same_key(candidate, item) or _same_title(candidate, item): 

229 return item 

230 if _jaccard(candidate["tokens"], item["tokens"]) >= threshold: 

231 return item 

232 return None 

233 

234 

235def _same_key(candidate: dict[str, Any], existing: dict[str, Any]) -> bool: 

236 return bool(candidate["dedupe_key"] and candidate["dedupe_key"] == existing["dedupe_key"]) 

237 

238 

239def _same_title(candidate: dict[str, Any], existing: dict[str, Any]) -> bool: 

240 return bool( 

241 candidate["normalized_title"] 

242 and candidate["normalized_title"] == existing["normalized_title"] 

243 ) 

244 

245 

246def _is_open(raw: dict[str, Any]) -> bool: 

247 state = _string(raw.get("state")).lower() 

248 # Missing state is treated as open so incomplete GitHub/search fixtures 

249 # suppress duplicates conservatively instead of creating duplicate work. 

250 return state in {"", "open"} 

251 

252 

253def _tokens(text: str) -> set[str]: 

254 return set(_TOKEN_RE.findall(_normalize_text(text))) 

255 

256 

257def _jaccard(left: set[str], right: set[str]) -> float: 

258 if not left or not right: 

259 return 0.0 

260 return len(left & right) / len(left | right) 

261 

262 

263def _normalize_text(text: str) -> str: 

264 return " ".join(_TOKEN_RE.findall(text.lower())) 

265 

266 

267def _string(value: Any) -> str: 

268 return value.strip() if isinstance(value, str) and value.strip() else "" 

269 

270 

271def _positive_int(value: Any, default: int) -> int: 

272 return value if isinstance(value, int) and value > 0 else default 

273 

274 

275def _confidence(value: Any, default: float) -> float: 

276 return value if isinstance(value, int | float) and 0 <= value <= 1 else default 

277 

278 

279def _count(decisions: list[dict[str, Any]], decision: str) -> int: 

280 return sum(item["decision"] == decision for item in decisions)