Coverage for src/keel/intake.py: 100%

206 statements  

« prev     ^ index     » next       coverage.py v7.16.2, created at 2026-10-02 20:26 +0000

1"""Pure issue intake and readiness classification for work-owning commands.""" 

2 

3from __future__ import annotations 

4 

5import re 

6from dataclasses import dataclass 

7from typing import Any 

8 

9READY = "ready" 

10NEEDS_INPUT = "needs-input" 

11BLOCKED = "blocked" 

12OUT_OF_SCOPE = "out-of-scope" 

13READINESS_STATUSES = (READY, NEEDS_INPUT, BLOCKED, OUT_OF_SCOPE) 

14 

15_SECTION_RE = re.compile(r"^(?P<hashes>#{1,6})\s+(?P<title>.+?)\s*$", re.MULTILINE) 

16_BULLET_RE = re.compile(r"^\s*(?:[-*+]|\d+[.)])\s+(?P<text>.+?)\s*$") 

17_BLOCKED_RE = re.compile( 

18 r"\b(blocked by|depends on|dependency|waiting on|needs dependency|blocked until)\b", 

19 re.IGNORECASE, 

20) 

21_NON_BLOCKING_DEPENDENCY_RE = re.compile( 

22 r"\b(no dependenc(?:y|ies)|dependenc(?:y|ies):\s*(?:none|no|n/a)|" 

23 r"depends on:\s*(?:none|no|n/a)|blocked by:\s*(?:none|no|n/a)|" 

24 r"waiting on\s+(?:none|no one|nobody|no-one))\b", 

25 re.IGNORECASE, 

26) 

27_AMBIGUOUS_RE = re.compile( 

28 r"\b(tbd|todo|unclear|ambiguous|maybe|not sure|needs clarification|decide later)\b", 

29 re.IGNORECASE, 

30) 

31_OUT_OF_SCOPE_LABELS = frozenset({"out-of-scope", "wontfix", "not-planned"}) 

32#: Headings whose whole section is a boundary statement **about the change** (#1168). 

33#: Normalised by :func:`_normalize_heading`, so ``## Non-goals`` and ``## non goals`` 

34#: are the same heading; ``not in this change`` is this repository's own spelling, 

35#: used by every issue written since #1182. 

36#: 

37#: Deliberately **not** here: ``Not planned``, ``Will not do``, ``Decision``, 

38#: ``Status``, ``Resolution``. Those are close-reason headings — a section saying the 

39#: issue will not be done — and dropping them would silence the very statement intake 

40#: exists to read. The round-1 gate seats caught exactly that: with them in the set, 

41#: `## Not planned` / `This issue is out of scope; closing.` came back `ready` with 

42#: `can_mutate_code: true`. The test is what the section says about, not how 

43#: negative it sounds. 

44_SCOPE_EXCLUSION_HEADINGS = frozenset( 

45 { 

46 "out of scope", 

47 "non goals", 

48 "non goal", 

49 "not in scope", 

50 "not in this change", 

51 } 

52) 

53#: The short form, anchored at the start, for the **title** and nowhere else. 

54#: 

55#: A title is one line, with nothing in front of it and nothing after it to qualify — it 

56#: names the issue's whole subject. A body sentence is not: `Out of scope for v1: the 

57#: Android client.` opening an issue is a boundary, `Not in scope for Windows.` is a 

58#: carve-out, and neither closes anything. Fifteen review rounds went into trying to tell 

59#: those from a closure by position, and each rule inverted on some real sentence. The 

60#: body asks the one question that has an answer: is the issue the subject. 

61_OUT_OF_SCOPE_OPENER_RE = re.compile( 

62 r"^(?:out[- ]of[- ]scope|not planned|wontfix|won't fix|not in scope)\b", 

63 re.IGNORECASE, 

64) 

65#: A declaration **in the body**: a sentence whose subject is the issue. 

66#: 

67#: Anchored at the start of the *sentence* — which is not the same as the start of a 

68#: line, and that distinction is the whole lesson of this change. Eight rounds tried to 

69#: anchor a short `Out of scope: …` form at a line start and it broke on every piece of 

70#: markdown that can precede one. A sentence start is stable, and requiring the issue to 

71#: be the **subject** is what separates a closure from a mention: "This issue is out of 

72#: scope" closes it, "A backport of this issue is out of scope" carves out a backport. 

73#: An unanchored match refused the second, which is #1168 inverted. 

74_OUT_OF_SCOPE_DECLARATION_RE = re.compile( 

75 r"^(?:this|the)\s+issue\s+(?:is|was|remains)\s+" 

76 r"(?:out[- ]of[- ]scope|not planned|wontfix|won't fix|not in scope)" 

77 r"(?![A-Za-z0-9])", 

78 re.IGNORECASE, 

79) 

80#: Markdown that can sit in front of a sentence without being part of it. Stripped before 

81#: the anchor is applied, so a bullet or a quote marker does not hide the subject. 

82_LEADING_MARKUP_RE = re.compile( 

83 r"^(?:" 

84 r"\s*(?:[-*+]|\d+[.)])\s+" # list marker 

85 r"|\s*>\s*" # block quote 

86 r"|\s*\[[ xX]?\]\s*" # task box 

87 r"|\s*(?:-{3,}|\*{3,}|_{3,})\s*" # thematic break 

88 # `[\s\S]` rather than `.`: an HTML comment may span lines, and `.` does not cross 

89 # one, so a multi-line comment in front of a sentence was left where it stood. 

90 r"|\s*<!--[\s\S]*?-->\s*" # html comment 

91 r"|\s*!\[[^\]]*\]\([^)]*\)\s*" # image 

92 r"|[*_]{1,3}" # emphasis run 

93 r"|\s+" 

94 r")+" 

95) 

96#: A short label opening a heading or a line — ``Decision — …``, ``Status: …``. Removed 

97#: before the anchor so the subject that follows is seen, while a sentence that merely 

98#: *mentions* the issue further in ("A backport of this issue is out of scope") is not, 

99#: because it carries no such separator before the subject. Bounded in length and 

100#: forbidden sentence-ending punctuation, so it cannot eat a real clause. 

101_LEADING_LABEL_RE = re.compile( 

102 # A label is a **short name**, not a clause: at most three words, each a word with 

103 # optional internal punctuation, then an explicit terminator. `Decision:`, 

104 # `Follow-up:`, `Update 2026-09-14:`, `Decision (2026-09-14) —` all qualify. 

105 # 

106 # The length-capped "anything up to a terminator" form this replaces let the label 

107 # strip re-anchor the pattern in the middle of a compound sentence: `Users need safer 

108 # sync — this issue is not in scope for Windows.` became `this issue is not in scope 

109 # for Windows.` and read as a closure, when it carves out Windows. That is #1168 

110 # inverted, and the subject test cannot see it because the subject really is there — 

111 # just not at the start of the sentence, which is the whole point of the anchor. 

112 r"^\w[\w.()/'’-]*(?:\s+[\w.()/'’-]+){0,2}\s*[:\u2014\u2013]\s*" 

113 # A spaced hyphen separates the same way; `-` inside a word does not. 

114 r"|^\w[\w.()/'’]*(?:\s+[\w.()/'’]+){0,2}\s+-\s+" 

115) 

116 

117 

118def _declaration_candidates(text: str, *, allow_label: bool = False) -> tuple[str, ...]: 

119 """Every prefix-stripped form of one sentence the declaration may be anchored in. 

120 

121 The sentence **as written** comes first, because the label pattern cannot tell a 

122 `Decision:` prefix from the declaration's own `out of scope: closing.` — stripping 

123 unconditionally ate the closure down to `closing.`. Then markdown removed; then a 

124 label removed and markdown removed *again*, because `**Decision:** This issue is…` 

125 leaves a closing `**` between the label and the subject that the first pass cannot 

126 see. Each round of this review found one of these. 

127 """ 

128 forms = [text.strip()] 

129 stripped = _LEADING_MARKUP_RE.sub("", forms[0]) 

130 if stripped != forms[0]: 

131 forms.append(stripped) 

132 if not allow_label: 

133 return tuple(forms) 

134 delabelled = _LEADING_LABEL_RE.sub("", stripped, count=1).strip() 

135 if delabelled != stripped: 

136 forms.append(_LEADING_MARKUP_RE.sub("", delabelled).strip()) 

137 return tuple(forms) 

138 

139 

140_DOCS_RE = re.compile(r"\b(doc|docs|documentation|readme|changelog)\b", re.IGNORECASE) 

141_TESTS_RE = re.compile(r"\b(test|tests|coverage|ci|lint)\b", re.IGNORECASE) 

142_BLOCKED_LABELS = frozenset({"blocked", "status:blocked", "needs-dependency"}) 

143_RISK_RE = re.compile( 

144 r"\b(security|release|migration|schema|api|breaking|billing|secret|credential|ci|" 

145 r"production|compatibility)\b", 

146 re.IGNORECASE, 

147) 

148 

149 

150@dataclass(frozen=True) 

151class IssueContext: 

152 """Issue text supplied by an adapter before code mutation starts.""" 

153 

154 title: str | None = None 

155 body: str | None = None 

156 labels: tuple[str, ...] = () 

157 

158 @property 

159 def provided(self) -> bool: 

160 return bool((self.title or "").strip() or (self.body or "").strip() or self.labels) 

161 

162 

163def assess_issue( 

164 *, 

165 title: str | None = None, 

166 body: str | None = None, 

167 labels: tuple[str, ...] = (), 

168) -> dict[str, Any]: 

169 """Return a deterministic readiness record for an issue-like work item.""" 

170 context = IssueContext(title=title, body=body, labels=tuple(labels)) 

171 sections = _sections(context.body or "") 

172 acceptance = _acceptance_criteria(sections) 

173 objective = _objective(context.title, context.body, sections) 

174 deliverable = _deliverable(sections, acceptance) 

175 combined = " ".join(filter(None, (context.title, context.body, " ".join(context.labels)))) 

176 normalized_labels = tuple(label.strip().lower() for label in context.labels) 

177 

178 missing: list[str] = [] 

179 blockers: list[str] = [] 

180 questions: list[str] = [] 

181 

182 if not objective: 

183 missing.append("objective") 

184 questions.append("What objective should this issue accomplish?") 

185 if not deliverable: 

186 missing.append("deliverable") 

187 questions.append("What concrete deliverable should be produced?") 

188 if not acceptance: 

189 missing.append("acceptance_criteria") 

190 questions.append("What acceptance criteria define done for this issue?") 

191 

192 status = READY 

193 reason = "Issue has an objective, deliverable, and acceptance criteria." 

194 out_of_scope_reason = _out_of_scope_reason( 

195 title=context.title, 

196 body=context.body, 

197 labels=normalized_labels, 

198 ) 

199 if out_of_scope_reason: 

200 status = OUT_OF_SCOPE 

201 reason = out_of_scope_reason 

202 questions = [] 

203 elif _is_blocked(combined, normalized_labels): 

204 status = BLOCKED 

205 reason = "Issue declares a dependency or waiting condition." 

206 blockers.append(_blocked_summary(combined)) 

207 questions.append("What dependency must clear before this issue can start?") 

208 elif missing: 

209 status = NEEDS_INPUT 

210 reason = f"Missing required intake field(s): {', '.join(missing)}." 

211 elif _AMBIGUOUS_RE.search(combined): 

212 status = NEEDS_INPUT 

213 reason = "Issue scope contains ambiguous or deferred wording." 

214 missing.append("scope_clarity") 

215 questions.append("Which exact scope should be implemented now?") 

216 

217 return { 

218 "schema_version": "keel.issue-intake.v1", 

219 "provided": context.provided, 

220 "status": status, 

221 "can_mutate_code": status == READY, 

222 "work_block_policy": { 

223 "skip_when_not_ready": status != READY, 

224 "continue_with_next_ready_issue": status != READY, 

225 "non_ready_statuses": [NEEDS_INPUT, BLOCKED, OUT_OF_SCOPE], 

226 }, 

227 "reason": reason, 

228 "objective": objective, 

229 "deliverable": deliverable, 

230 "acceptance_criteria": acceptance, 

231 "risk_tier_inputs": _risk_inputs(combined, normalized_labels), 

232 "required_docs_tests": _required_docs_tests(combined, acceptance), 

233 "missing_info": missing, 

234 "blockers": blockers, 

235 "questions": _unique(questions)[:3], 

236 "ledger_record": { 

237 "readiness": status, 

238 "mutation_allowed": status == READY, 

239 "skip_reason": None if status == READY else reason, 

240 "question_count": len(_unique(questions)[:3]), 

241 }, 

242 } 

243 

244 

245def _sections(body: str) -> dict[str, str]: 

246 matches = list(_SECTION_RE.finditer(body)) 

247 if not matches: 

248 return {} 

249 sections: dict[str, str] = {} 

250 for index, match in enumerate(matches): 

251 title = _normalize_heading(match.group("title")) 

252 start = match.end() 

253 end = matches[index + 1].start() if index + 1 < len(matches) else len(body) 

254 sections[title] = body[start:end].strip() 

255 return sections 

256 

257 

258def _normalize_heading(value: str) -> str: 

259 return re.sub(r"[^a-z0-9]+", " ", value.lower()).strip() 

260 

261 

262def _objective(title: str | None, body: str | None, sections: dict[str, str]) -> str | None: 

263 for key in ("objective", "problem", "summary", "context"): 

264 if value := _first_sentence_or_bullet(sections.get(key, "")): 

265 return value 

266 if body and not sections: 

267 return _first_sentence_or_bullet(body) 

268 return title.strip() if title and title.strip() else None 

269 

270 

271def _deliverable(sections: dict[str, str], acceptance: list[str]) -> str | None: 

272 del acceptance 

273 for key in ("deliverable", "proposed direction", "proposal", "scope", "implementation"): 

274 if value := _first_sentence_or_bullet(sections.get(key, "")): 

275 return value 

276 return None 

277 

278 

279def _acceptance_criteria(sections: dict[str, str]) -> list[str]: 

280 for key in ( 

281 "acceptance criteria", 

282 "acceptance", 

283 "definition of done", 

284 "done when", 

285 "dod", 

286 ): 

287 if key in sections: 

288 bullets = _bullets(sections[key]) 

289 return bullets if bullets else _sentences(sections[key]) 

290 return [] 

291 

292 

293def _first_sentence_or_bullet(text: str) -> str | None: 

294 bullets = _bullets(text) 

295 if bullets: 

296 return bullets[0] 

297 sentences = _sentences(text) 

298 return sentences[0] if sentences else None 

299 

300 

301def _bullets(text: str) -> list[str]: 

302 items: list[str] = [] 

303 for line in text.splitlines(): 

304 if match := _BULLET_RE.match(line): 

305 item = match.group("text").strip() 

306 if item: 

307 items.append(item) 

308 return items 

309 

310 

311def _bullet_lines(text: str) -> list[str]: 

312 """List items, markers removed — each a statement in its own right. 

313 

314 `_sentences` joins lines carrying no terminator, so a plain bullet above a 

315 declaration (`- Discussed with the team` then `- This issue is out of scope.`) puts 

316 the declaration mid-sentence, past the anchor. A list item is a statement whether or 

317 not it ends in a full stop. 

318 

319 List items **only**. Every line was tried and refused prose the moment a wrap put the 

320 phrase at a line start; block quotes were tried and refused it too, because Markdown 

321 prefixes each continuation line of a quote with `>`. Neither shape carries a list 

322 marker. The subject test is what makes this safe now — a bullet has to name the issue 

323 to match anything at all. 

324 """ 

325 return [item for line in text.splitlines() if (item := _bullet_text(line))] 

326 

327 

328def _bullet_text(line: str) -> str: 

329 match = _BULLET_RE.match(line) 

330 return match.group("text").strip() if match else "" 

331 

332 

333def _sentences(text: str) -> list[str]: 

334 compact = " ".join(line.strip() for line in text.splitlines() if line.strip()) 

335 if not compact: 

336 return [] 

337 parts = [part.strip() for part in re.split(r"(?<=[.!?])\s+", compact) if part.strip()] 

338 return parts or [compact] 

339 

340 

341def _is_scope_exclusion_heading(title: str) -> bool: 

342 """Does this heading open a section that bounds the **change**? 

343 

344 The test is *starts with*, not equals. ``## Out of scope for v1``, 

345 ``## Out of scope: mobile UI`` and ``## Non-goals for now`` are the same kind of 

346 section as ``## Out of scope``, and an equality test refused the issues that wrote 

347 them — #1168 coming back through the heading, which is now read as prose. A section 

348 titled "Out of scope…" is a boundary whatever qualifies it. 

349 

350 A close-reason heading (``Not planned``, ``Decision``, ``Status``) is deliberately 

351 not in the set: it names a status for the issue, and dropping it would silence the 

352 statement intake exists to read. 

353 """ 

354 normalised = _normalize_heading(title) 

355 return any( 

356 normalised == heading or normalised.startswith(heading + " ") 

357 for heading in _SCOPE_EXCLUSION_HEADINGS 

358 ) 

359 

360 

361def _scannable_chunks(body: str) -> list[tuple[bool, str]]: 

362 """``body`` split into the pieces a declaration may hide in, boundaries removed. 

363 

364 A scope-exclusion section goes whole — heading and content. What remains comes back 

365 per section, with the heading's own text as a chunk **separate from** its body: 

366 inline, it glues onto the sentence below (`- works ## Decision Out of scope: 

367 closing.`) and the declaration pattern is anchored at a sentence start; dropped, a 

368 declaration written *in* a heading becomes invisible. Each gate round found one of 

369 those halves. 

370 """ 

371 matches = list(_SECTION_RE.finditer(body)) 

372 if not matches: 

373 return [(False, body)] 

374 chunks: list[tuple[bool, str]] = [(False, body[: matches[0].start()])] 

375 for index, match in enumerate(matches): 

376 # An exclusion section owns its own body and nothing else. Owning nested headings 

377 # was tried and failed in both directions: `# Out of scope` over a run of `##` 

378 # sections swallowed the `## Decision` that closed the issue, and every heuristic 

379 # for telling a title level from a section level inverted on some real body. What 

380 # makes the simple rule safe is the subject test above — a bullet under a boundary 

381 # section saying `Out of scope: the mobile client` names no issue and matches 

382 # nothing, so the section's *contents* no longer need hiding, only its own prose. 

383 if _is_scope_exclusion_heading(match.group("title")): 

384 continue 

385 end = matches[index + 1].start() if index + 1 < len(matches) else len(body) 

386 # Flagged as a heading: its text may carry a record label (`Decision —`) in front 

387 # of the declaration. It does **not** get the short form — `## Not planned` over a 

388 # list of features is a boundary section, and reading its title as a closure 

389 # refused the issue while the identical bullets under `## Non-goals` passed. 

390 chunks.append((True, match.group("title"))) 

391 chunks.append((False, body[match.end() : end])) 

392 return [(is_heading, chunk) for is_heading, chunk in chunks if chunk.strip()] 

393 

394 

395def _out_of_scope_reason( 

396 *, 

397 title: str | None, 

398 body: str | None, 

399 labels: tuple[str, ...], 

400) -> str | None: 

401 """Return the concrete scope declaration, ignoring structural exclusions. 

402 

403 Two independent filters, and each does one job: 

404 

405 * **Structurally**, a scope-exclusion section is removed whole. Its heading and its 

406 bullets describe what the change leaves out; none of it says the issue is closed. 

407 * **In what remains**, a declaration is a sentence whose *subject is the issue* — 

408 ``this issue is out of scope``, opening the sentence. Not ``Out of scope: …``, 

409 which is the same string as a boundary bullet; not a mention in the middle of a 

410 sentence, which carves out a part rather than closing the whole. The short form 

411 is the title's, where one line and no markdown make an anchor safe. 

412 

413 Everything left is searched, title and body alike. A maintainer writing 

414 ``## Decision — this issue is out of scope; closing`` is declaring it closed, and 

415 intake must not hand that to s2. 

416 

417 ``non-goal`` is deliberately **not** a declaration marker. It names a non-goal of 

418 the change ("Non-goal: rewrite the parser"), which is a boundary like the section 

419 it usually appears under, not a statement that the issue will not be done. The 

420 ``## Non-goals`` heading is handled structurally above. 

421 """ 

422 for label in labels: 

423 if label in _OUT_OF_SCOPE_LABELS: 

424 return f"Issue carries out-of-scope label: {label}." 

425 

426 # The title takes the short form: it names the issue's whole subject, with nothing 

427 # before it and nothing after it to qualify. 

428 if title and (heading := title.strip()): 

429 if _OUT_OF_SCOPE_OPENER_RE.search(heading) or _OUT_OF_SCOPE_DECLARATION_RE.search(heading): 

430 return f"Issue declares itself out of scope: {heading}" 

431 

432 #: `(text, may a leading label be stripped)`. A heading's `Decision —` prefix is a 

433 #: record label; the same shape inside prose is a prepositional phrase (`For Windows: 

434 #: this issue is not in scope.`), and stripping it turned a carve-out into a closure. 

435 candidates: list[tuple[str, bool]] = [] 

436 if body: 

437 for is_heading, chunk in _scannable_chunks(body): 

438 candidates.extend((sentence, is_heading) for sentence in _sentences(chunk)) 

439 candidates.extend((line, False) for line in _bullet_lines(chunk)) 

440 

441 for candidate, allow_label in candidates: 

442 # Leading markdown removed first, so the anchor sees the sentence rather than the 

443 # bullet in front of it. 

444 if any( 

445 _OUT_OF_SCOPE_DECLARATION_RE.search(form) 

446 for form in _declaration_candidates(candidate, allow_label=allow_label) 

447 ): 

448 return f"Issue declares itself out of scope: {candidate.strip()}" 

449 return None 

450 

451 

452def _is_blocked(combined: str, labels: tuple[str, ...]) -> bool: 

453 if not _BLOCKED_LABELS.isdisjoint(labels): 

454 return True 

455 

456 compact = " ".join(line.strip() for line in combined.splitlines() if line.strip()) 

457 # Fast path: bypass expensive sentence iteration if no 

458 # blocked regex match exists globally in normalized text 

459 if not _BLOCKED_RE.search(compact): 

460 return False 

461 

462 # ⚡ Bolt Optimization: Unroll any() generator to avoid generator overhead 

463 for sentence in _sentences(compact): 

464 if _is_actionable_blocker(sentence): 

465 return True 

466 return False 

467 

468 

469def _blocked_summary(combined: str) -> str: 

470 sentences = _sentences(combined) 

471 for sentence in sentences: 

472 if _is_actionable_blocker(sentence): 

473 return sentence 

474 return "Declared blocked dependency." 

475 

476 

477def _is_actionable_blocker(text: str) -> bool: 

478 return bool(_BLOCKED_RE.search(text)) and not bool(_NON_BLOCKING_DEPENDENCY_RE.search(text)) 

479 

480 

481def _risk_inputs(combined: str, labels: tuple[str, ...]) -> dict[str, Any]: 

482 keywords = _unique(match.group(0).lower() for match in _RISK_RE.finditer(combined)) 

483 risk_labels = [label for label in labels if "risk" in label or label.startswith("tier")] 

484 return { 

485 "keywords": keywords, 

486 "labels": risk_labels, 

487 "has_high_risk_signal": bool(keywords or risk_labels), 

488 } 

489 

490 

491def _required_docs_tests(combined: str, acceptance: list[str]) -> dict[str, Any]: 

492 text = " ".join([combined, *acceptance]) 

493 return { 

494 "docs": "required" if _DOCS_RE.search(text) else "unspecified", 

495 "tests": "required" if _TESTS_RE.search(text) else "unspecified", 

496 } 

497 

498 

499def _unique(items) -> list[str]: 

500 seen = set() 

501 result = [] 

502 for item in items: 

503 v = str(item).strip() 

504 if v and v not in seen: 

505 seen.add(v) 

506 result.append(v) 

507 return result