Coverage for src/keel/intake.py: 100%
206 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-02 20:26 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-02 20:26 +0000
1"""Pure issue intake and readiness classification for work-owning commands."""
3from __future__ import annotations
5import re
6from dataclasses import dataclass
7from typing import Any
9READY = "ready"
10NEEDS_INPUT = "needs-input"
11BLOCKED = "blocked"
12OUT_OF_SCOPE = "out-of-scope"
13READINESS_STATUSES = (READY, NEEDS_INPUT, BLOCKED, OUT_OF_SCOPE)
15_SECTION_RE = re.compile(r"^(?P<hashes>#{1,6})\s+(?P<title>.+?)\s*$", re.MULTILINE)
16_BULLET_RE = re.compile(r"^\s*(?:[-*+]|\d+[.)])\s+(?P<text>.+?)\s*$")
17_BLOCKED_RE = re.compile(
18 r"\b(blocked by|depends on|dependency|waiting on|needs dependency|blocked until)\b",
19 re.IGNORECASE,
20)
21_NON_BLOCKING_DEPENDENCY_RE = re.compile(
22 r"\b(no dependenc(?:y|ies)|dependenc(?:y|ies):\s*(?:none|no|n/a)|"
23 r"depends on:\s*(?:none|no|n/a)|blocked by:\s*(?:none|no|n/a)|"
24 r"waiting on\s+(?:none|no one|nobody|no-one))\b",
25 re.IGNORECASE,
26)
27_AMBIGUOUS_RE = re.compile(
28 r"\b(tbd|todo|unclear|ambiguous|maybe|not sure|needs clarification|decide later)\b",
29 re.IGNORECASE,
30)
31_OUT_OF_SCOPE_LABELS = frozenset({"out-of-scope", "wontfix", "not-planned"})
32#: Headings whose whole section is a boundary statement **about the change** (#1168).
33#: Normalised by :func:`_normalize_heading`, so ``## Non-goals`` and ``## non goals``
34#: are the same heading; ``not in this change`` is this repository's own spelling,
35#: used by every issue written since #1182.
36#:
37#: Deliberately **not** here: ``Not planned``, ``Will not do``, ``Decision``,
38#: ``Status``, ``Resolution``. Those are close-reason headings — a section saying the
39#: issue will not be done — and dropping them would silence the very statement intake
40#: exists to read. The round-1 gate seats caught exactly that: with them in the set,
41#: `## Not planned` / `This issue is out of scope; closing.` came back `ready` with
42#: `can_mutate_code: true`. The test is what the section says about, not how
43#: negative it sounds.
44_SCOPE_EXCLUSION_HEADINGS = frozenset(
45 {
46 "out of scope",
47 "non goals",
48 "non goal",
49 "not in scope",
50 "not in this change",
51 }
52)
53#: The short form, anchored at the start, for the **title** and nowhere else.
54#:
55#: A title is one line, with nothing in front of it and nothing after it to qualify — it
56#: names the issue's whole subject. A body sentence is not: `Out of scope for v1: the
57#: Android client.` opening an issue is a boundary, `Not in scope for Windows.` is a
58#: carve-out, and neither closes anything. Fifteen review rounds went into trying to tell
59#: those from a closure by position, and each rule inverted on some real sentence. The
60#: body asks the one question that has an answer: is the issue the subject.
61_OUT_OF_SCOPE_OPENER_RE = re.compile(
62 r"^(?:out[- ]of[- ]scope|not planned|wontfix|won't fix|not in scope)\b",
63 re.IGNORECASE,
64)
65#: A declaration **in the body**: a sentence whose subject is the issue.
66#:
67#: Anchored at the start of the *sentence* — which is not the same as the start of a
68#: line, and that distinction is the whole lesson of this change. Eight rounds tried to
69#: anchor a short `Out of scope: …` form at a line start and it broke on every piece of
70#: markdown that can precede one. A sentence start is stable, and requiring the issue to
71#: be the **subject** is what separates a closure from a mention: "This issue is out of
72#: scope" closes it, "A backport of this issue is out of scope" carves out a backport.
73#: An unanchored match refused the second, which is #1168 inverted.
74_OUT_OF_SCOPE_DECLARATION_RE = re.compile(
75 r"^(?:this|the)\s+issue\s+(?:is|was|remains)\s+"
76 r"(?:out[- ]of[- ]scope|not planned|wontfix|won't fix|not in scope)"
77 r"(?![A-Za-z0-9])",
78 re.IGNORECASE,
79)
80#: Markdown that can sit in front of a sentence without being part of it. Stripped before
81#: the anchor is applied, so a bullet or a quote marker does not hide the subject.
82_LEADING_MARKUP_RE = re.compile(
83 r"^(?:"
84 r"\s*(?:[-*+]|\d+[.)])\s+" # list marker
85 r"|\s*>\s*" # block quote
86 r"|\s*\[[ xX]?\]\s*" # task box
87 r"|\s*(?:-{3,}|\*{3,}|_{3,})\s*" # thematic break
88 # `[\s\S]` rather than `.`: an HTML comment may span lines, and `.` does not cross
89 # one, so a multi-line comment in front of a sentence was left where it stood.
90 r"|\s*<!--[\s\S]*?-->\s*" # html comment
91 r"|\s*!\[[^\]]*\]\([^)]*\)\s*" # image
92 r"|[*_]{1,3}" # emphasis run
93 r"|\s+"
94 r")+"
95)
96#: A short label opening a heading or a line — ``Decision — …``, ``Status: …``. Removed
97#: before the anchor so the subject that follows is seen, while a sentence that merely
98#: *mentions* the issue further in ("A backport of this issue is out of scope") is not,
99#: because it carries no such separator before the subject. Bounded in length and
100#: forbidden sentence-ending punctuation, so it cannot eat a real clause.
101_LEADING_LABEL_RE = re.compile(
102 # A label is a **short name**, not a clause: at most three words, each a word with
103 # optional internal punctuation, then an explicit terminator. `Decision:`,
104 # `Follow-up:`, `Update 2026-09-14:`, `Decision (2026-09-14) —` all qualify.
105 #
106 # The length-capped "anything up to a terminator" form this replaces let the label
107 # strip re-anchor the pattern in the middle of a compound sentence: `Users need safer
108 # sync — this issue is not in scope for Windows.` became `this issue is not in scope
109 # for Windows.` and read as a closure, when it carves out Windows. That is #1168
110 # inverted, and the subject test cannot see it because the subject really is there —
111 # just not at the start of the sentence, which is the whole point of the anchor.
112 r"^\w[\w.()/'’-]*(?:\s+[\w.()/'’-]+){0,2}\s*[:\u2014\u2013]\s*"
113 # A spaced hyphen separates the same way; `-` inside a word does not.
114 r"|^\w[\w.()/'’]*(?:\s+[\w.()/'’]+){0,2}\s+-\s+"
115)
118def _declaration_candidates(text: str, *, allow_label: bool = False) -> tuple[str, ...]:
119 """Every prefix-stripped form of one sentence the declaration may be anchored in.
121 The sentence **as written** comes first, because the label pattern cannot tell a
122 `Decision:` prefix from the declaration's own `out of scope: closing.` — stripping
123 unconditionally ate the closure down to `closing.`. Then markdown removed; then a
124 label removed and markdown removed *again*, because `**Decision:** This issue is…`
125 leaves a closing `**` between the label and the subject that the first pass cannot
126 see. Each round of this review found one of these.
127 """
128 forms = [text.strip()]
129 stripped = _LEADING_MARKUP_RE.sub("", forms[0])
130 if stripped != forms[0]:
131 forms.append(stripped)
132 if not allow_label:
133 return tuple(forms)
134 delabelled = _LEADING_LABEL_RE.sub("", stripped, count=1).strip()
135 if delabelled != stripped:
136 forms.append(_LEADING_MARKUP_RE.sub("", delabelled).strip())
137 return tuple(forms)
140_DOCS_RE = re.compile(r"\b(doc|docs|documentation|readme|changelog)\b", re.IGNORECASE)
141_TESTS_RE = re.compile(r"\b(test|tests|coverage|ci|lint)\b", re.IGNORECASE)
142_BLOCKED_LABELS = frozenset({"blocked", "status:blocked", "needs-dependency"})
143_RISK_RE = re.compile(
144 r"\b(security|release|migration|schema|api|breaking|billing|secret|credential|ci|"
145 r"production|compatibility)\b",
146 re.IGNORECASE,
147)
150@dataclass(frozen=True)
151class IssueContext:
152 """Issue text supplied by an adapter before code mutation starts."""
154 title: str | None = None
155 body: str | None = None
156 labels: tuple[str, ...] = ()
158 @property
159 def provided(self) -> bool:
160 return bool((self.title or "").strip() or (self.body or "").strip() or self.labels)
163def assess_issue(
164 *,
165 title: str | None = None,
166 body: str | None = None,
167 labels: tuple[str, ...] = (),
168) -> dict[str, Any]:
169 """Return a deterministic readiness record for an issue-like work item."""
170 context = IssueContext(title=title, body=body, labels=tuple(labels))
171 sections = _sections(context.body or "")
172 acceptance = _acceptance_criteria(sections)
173 objective = _objective(context.title, context.body, sections)
174 deliverable = _deliverable(sections, acceptance)
175 combined = " ".join(filter(None, (context.title, context.body, " ".join(context.labels))))
176 normalized_labels = tuple(label.strip().lower() for label in context.labels)
178 missing: list[str] = []
179 blockers: list[str] = []
180 questions: list[str] = []
182 if not objective:
183 missing.append("objective")
184 questions.append("What objective should this issue accomplish?")
185 if not deliverable:
186 missing.append("deliverable")
187 questions.append("What concrete deliverable should be produced?")
188 if not acceptance:
189 missing.append("acceptance_criteria")
190 questions.append("What acceptance criteria define done for this issue?")
192 status = READY
193 reason = "Issue has an objective, deliverable, and acceptance criteria."
194 out_of_scope_reason = _out_of_scope_reason(
195 title=context.title,
196 body=context.body,
197 labels=normalized_labels,
198 )
199 if out_of_scope_reason:
200 status = OUT_OF_SCOPE
201 reason = out_of_scope_reason
202 questions = []
203 elif _is_blocked(combined, normalized_labels):
204 status = BLOCKED
205 reason = "Issue declares a dependency or waiting condition."
206 blockers.append(_blocked_summary(combined))
207 questions.append("What dependency must clear before this issue can start?")
208 elif missing:
209 status = NEEDS_INPUT
210 reason = f"Missing required intake field(s): {', '.join(missing)}."
211 elif _AMBIGUOUS_RE.search(combined):
212 status = NEEDS_INPUT
213 reason = "Issue scope contains ambiguous or deferred wording."
214 missing.append("scope_clarity")
215 questions.append("Which exact scope should be implemented now?")
217 return {
218 "schema_version": "keel.issue-intake.v1",
219 "provided": context.provided,
220 "status": status,
221 "can_mutate_code": status == READY,
222 "work_block_policy": {
223 "skip_when_not_ready": status != READY,
224 "continue_with_next_ready_issue": status != READY,
225 "non_ready_statuses": [NEEDS_INPUT, BLOCKED, OUT_OF_SCOPE],
226 },
227 "reason": reason,
228 "objective": objective,
229 "deliverable": deliverable,
230 "acceptance_criteria": acceptance,
231 "risk_tier_inputs": _risk_inputs(combined, normalized_labels),
232 "required_docs_tests": _required_docs_tests(combined, acceptance),
233 "missing_info": missing,
234 "blockers": blockers,
235 "questions": _unique(questions)[:3],
236 "ledger_record": {
237 "readiness": status,
238 "mutation_allowed": status == READY,
239 "skip_reason": None if status == READY else reason,
240 "question_count": len(_unique(questions)[:3]),
241 },
242 }
245def _sections(body: str) -> dict[str, str]:
246 matches = list(_SECTION_RE.finditer(body))
247 if not matches:
248 return {}
249 sections: dict[str, str] = {}
250 for index, match in enumerate(matches):
251 title = _normalize_heading(match.group("title"))
252 start = match.end()
253 end = matches[index + 1].start() if index + 1 < len(matches) else len(body)
254 sections[title] = body[start:end].strip()
255 return sections
258def _normalize_heading(value: str) -> str:
259 return re.sub(r"[^a-z0-9]+", " ", value.lower()).strip()
262def _objective(title: str | None, body: str | None, sections: dict[str, str]) -> str | None:
263 for key in ("objective", "problem", "summary", "context"):
264 if value := _first_sentence_or_bullet(sections.get(key, "")):
265 return value
266 if body and not sections:
267 return _first_sentence_or_bullet(body)
268 return title.strip() if title and title.strip() else None
271def _deliverable(sections: dict[str, str], acceptance: list[str]) -> str | None:
272 del acceptance
273 for key in ("deliverable", "proposed direction", "proposal", "scope", "implementation"):
274 if value := _first_sentence_or_bullet(sections.get(key, "")):
275 return value
276 return None
279def _acceptance_criteria(sections: dict[str, str]) -> list[str]:
280 for key in (
281 "acceptance criteria",
282 "acceptance",
283 "definition of done",
284 "done when",
285 "dod",
286 ):
287 if key in sections:
288 bullets = _bullets(sections[key])
289 return bullets if bullets else _sentences(sections[key])
290 return []
293def _first_sentence_or_bullet(text: str) -> str | None:
294 bullets = _bullets(text)
295 if bullets:
296 return bullets[0]
297 sentences = _sentences(text)
298 return sentences[0] if sentences else None
301def _bullets(text: str) -> list[str]:
302 items: list[str] = []
303 for line in text.splitlines():
304 if match := _BULLET_RE.match(line):
305 item = match.group("text").strip()
306 if item:
307 items.append(item)
308 return items
311def _bullet_lines(text: str) -> list[str]:
312 """List items, markers removed — each a statement in its own right.
314 `_sentences` joins lines carrying no terminator, so a plain bullet above a
315 declaration (`- Discussed with the team` then `- This issue is out of scope.`) puts
316 the declaration mid-sentence, past the anchor. A list item is a statement whether or
317 not it ends in a full stop.
319 List items **only**. Every line was tried and refused prose the moment a wrap put the
320 phrase at a line start; block quotes were tried and refused it too, because Markdown
321 prefixes each continuation line of a quote with `>`. Neither shape carries a list
322 marker. The subject test is what makes this safe now — a bullet has to name the issue
323 to match anything at all.
324 """
325 return [item for line in text.splitlines() if (item := _bullet_text(line))]
328def _bullet_text(line: str) -> str:
329 match = _BULLET_RE.match(line)
330 return match.group("text").strip() if match else ""
333def _sentences(text: str) -> list[str]:
334 compact = " ".join(line.strip() for line in text.splitlines() if line.strip())
335 if not compact:
336 return []
337 parts = [part.strip() for part in re.split(r"(?<=[.!?])\s+", compact) if part.strip()]
338 return parts or [compact]
341def _is_scope_exclusion_heading(title: str) -> bool:
342 """Does this heading open a section that bounds the **change**?
344 The test is *starts with*, not equals. ``## Out of scope for v1``,
345 ``## Out of scope: mobile UI`` and ``## Non-goals for now`` are the same kind of
346 section as ``## Out of scope``, and an equality test refused the issues that wrote
347 them — #1168 coming back through the heading, which is now read as prose. A section
348 titled "Out of scope…" is a boundary whatever qualifies it.
350 A close-reason heading (``Not planned``, ``Decision``, ``Status``) is deliberately
351 not in the set: it names a status for the issue, and dropping it would silence the
352 statement intake exists to read.
353 """
354 normalised = _normalize_heading(title)
355 return any(
356 normalised == heading or normalised.startswith(heading + " ")
357 for heading in _SCOPE_EXCLUSION_HEADINGS
358 )
361def _scannable_chunks(body: str) -> list[tuple[bool, str]]:
362 """``body`` split into the pieces a declaration may hide in, boundaries removed.
364 A scope-exclusion section goes whole — heading and content. What remains comes back
365 per section, with the heading's own text as a chunk **separate from** its body:
366 inline, it glues onto the sentence below (`- works ## Decision Out of scope:
367 closing.`) and the declaration pattern is anchored at a sentence start; dropped, a
368 declaration written *in* a heading becomes invisible. Each gate round found one of
369 those halves.
370 """
371 matches = list(_SECTION_RE.finditer(body))
372 if not matches:
373 return [(False, body)]
374 chunks: list[tuple[bool, str]] = [(False, body[: matches[0].start()])]
375 for index, match in enumerate(matches):
376 # An exclusion section owns its own body and nothing else. Owning nested headings
377 # was tried and failed in both directions: `# Out of scope` over a run of `##`
378 # sections swallowed the `## Decision` that closed the issue, and every heuristic
379 # for telling a title level from a section level inverted on some real body. What
380 # makes the simple rule safe is the subject test above — a bullet under a boundary
381 # section saying `Out of scope: the mobile client` names no issue and matches
382 # nothing, so the section's *contents* no longer need hiding, only its own prose.
383 if _is_scope_exclusion_heading(match.group("title")):
384 continue
385 end = matches[index + 1].start() if index + 1 < len(matches) else len(body)
386 # Flagged as a heading: its text may carry a record label (`Decision —`) in front
387 # of the declaration. It does **not** get the short form — `## Not planned` over a
388 # list of features is a boundary section, and reading its title as a closure
389 # refused the issue while the identical bullets under `## Non-goals` passed.
390 chunks.append((True, match.group("title")))
391 chunks.append((False, body[match.end() : end]))
392 return [(is_heading, chunk) for is_heading, chunk in chunks if chunk.strip()]
395def _out_of_scope_reason(
396 *,
397 title: str | None,
398 body: str | None,
399 labels: tuple[str, ...],
400) -> str | None:
401 """Return the concrete scope declaration, ignoring structural exclusions.
403 Two independent filters, and each does one job:
405 * **Structurally**, a scope-exclusion section is removed whole. Its heading and its
406 bullets describe what the change leaves out; none of it says the issue is closed.
407 * **In what remains**, a declaration is a sentence whose *subject is the issue* —
408 ``this issue is out of scope``, opening the sentence. Not ``Out of scope: …``,
409 which is the same string as a boundary bullet; not a mention in the middle of a
410 sentence, which carves out a part rather than closing the whole. The short form
411 is the title's, where one line and no markdown make an anchor safe.
413 Everything left is searched, title and body alike. A maintainer writing
414 ``## Decision — this issue is out of scope; closing`` is declaring it closed, and
415 intake must not hand that to s2.
417 ``non-goal`` is deliberately **not** a declaration marker. It names a non-goal of
418 the change ("Non-goal: rewrite the parser"), which is a boundary like the section
419 it usually appears under, not a statement that the issue will not be done. The
420 ``## Non-goals`` heading is handled structurally above.
421 """
422 for label in labels:
423 if label in _OUT_OF_SCOPE_LABELS:
424 return f"Issue carries out-of-scope label: {label}."
426 # The title takes the short form: it names the issue's whole subject, with nothing
427 # before it and nothing after it to qualify.
428 if title and (heading := title.strip()):
429 if _OUT_OF_SCOPE_OPENER_RE.search(heading) or _OUT_OF_SCOPE_DECLARATION_RE.search(heading):
430 return f"Issue declares itself out of scope: {heading}"
432 #: `(text, may a leading label be stripped)`. A heading's `Decision —` prefix is a
433 #: record label; the same shape inside prose is a prepositional phrase (`For Windows:
434 #: this issue is not in scope.`), and stripping it turned a carve-out into a closure.
435 candidates: list[tuple[str, bool]] = []
436 if body:
437 for is_heading, chunk in _scannable_chunks(body):
438 candidates.extend((sentence, is_heading) for sentence in _sentences(chunk))
439 candidates.extend((line, False) for line in _bullet_lines(chunk))
441 for candidate, allow_label in candidates:
442 # Leading markdown removed first, so the anchor sees the sentence rather than the
443 # bullet in front of it.
444 if any(
445 _OUT_OF_SCOPE_DECLARATION_RE.search(form)
446 for form in _declaration_candidates(candidate, allow_label=allow_label)
447 ):
448 return f"Issue declares itself out of scope: {candidate.strip()}"
449 return None
452def _is_blocked(combined: str, labels: tuple[str, ...]) -> bool:
453 if not _BLOCKED_LABELS.isdisjoint(labels):
454 return True
456 compact = " ".join(line.strip() for line in combined.splitlines() if line.strip())
457 # Fast path: bypass expensive sentence iteration if no
458 # blocked regex match exists globally in normalized text
459 if not _BLOCKED_RE.search(compact):
460 return False
462 # ⚡ Bolt Optimization: Unroll any() generator to avoid generator overhead
463 for sentence in _sentences(compact):
464 if _is_actionable_blocker(sentence):
465 return True
466 return False
469def _blocked_summary(combined: str) -> str:
470 sentences = _sentences(combined)
471 for sentence in sentences:
472 if _is_actionable_blocker(sentence):
473 return sentence
474 return "Declared blocked dependency."
477def _is_actionable_blocker(text: str) -> bool:
478 return bool(_BLOCKED_RE.search(text)) and not bool(_NON_BLOCKING_DEPENDENCY_RE.search(text))
481def _risk_inputs(combined: str, labels: tuple[str, ...]) -> dict[str, Any]:
482 keywords = _unique(match.group(0).lower() for match in _RISK_RE.finditer(combined))
483 risk_labels = [label for label in labels if "risk" in label or label.startswith("tier")]
484 return {
485 "keywords": keywords,
486 "labels": risk_labels,
487 "has_high_risk_signal": bool(keywords or risk_labels),
488 }
491def _required_docs_tests(combined: str, acceptance: list[str]) -> dict[str, Any]:
492 text = " ".join([combined, *acceptance])
493 return {
494 "docs": "required" if _DOCS_RE.search(text) else "unspecified",
495 "tests": "required" if _TESTS_RE.search(text) else "unspecified",
496 }
499def _unique(items) -> list[str]:
500 seen = set()
501 result = []
502 for item in items:
503 v = str(item).strip()
504 if v and v not in seen:
505 seen.add(v)
506 result.append(v)
507 return result