Coverage for src/keel/capture.py: 100%
814 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-02 20:26 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-02 20:26 +0000
1"""Consumer-neutral post-merge capture contract and verification helpers."""
3from __future__ import annotations
5import json
6import os.path
7import posixpath
8import re
9from dataclasses import dataclass
10from pathlib import Path
11from typing import Any, Protocol
12from urllib.parse import quote
14# `workspace` imports nothing from this package's config layer, so naming it here
15# keeps the import graph acyclic — see `_HasPolicyPack` for why that matters.
16from . import workspace
17from . import yaml_helper as yaml
20class _HasPolicyPack(Protocol):
21 """Duck type for a loaded ``ProjectConfig``.
23 This module only reads ``policy_pack``. Naming ``config.ProjectConfig`` here
24 would import ``config``, and ``config`` already imports this module to
25 validate ``policy_pack.capture.learning``. CodeQL counts even a
26 ``TYPE_CHECKING`` import as that reverse edge — the cycle it reports as
27 ``py/cyclic-import``, because ``ProjectConfig`` is defined *after*
28 ``config``'s import of ``capture``.
29 """
31 policy_pack: Any
34CAPTURE_SCHEMA_VERSION = "keel.capture.v1"
35RECONCILE_SCHEMA_VERSION = "keel.capture-reconcile.v1"
36LEARNING_DECISION_SCHEMA_VERSION = "keel.capture-learning.v1"
37MARKER_PREFIX = "compound-learning"
38STATUSES = ("applied", "deferred", "skipped")
39SKIP_REASONS = (
40 "dry-run",
41 "deferred",
42 "merge-failed",
43 "recursion-guard",
44 "capability-unavailable",
45 "no-policy",
46)
47LEARNING_DECISIONS = ("create-learning", "marker-only", "defer", "duplicate")
49_MARKER_RE = re.compile(
50 r"^compound-learning:\s+pr=(?P<pr>[1-9][0-9]*)\s+status="
51 r"(?P<status>applied|deferred|skipped(?::[a-z0-9-]+)?)$"
52)
55class CaptureError(ValueError):
56 """Raised when a capture marker or capture record is invalid."""
59@dataclass(frozen=True)
60class CaptureMarker:
61 """One stable capture marker emitted after a merged PR."""
63 pr_number: int
64 status: str
65 reason: str | None = None
67 def as_text(self) -> str:
68 return marker_text(
69 pr_number=self.pr_number,
70 status=self.status,
71 reason=self.reason,
72 )
74 def as_dict(self) -> dict[str, Any]:
75 return {
76 "schema_version": CAPTURE_SCHEMA_VERSION,
77 "prefix": MARKER_PREFIX,
78 "pr": self.pr_number,
79 "status": self.status,
80 "reason": self.reason,
81 "text": self.as_text(),
82 }
85def contract_as_dict(config: _HasPolicyPack | None = None) -> dict[str, Any]:
86 """Return the stable capture contract consumed by adapters and verifiers."""
87 capture_policy = _capture_policy(config)
88 # **The sink core will actually use**, not merely one written down. A dormant
89 # `sink:` block under `capture.enabled: false` or `mode: marker-only` published
90 # `project_destination: "sink"` — telling an adapter core would write — while
91 # `learning_sink_writes` refused, so neither wrote and the contract had promised
92 # one of them would. Third reader of the same question; they all ask it here.
93 sink_policy = learning_sink_policy(config) if capture_hook_enabled(config) else None
94 return {
95 "schema_version": CAPTURE_SCHEMA_VERSION,
96 "marker": {
97 "prefix": MARKER_PREFIX,
98 "format": "compound-learning: pr=<N> status=<applied|deferred|skipped:reason>",
99 "statuses": list(STATUSES),
100 "skip_reasons": list(SKIP_REASONS),
101 "required_after_merged_pr": True,
102 },
103 "extension_slots": ["capture", "post-merge"],
104 "policy_source": "policy_pack.capture + capture/post-merge extensions",
105 "policy_enabled": bool(capture_policy.get("enabled", False)),
106 "policy_mode": capture_policy.get("mode", "extension"),
107 "recursion_guard": {
108 "enabled": True,
109 "reason": "recursion-guard",
110 "never_capture_capture_work": True,
111 },
112 "fail_soft": {
113 "enabled": True,
114 "merge_revert_on_capture_failure": False,
115 "failure_marker": "skipped:capability-unavailable",
116 },
117 "durable_artifacts": {
118 "requires_redaction": True,
119 "redaction_contract": "run_ledger.capture_redaction",
120 "core_destination": "run-ledger",
121 # `extension-owned` was the whole truth until keel shipped a writer.
122 # A project that configures `learning.sink` has **core** writing the
123 # file and filling `capture.artifact`, and an adapter that read this
124 # block and wrote its own would have had it overwritten (#1154).
125 "project_destination": "sink" if sink_policy is not None else "extension-owned",
126 # Defaulted the way `learning_sink_plan` defaults it. Read without the
127 # fallback, a sink that did not spell out its `kind` — which the schema,
128 # the docs and the validator all allow — published
129 # `{project_destination: sink, sink: None}`: a contract disagreeing with
130 # itself about the writer it had just named.
131 "sink": (
132 sink_policy.get("kind", LEARNING_SINK_KINDS[0]) if sink_policy is not None else None
133 ),
134 # **Whether the file lands in the working tree, and so has to be
135 # committed.** keel writes it and stops there. A relative sink path is
136 # inside the repository, where an uncommitted file is one the next
137 # worktree — cut from `origin/<base>` — and every CI runner never see:
138 # keel would be writing a learning and then throwing it away, which is
139 # the failure classifying `.keel/learning` as committed was for. An
140 # absolute or `~` path is outside the checkout and git never sees it.
141 "commit_required": learning_sink_in_worktree(config),
142 # **How a `commit_required` artifact reaches the base branch (#1163).**
143 # `commit_required` answered *whether* the file has to be committed and
144 # nothing answered *how*, so keel wrote a learning on every merge and
145 # threw it away: s2 cuts the next worktree from `origin/<base>`, every CI
146 # runner clones fresh, and s10's pre-clean deletes the worktree outright.
147 # The recipe cannot be "switch to the base branch and commit" — s2,
148 # `overnight` and `swarm` all run inside a worktree while the primary
149 # checkout holds the base branch, so `git switch <base>` there exits 128
150 # with *'<base>' is already used by worktree*. `keel capture-land` builds
151 # the commit with plumbing against `origin/<base>` and never checks it
152 # out, which is why it is topology-independent.
153 "land_command": "keel capture-land" if learning_sink_in_worktree(config) else None,
154 },
155 "learning_quality": learning_quality_contract_as_dict(config),
156 "learning_retrieval": learning_retrieval_contract_as_dict(config),
157 "session_end_verifier": {
158 "primitive": "capture.verify_session",
159 "cli": "keel capture-verify",
160 "missing_marker_status": "missing",
161 "invalid_marker_status": "invalid",
162 },
163 "reconcile": {
164 "schema_version": RECONCILE_SCHEMA_VERSION,
165 "primitive": "capture.reconcile_session",
166 "cli": "keel capture-reconcile",
167 "idempotent": True,
168 "never_reopens_implementation": True,
169 "never_pushes_code": True,
170 "never_merges_prs": True,
171 "actions": [
172 "emit-capture-marker",
173 "run-capture-extension",
174 "post-closure-summary",
175 "close-linked-issue",
176 "record-skip",
177 ],
178 },
179 }
182def learning_retrieval_contract_as_dict(
183 config: _HasPolicyPack | None = None,
184) -> dict[str, Any]:
185 """The read side of capture, declared so an adapter knows the section exists.
187 Policy only — where this project reads learnings from and how many a brief
188 carries. What was actually *found* is measured per run and travels on the ship
189 contract, because reading a directory is I/O and this contract is pure.
190 """
191 return {
192 "schema_version": LEARNING_RETRIEVAL_SCHEMA_VERSION,
193 "policy_source": "policy_pack.capture.learning.source",
194 # As **configured**, not as resolved: this contract is pure and has no
195 # `{repo}` to expand with, so a resolved list would report an empty
196 # `sources` for a templated setting that reads perfectly well at run time.
197 "sources": learning_source_entries(config),
198 "limit": DEFAULT_LEARNING_RETRIEVAL_LIMIT,
199 "heading": LEARNING_BRIEF_HEADING,
200 "briefs": ["implement", "review"],
201 "reader": "capture.retrieve_relevant_learnings",
202 "ledger_field": "capture.retrieved",
203 "silent_when_empty": True,
204 }
207def learning_quality_contract_as_dict(config: _HasPolicyPack | None = None) -> dict[str, Any]:
208 """Return the consumer-neutral durable-learning quality contract."""
209 policy = _learning_policy(config)
210 dedupe = policy.get("dedupe") if isinstance(policy.get("dedupe"), dict) else {}
211 return {
212 "schema_version": LEARNING_DECISION_SCHEMA_VERSION,
213 "decisions": list(LEARNING_DECISIONS),
214 "policy_source": "policy_pack.capture.learning",
215 "policy_enabled": bool(policy.get("enabled", False)),
216 "policy_mode": policy.get("mode", "policy-unavailable"),
217 "default_decision": "marker-only",
218 "default_reason": "policy-unavailable",
219 "marker_required_for_every_merge": True,
220 "durable_learning_optional": True,
221 "dedupe": {
222 "enabled": bool(dedupe.get("enabled", True)),
223 "fingerprint": "sha256(normalized title + labels + changed files)",
224 "matching": "stable fingerprint plus configured matching rules",
225 },
226 "ledger_field": "capture.learning",
227 "closure_summary_field": "Capture",
228 }
231def marker_text(*, pr_number: int, status: str, reason: str | None = None) -> str:
232 """Render one stable capture marker."""
233 marker = build_marker(pr_number=pr_number, status=status, reason=reason)
234 suffix = marker.status if marker.reason is None else f"{marker.status}:{marker.reason}"
235 return f"{MARKER_PREFIX}: pr={marker.pr_number} status={suffix}"
238def build_marker(*, pr_number: int, status: str, reason: str | None = None) -> CaptureMarker:
239 """Validate and build a capture marker."""
240 if pr_number <= 0:
241 raise CaptureError("capture marker requires a positive PR number")
242 status, reason = normalize_status(status, reason)
243 return CaptureMarker(pr_number=pr_number, status=status, reason=reason)
246def normalize_status(status: str | None, reason: str | None = None) -> tuple[str, str | None]:
247 """Normalize ``skipped:<reason>`` into a structured status and reason."""
248 if not status:
249 raise CaptureError("capture status is required")
250 raw = status.strip()
251 if raw.startswith("skipped:"):
252 raw, embedded_reason = raw.split(":", 1)
253 reason = embedded_reason
254 if raw not in STATUSES:
255 raise CaptureError(f"unsupported capture status: {status}")
256 clean_reason = reason.strip() if isinstance(reason, str) and reason.strip() else None
257 if raw == "skipped":
258 if clean_reason not in SKIP_REASONS:
259 raise CaptureError("skipped capture requires an allowed skip reason")
260 else:
261 clean_reason = None
262 return raw, clean_reason
265def parse_marker(text: str) -> CaptureMarker:
266 """Parse a stable marker string into structured data."""
267 match = _MARKER_RE.match(text.strip())
268 if not match:
269 raise CaptureError("invalid capture marker")
270 status_text = match.group("status")
271 status, reason = normalize_status(status_text)
272 return CaptureMarker(
273 pr_number=int(match.group("pr")),
274 status=status,
275 reason=reason,
276 )
279def record_marker(
280 *,
281 pr_number: int | None,
282 status: str | None,
283 reason: str | None = None,
284 artifact: str | None = None,
285 title: str | None = None,
286 labels: list[str] | tuple[str, ...] = (),
287 changed_files: list[str] | tuple[str, ...] = (),
288 existing_records: list[dict[str, Any]] | tuple[dict[str, Any], ...] = (),
289 config: _HasPolicyPack | None = None,
290 not_run: bool = False,
291 retrieved: list[str] | tuple[str, ...] = (),
292) -> dict[str, Any]:
293 """Build the capture block stored in a ship run ledger record.
295 ``artifact`` is an optional reference (path or content hash) to the durable
296 capture artifact. It is the proof that an ``applied`` capture actually
297 produced something; capture reconcile treats ``applied`` with no artifact as
298 a finding. ``deferred``/``skipped`` need no artifact.
300 ``retrieved`` is the fingerprints of the learnings this run put in front of the
301 implementer and the reviewers (#1155). Recording them is what lets a later run
302 tell a lesson nobody had from one that was surfaced and still not applied —
303 the difference between a retrieval gap and a discipline gap.
305 ``not_run`` marks a record whose run never reached capture, so a reader can
306 tell it apart from one that reached capture and lost its marker. Both carry
307 ``marker: None``, and without the flag a capture-health reader can only
308 assume the second and report a gap that is not there (#945). It is a
309 *declaration* by the operator, deliberately: inferring it from the null would
310 reclassify every genuinely missing marker as "never attempted", which is the
311 fail-open this field exists to avoid.
312 """
313 clean_artifact = artifact.strip() if isinstance(artifact, str) and artifact.strip() else None
314 clean_retrieved = _strings(retrieved)
315 if status is None:
316 return {
317 "schema_version": CAPTURE_SCHEMA_VERSION,
318 "status": None,
319 "reason": reason,
320 "marker_reason": None,
321 "marker": None,
322 "not_run": not_run,
323 "artifact": clean_artifact,
324 "artifact_scope": artifact_scope(clean_artifact, config),
325 "retrieved": clean_retrieved,
326 "fail_soft": True,
327 "learning": learning_decision(
328 title=title,
329 labels=labels,
330 changed_files=changed_files,
331 capture_status=None,
332 capture_reason=reason,
333 existing_records=existing_records,
334 config=config,
335 ),
336 }
337 learning = learning_decision(
338 title=title,
339 labels=labels,
340 changed_files=changed_files,
341 capture_status=status,
342 capture_reason=reason,
343 existing_records=existing_records,
344 config=config,
345 )
346 marker_reason = _marker_reason(status, reason)
347 if pr_number is None:
348 clean_status, clean_marker_reason = normalize_status(status, marker_reason)
349 return {
350 "schema_version": CAPTURE_SCHEMA_VERSION,
351 "status": clean_status,
352 "reason": reason,
353 "marker_reason": clean_marker_reason,
354 "marker": None,
355 "artifact": clean_artifact,
356 "artifact_scope": artifact_scope(clean_artifact, config),
357 "retrieved": clean_retrieved,
358 "fail_soft": True,
359 "learning": learning,
360 }
361 marker = build_marker(pr_number=pr_number, status=status, reason=marker_reason)
362 return {
363 "schema_version": CAPTURE_SCHEMA_VERSION,
364 "status": marker.status,
365 "reason": reason,
366 "marker_reason": marker.reason,
367 "marker": marker.as_text(),
368 "artifact": clean_artifact,
369 "artifact_scope": artifact_scope(clean_artifact, config),
370 "retrieved": clean_retrieved,
371 "fail_soft": True,
372 "learning": learning,
373 }
376def learning_decision(
377 *,
378 title: str | None = None,
379 labels: list[str] | tuple[str, ...] = (),
380 changed_files: list[str] | tuple[str, ...] = (),
381 capture_status: str | None = None,
382 capture_reason: str | None = None,
383 existing_records: list[dict[str, Any]] | tuple[dict[str, Any], ...] = (),
384 config: _HasPolicyPack | None = None,
385) -> dict[str, Any]:
386 """Classify whether a merged PR deserves a durable learning artifact.
388 The marker is mandatory and independent from this decision. Durable learning is
389 optional, policy-driven, and deduped by a stable fingerprint so routine merges can stay
390 marker-only without losing auditability.
391 """
392 policy = _learning_policy(config)
393 fingerprint = learning_fingerprint(
394 title=title,
395 labels=labels,
396 changed_files=changed_files,
397 )
398 if _learning_dedupe_enabled(policy):
399 duplicate_of = _duplicate_learning_fingerprint(fingerprint, existing_records)
400 if duplicate_of is not None:
401 return _learning_result(
402 "duplicate",
403 reason="duplicate-learning",
404 fingerprint=fingerprint,
405 duplicate_of=duplicate_of,
406 policy=policy,
407 )
408 # **The same parent pair the writer consults.** Gating only the write left the
409 # record claiming `create-learning` with `durable_artifact: true` for a run that
410 # produced nothing — the write/record disagreement this feature already treats
411 # as load-bearing one level down, inverted. One predicate answers both.
412 if not policy.get("enabled") or not capture_hook_enabled(config):
413 return _learning_result(
414 "marker-only",
415 reason="policy-unavailable",
416 fingerprint=fingerprint,
417 policy=policy,
418 )
419 mode = policy.get("mode", "marker-only")
420 if mode == "create-learning":
421 # **Anything but `applied`**, not just `skipped:*`. `deferred` fell through
422 # and answered `create-learning` with `durable_artifact: true`, while
423 # `learning_sink_writes` refuses every status but `applied` — the same
424 # write/record disagreement, one status over.
425 if capture_status != "applied":
426 return _learning_result(
427 "marker-only",
428 reason=(
429 "capture-skipped"
430 if (capture_status or "").startswith("skipped")
431 else "capture-not-applied"
432 ),
433 fingerprint=fingerprint,
434 policy=policy,
435 )
436 return _learning_result(
437 "create-learning",
438 reason=_policy_reason(policy, "policy-requested-learning"),
439 fingerprint=fingerprint,
440 policy=policy,
441 )
442 if mode == "defer":
443 return _learning_result(
444 "defer",
445 reason=_policy_reason(policy, "policy-deferred"),
446 fingerprint=fingerprint,
447 policy=policy,
448 )
449 return _learning_result(
450 "marker-only",
451 reason=_policy_reason(policy, "marker-only-policy"),
452 fingerprint=fingerprint,
453 policy=policy,
454 )
457def capture_hook_enabled(config: _HasPolicyPack | None) -> bool:
458 """Whether this project runs a post-merge **content hook** at all.
460 `policy_pack.capture.enabled` is the project saying it intends to; `mode:
461 marker-only` records the core marker *without* one, which is the schema's own
462 wording. The learning sink is that hook, and so is the decision that says a
463 durable artifact is wanted — both ask here, because a writer and a record that
464 answer this differently is the disagreement this feature keeps producing.
465 `_reconcile_marker_decision` reads the same pair for `skipped:no-policy`.
466 """
467 policy = _capture_policy(config)
468 return bool(policy.get("enabled")) and policy.get("mode", "extension") == "extension"
471def learning_fingerprint(
472 *,
473 title: str | None = None,
474 labels: list[str] | tuple[str, ...] = (),
475 changed_files: list[str] | tuple[str, ...] = (),
476) -> str:
477 """Return a stable, consumer-neutral dedupe fingerprint for learning candidates."""
478 import hashlib
480 payload = {
481 "title": _normalize_text(title),
482 "labels": sorted(_normalize_text(label) for label in _strings(labels)),
483 "changed_files": sorted(_normalize_path(path) for path in _strings(changed_files)),
484 }
485 encoded = json.dumps(payload, sort_keys=True, separators=(",", ":"))
486 return hashlib.sha256(encoded.encode("utf-8")).hexdigest()
489def verify_session(
490 records: list[dict[str, Any]],
491 merged_prs: list[int] | tuple[int, ...],
492) -> dict[str, Any]:
493 """Verify that each merged PR has an applied/deferred/allowed-skip capture marker."""
494 results = [_verify_pr(records, pr) for pr in merged_prs]
495 missing = [item for item in results if item["status"] == "missing"]
496 invalid = [item for item in results if item["status"] == "invalid"]
497 status = "complete" if not missing and not invalid else "incomplete"
498 return {
499 "schema_version": CAPTURE_SCHEMA_VERSION,
500 "status": status,
501 "expected_prs": list(merged_prs),
502 "results": results,
503 "summary": {
504 "ok": sum(1 for item in results if item["ok"]),
505 "missing": len(missing),
506 "invalid": len(invalid),
507 },
508 }
511def reconcile_session(
512 records: list[dict[str, Any]],
513 merged_prs: list[int | dict[str, Any]] | tuple[int | dict[str, Any], ...],
514 *,
515 config: _HasPolicyPack | None = None,
516 capture_capability_available: bool = False,
517) -> dict[str, Any]:
518 """Plan idempotent post-merge reconciliation actions for capture gaps.
520 The returned plan is pure data. It never writes ledger records, comments, issues, git
521 state, or PR state; adapters may apply the listed actions after their own transport and
522 consent checks. This keeps reconcile recovery deterministic and safe to run repeatedly.
523 """
524 items = [_merged_pr_info(item) for item in merged_prs]
525 results = [
526 _reconcile_pr(
527 records,
528 item,
529 config=config,
530 capture_capability_available=capture_capability_available,
531 )
532 for item in items
533 ]
534 actionable = [item for item in results if item["actions"]]
535 blocked = [item for item in results if item["status"] in {"invalid", "ambiguous"}]
536 complete = [item for item in results if item["status"] == "complete"]
537 status = "blocked" if blocked else "actionable" if actionable else "complete"
538 return {
539 "schema_version": RECONCILE_SCHEMA_VERSION,
540 "status": status,
541 "dry_run_safe": True,
542 "idempotent": True,
543 "no_code_mutations": True,
544 "expected_prs": [item["number"] for item in items],
545 "results": results,
546 "summary": {
547 "complete": len(complete),
548 "actionable": len(actionable),
549 "blocked": len(blocked),
550 },
551 }
554def recursion_guard(
555 *,
556 title: str | None = None,
557 labels: list[str] | tuple[str, ...] = (),
558 changed_files: list[str] | tuple[str, ...] = (),
559) -> bool:
560 """Return true when capture should skip to avoid capture-on-capture recursion."""
561 # ⚡ Bolt: Return early if title matches to avoid expensive loops on labels and files
562 if title and "capture" in title.lower():
563 return True
565 # ⚡ Bolt: Return early if label matches to avoid expensive loops on files
566 for label in labels:
567 if label.lower() == "capture":
568 return True
570 # ⚡ Bolt: Avoid generator overhead with explicit loop for paths (~90x speedup when early match)
571 for path in changed_files:
572 p = path.lower()
573 if "/capture" in p or p.endswith("capture.py"):
574 return True
576 return False
579def _merged_pr_info(item: int | dict[str, Any]) -> dict[str, Any]:
580 if isinstance(item, int):
581 return {
582 "number": item,
583 "title": None,
584 "labels": [],
585 "changed_files": [],
586 "issue_numbers": [],
587 }
588 number = item.get("number")
589 if not isinstance(number, int) or number <= 0:
590 raise CaptureError("merged PR entry requires a positive number")
591 return {
592 "number": number,
593 "title": item.get("title") if isinstance(item.get("title"), str) else None,
594 "labels": _strings(item.get("labels")),
595 "changed_files": _strings(item.get("changed_files")),
596 "issue_numbers": _positive_ints(item.get("issue_numbers")),
597 }
600def _reconcile_pr(
601 records: list[dict[str, Any]],
602 item: dict[str, Any],
603 *,
604 config: _HasPolicyPack | None,
605 capture_capability_available: bool,
606) -> dict[str, Any]:
607 pr_number = item["number"]
608 verification = _verify_pr(records, pr_number)
609 issue_numbers = _linked_issue_numbers(records, item)
610 if len(issue_numbers) > 1:
611 return _reconcile_result(
612 pr_number,
613 status="ambiguous",
614 reason="multiple linked issues found for merged PR",
615 verification=verification,
616 issue_numbers=issue_numbers,
617 blocked=True,
618 )
619 if verification["ok"]:
620 if len(issue_numbers) == 1:
621 return _reconcile_result(
622 pr_number,
623 status="actionable",
624 reason="capture marker already present; linked issue closeout can be reconciled",
625 verification=verification,
626 issue_numbers=issue_numbers,
627 marker=verification["marker"],
628 actions=[
629 _action(
630 "close-linked-issue", pr_number=pr_number, issue_number=issue_numbers[0]
631 ),
632 ],
633 )
634 return _reconcile_result(
635 pr_number,
636 status="complete",
637 reason="capture marker already present",
638 verification=verification,
639 )
640 if verification["status"] == "invalid":
641 return _reconcile_result(
642 pr_number,
643 status="invalid",
644 reason=verification["reason"],
645 verification=verification,
646 issue_numbers=issue_numbers,
647 blocked=True,
648 )
649 marker_status, marker_reason, reason = _reconcile_marker_decision(
650 item,
651 config=config,
652 capture_capability_available=capture_capability_available,
653 )
654 marker = marker_text(
655 pr_number=pr_number,
656 status=marker_status,
657 reason=marker_reason,
658 )
659 actions = [
660 _action(
661 "emit-capture-marker",
662 pr_number=pr_number,
663 marker=marker,
664 status=marker_status,
665 reason=marker_reason,
666 ),
667 _action("post-closure-summary", pr_number=pr_number),
668 ]
669 if marker_status == "deferred":
670 actions.insert(0, _action("run-capture-extension", pr_number=pr_number))
671 if marker_status == "skipped":
672 actions.append(_action("record-skip", pr_number=pr_number, reason=marker_reason))
673 if len(issue_numbers) == 1:
674 actions.append(
675 _action("close-linked-issue", pr_number=pr_number, issue_number=issue_numbers[0])
676 )
677 return _reconcile_result(
678 pr_number,
679 status="actionable",
680 reason=reason,
681 verification=verification,
682 issue_numbers=issue_numbers,
683 marker=marker,
684 actions=actions,
685 )
688def _marker_of(record: dict[str, Any]) -> str | None:
689 """This record's capture marker, or ``None`` when it carries none."""
690 capture_block = record.get("capture")
691 marker = capture_block.get("marker") if isinstance(capture_block, dict) else None
692 return marker if marker else None
695def _record_head(record: dict[str, Any]) -> str | None:
696 """The head a ship_run record was written for, or ``None`` when it names none."""
697 git = record.get("git")
698 head = git.get("head_sha") if isinstance(git, dict) else None
699 return head.strip() if isinstance(head, str) and head.strip() else None
702def _verify_pr(records: list[dict[str, Any]], pr_number: int) -> dict[str, Any]:
703 candidates = [
704 record
705 for record in records
706 if record.get("record_type") == "ship_run"
707 and (record.get("pull_request") or {}).get("number") == pr_number
708 ]
709 # Markers are counted on one head, not across every head this pull request
710 # ever had (#1157). `existing_capture_marker` refuses a second marker per
711 # (pull request, head); counting per pull request here would move the same
712 # deadlock one step later — a pull request whose superseded head left a
713 # marker would fail verification for carrying two, and the only exit would
714 # again be editing an append-only ledger by hand.
715 #
716 # The head is taken from the **last record that carries a marker**, not from
717 # the last record. `ledger.latest_ship_run_for_pr` documents why the latter
718 # is the wrong proxy, and the concrete case is #945's: a run that never
719 # reached capture re-records gates on a new head and writes `not_run` with no
720 # marker. Reading the head off that row would look past a real applied marker
721 # on the merged head and report the capture missing, and `reconcile_session`
722 # would then plan to emit a marker that already exists. A row with no marker
723 # cannot move the head the markers are counted on, because it is not evidence
724 # about capture at all.
725 marked = [record for record in candidates if _marker_of(record)]
726 merged_head = _record_head(marked[-1]) if marked else None
727 markers = [
728 marker
729 for record in marked
730 if _record_head(record) == merged_head and (marker := _marker_of(record))
731 ]
732 if len(markers) > 1:
733 return {
734 "pr": pr_number,
735 "ok": False,
736 "status": "invalid",
737 "reason": "multiple capture markers found for merged PR",
738 "marker": markers[-1],
739 "marker_count": len(markers),
740 }
741 for marker in markers:
742 try:
743 parsed = parse_marker(marker)
744 except CaptureError as exc:
745 return {
746 "pr": pr_number,
747 "ok": False,
748 "status": "invalid",
749 "reason": str(exc),
750 "marker": marker,
751 }
752 if parsed.pr_number != pr_number:
753 return {
754 "pr": pr_number,
755 "ok": False,
756 "status": "invalid",
757 "reason": "marker PR does not match ledger PR",
758 "marker": marker,
759 }
760 return {
761 "pr": pr_number,
762 "ok": True,
763 "status": parsed.status,
764 "reason": parsed.reason,
765 "marker": marker,
766 }
767 return {
768 "pr": pr_number,
769 "ok": False,
770 "status": "missing",
771 "reason": "no capture marker found for merged PR",
772 "marker": None,
773 }
776def _reconcile_result(
777 pr_number: int,
778 *,
779 status: str,
780 reason: str,
781 verification: dict[str, Any],
782 issue_numbers: list[int] | None = None,
783 marker: str | None = None,
784 actions: list[dict[str, Any]] | None = None,
785 blocked: bool = False,
786) -> dict[str, Any]:
787 return {
788 "pr": pr_number,
789 "status": status,
790 "reason": reason,
791 "verification_status": verification["status"],
792 "blocked": blocked,
793 "issue_numbers": list(issue_numbers or ()),
794 "marker": marker,
795 "actions": list(actions or ()),
796 }
799def _reconcile_marker_decision(
800 item: dict[str, Any],
801 *,
802 config: _HasPolicyPack | None,
803 capture_capability_available: bool,
804) -> tuple[str, str | None, str]:
805 if recursion_guard(
806 title=item["title"],
807 labels=item["labels"],
808 changed_files=item["changed_files"],
809 ):
810 return "skipped", "recursion-guard", "capture recursion guard matched"
811 policy = _capture_policy(config)
812 if policy.get("enabled") and policy.get("mode", "extension") == "marker-only":
813 return "applied", None, "marker-only capture policy configured"
814 if policy.get("enabled") and policy.get("mode", "extension") == "extension":
815 if capture_capability_available:
816 return "deferred", None, "capture extension can be rerun"
817 return "skipped", "capability-unavailable", "capture extension capability unavailable"
818 return "skipped", "no-policy", "no capture policy configured"
821def _linked_issue_numbers(records: list[dict[str, Any]], item: dict[str, Any]) -> list[int]:
822 numbers = set(item["issue_numbers"])
823 pr_number = item["number"]
824 for record in records:
825 if record.get("record_type") != "ship_run":
826 continue
827 if (record.get("pull_request") or {}).get("number") != pr_number:
828 continue
829 issue_number = (record.get("issue") or {}).get("number")
830 if isinstance(issue_number, int) and issue_number > 0:
831 numbers.add(issue_number)
832 return sorted(numbers)
835def _action(
836 action_type: str,
837 *,
838 pr_number: int,
839 marker: str | None = None,
840 status: str | None = None,
841 reason: str | None = None,
842 issue_number: int | None = None,
843) -> dict[str, Any]:
844 action = {
845 "type": action_type,
846 "pr": pr_number,
847 "idempotency_key": f"{action_type}:pr-{pr_number}",
848 }
849 if marker is not None:
850 action["marker"] = marker
851 if status is not None:
852 action["status"] = status
853 if reason is not None:
854 action["reason"] = reason
855 if issue_number is not None:
856 action["issue"] = issue_number
857 action["idempotency_key"] = f"{action_type}:issue-{issue_number}:pr-{pr_number}"
858 return action
861def _capture_policy(config: _HasPolicyPack | None) -> dict[str, Any]:
862 if config is None or not isinstance(config.policy_pack, dict):
863 return {}
864 policy = config.policy_pack.get("capture")
865 return policy if isinstance(policy, dict) else {}
868def _learning_policy(config: _HasPolicyPack | None) -> dict[str, Any]:
869 policy = _capture_policy(config)
870 learning = policy.get("learning") if isinstance(policy, dict) else None
871 return learning if isinstance(learning, dict) else {}
874def _learning_dedupe_enabled(policy: dict[str, Any]) -> bool:
875 dedupe = policy.get("dedupe")
876 if not isinstance(dedupe, dict):
877 return True
878 return bool(dedupe.get("enabled", True))
881def _marker_reason(status: str, reason: str | None) -> str | None:
882 raw = status.strip()
883 if raw.startswith("skipped:"):
884 return None
885 if raw != "skipped":
886 return None
887 if reason in SKIP_REASONS:
888 return reason
889 return "no-policy"
892def _strings(value: Any) -> list[str]:
893 if not isinstance(value, list | tuple):
894 return []
895 return [item for item in value if isinstance(item, str)]
898def _positive_ints(value: Any) -> list[int]:
899 if not isinstance(value, list | tuple):
900 return []
901 return [item for item in value if isinstance(item, int) and item > 0]
904def _duplicate_learning_fingerprint(
905 fingerprint: str,
906 records: list[dict[str, Any]] | tuple[dict[str, Any], ...],
907) -> str | None:
908 for record in records:
909 if not isinstance(record, dict):
910 continue
911 capture_block = record.get("capture")
912 learning = capture_block.get("learning") if isinstance(capture_block, dict) else None
913 if not isinstance(learning, dict):
914 continue
915 if learning.get("fingerprint") != fingerprint:
916 continue
917 decision = learning.get("decision")
918 if decision in {"create-learning", "duplicate"}:
919 return str(record.get("run_id") or (record.get("pull_request") or {}).get("number"))
920 return None
923def _learning_result(
924 decision: str,
925 *,
926 reason: str,
927 fingerprint: str,
928 policy: dict[str, Any],
929 duplicate_of: str | None = None,
930) -> dict[str, Any]:
931 if decision not in LEARNING_DECISIONS:
932 raise CaptureError(f"unsupported learning decision: {decision}")
933 result = {
934 "schema_version": LEARNING_DECISION_SCHEMA_VERSION,
935 "decision": decision,
936 "reason": reason,
937 "fingerprint": fingerprint,
938 "policy_source": "policy_pack.capture.learning",
939 "policy_mode": policy.get("mode", "policy-unavailable"),
940 "durable_artifact": decision == "create-learning",
941 }
942 if duplicate_of is not None:
943 result["duplicate_of"] = duplicate_of
944 return result
947def _policy_reason(policy: dict[str, Any], default: str) -> str:
948 reason = policy.get("reason")
949 return reason.strip() if isinstance(reason, str) and reason.strip() else default
952def _normalize_text(value: str | None) -> str:
953 return " ".join(value.lower().split()) if isinstance(value, str) else ""
956def _normalize_path(value: str) -> str:
957 return "/".join(value.strip().lower().replace("\\", "/").split("/"))
960#: The one sink kind this issue ships. A directory of Markdown with stable
961#: frontmatter is the whole contract — no vault format, no wikilinks, no plugin
962#: API — so a project can point it at whatever reads Markdown and keel never
963#: learns what that is.
964LEARNING_SINK_KINDS = ("markdown-dir",)
966#: Where a sink writes when a project configures capture but names no path. The
967#: existing `.keel/learning/` convention, so turning the sink on changes where
968#: files appear only for a project that asked it to.
969DEFAULT_LEARNING_SINK_PATH = ".keel/learning"
971#: **The fingerprint is in the name because the fingerprint is the identity.** Date,
972#: PR and slug do not distinguish two lessons: a second `create-learning` run on the
973#: same PR the same day — different labels, different files, a different lesson —
974#: resolved to the same path and `os.replace` destroyed the first, leaving the
975#: earlier ledger record pointing at a document that says something else. The dedupe
976#: cannot help; it suppresses *identical* fingerprints, and these differ.
977DEFAULT_LEARNING_SINK_FILENAME = "{date}-pr{pr}-{slug}-{fingerprint}.md"
979#: The suffixes the reader opens. Named because the *writer* is validated against
980#: them: a sink filename ending `.markdown`, or in nothing at all, was written
981#: successfully into the sink and then skipped by the only thing that reads it —
982#: the writer/reader disagreement this feature exists inside, arriving through a
983#: template the validator accepted.
984LEARNING_READ_SUFFIXES = (".md", ".json", ".txt")
986#: How much of the fingerprint a filename carries. A sha256 prefix this long
987#: distinguishes every learning a project will ever write without making the name
988#: unreadable.
989LEARNING_FINGERPRINT_SLICE = 12
991#: The document's fourth section (#1166): one entry per ``changed_files`` path, written
992#: as a Markdown link so a **link-following** reader — a knowledge-graph builder, a
993#: wiki — gets the file ↔ lesson edge. The front-matter list is a string to such a
994#: reader; keel's own reader keeps matching on the list, which is why it stays.
995LEARNING_FILES_HEADING = "## Files"
996LEARNING_NO_FILES = "_No files recorded._"
998#: Characters that end, escape, or open something inside a CommonMark link *text*:
999#: the bracket pair and the backslash; the angle brackets that would start an autolink
1000#: or raw HTML from a file name (a merged PR chooses those bytes); the emphasis and
1001#: strikethrough delimiters that turned ``__init__.py`` — the most common Python file
1002#: name — into bold ``init``; the ampersand that would decode an entity reference; and
1003#: the backtick, because a code span binds more tightly than the link's brackets and
1004#: swallows the characters between a pair. Every one is ASCII punctuation, which
1005#: CommonMark lets a backslash escape. The destination is percent-encoded instead, so
1006#: the two halves of a link never disagree about where a path ends.
1007_LINK_TEXT_UNSAFE = re.compile(r"([\\\[\]<>_*~&`])")
1009#: The frontmatter contract the reader depends on. Fixed and small on purpose:
1010#: `retrieve_relevant_learnings` reads `title` and `description` out of it, so a
1011#: field added here is a field that side has to be taught.
1012LEARNING_SCHEMA_VERSION = "keel.learning.v1"
1014#: Every placeholder a `path` or `filename` template may use. Named rather than
1015#: open-ended: an unknown placeholder is a typo that would otherwise write a
1016#: directory called `{repoo}` and look like it worked.
1017LEARNING_SINK_PLACEHOLDERS = (
1018 "owner",
1019 "repo",
1020 "base_branch",
1021 "date",
1022 "pr",
1023 "slug",
1024 "fingerprint",
1025)
1027_SLUG_STRIP = re.compile(r"[^a-z0-9]+")
1030def _slugify(text: str | None, *, limit: int = 48) -> str:
1031 """A filename-safe slug, or `learning` when the title reduces to nothing."""
1032 slug = _SLUG_STRIP.sub("-", (text or "").lower()).strip("-")
1033 if not slug:
1034 return "learning"
1035 return slug[:limit].rstrip("-")
1038def learning_sink_policy(config: _HasPolicyPack | None) -> dict[str, Any] | None:
1039 """The `policy_pack.capture.learning.sink` block, or `None` when unset.
1041 `None` and `{}` are different answers and the difference is load-bearing: every
1042 field of a sink is optional, so `sink: {}` is a project taking the documented
1043 defaults — `markdown-dir` into `.keel/learning`. Collapsed to `{}`, that
1044 declaration read as *no sink at all* and the project got the pre-#1154
1045 behaviour of writing nothing, while the same block naming its `kind` wrote.
1046 """
1047 sink = _learning_policy(config).get("sink")
1048 return sink if isinstance(sink, dict) else None
1051#: Where a recorded `capture.artifact` can be read from.
1052ARTIFACT_SCOPE_REPOSITORY = "repository"
1053ARTIFACT_SCOPE_MACHINE = "machine"
1056def artifact_scope(artifact: str | None, config: _HasPolicyPack | None = None) -> str | None:
1057 """Where this project's sink writes: `repository`, `machine`, or ``None`` for neither.
1059 An in-repo sink's path is recorded relative to ``--root``, so it means the same thing
1060 in every clone. A sink outside the checkout — a shared `~/knowledge` folder — is
1061 recorded absolute, which names nothing on any other host. The run ledger is gitignored
1062 per-host state by default (`.keel/state/run-ledger.jsonl`), but
1063 `policy_pack.reports.run_ledger` may point it at a tracked file, and then that path
1064 travels to teammates and CI runners with it. Saying so in the record is what lets
1065 `keel capture-verify` tell "written somewhere this host cannot see" from "never
1066 written", which are the same absence and very different facts (#1185).
1068 It describes the **sink**, not the artifact, so it is recorded even when this run
1069 wrote nothing: a cross-host duplicate drops the unreadable path it would have reused,
1070 and without the scope beside it that record is indistinguishable from a capture that
1071 produced no file at all.
1073 ``None`` when the project has no sink, which is not the same as a sink somewhere else
1074 — `learning_sink_in_worktree` answers False for "no sink" and for "capture disabled"
1075 as well as for an outside one, so the sink has to be looked for separately. With no
1076 config, the path's own shape decides: an anchored path was always an outside sink.
1077 """
1078 if config is not None:
1079 sink = learning_sink_policy(config)
1080 if sink is None:
1081 return None
1082 return (
1083 ARTIFACT_SCOPE_REPOSITORY if _sink_writes_in_worktree(sink) else ARTIFACT_SCOPE_MACHINE
1084 )
1085 if not isinstance(artifact, str) or not artifact.strip():
1086 return None
1087 path = artifact.strip()
1088 anchored = path.startswith("~") or workspace.is_root_anchored(path)
1089 return ARTIFACT_SCOPE_MACHINE if anchored else ARTIFACT_SCOPE_REPOSITORY
1092def learning_sink_in_worktree(config: _HasPolicyPack | None) -> bool:
1093 """Does this project's sink write **inside the repository**?
1095 Pure, and answered from the path's shape rather than the filesystem: a relative
1096 path resolves against `--root`, which is the checkout, while an absolute or `~`
1097 path is a folder somewhere else. It decides who has to commit the file — keel
1098 writes it and does not, so one inside the working tree is lost to the next
1099 worktree and to every CI runner unless the run commits it.
1100 """
1101 # Gated on the hook too: a dormant `sink:` under a disabled capture writes
1102 # nothing, so nothing needs committing. Fourth reader of the same question.
1103 sink = learning_sink_policy(config) if capture_hook_enabled(config) else None
1104 return sink is not None and _sink_writes_in_worktree(sink)
1107def _sink_writes_in_worktree(sink: dict[str, Any]) -> bool:
1108 """Does this sink block's **path** name somewhere inside the checkout?
1110 Split out from :func:`learning_sink_in_worktree` because the two callers ask
1111 different questions of the same block. *Who commits the file* is gated on the
1112 capture hook — a dormant sink writes nothing, so nothing needs committing — but
1113 *what shape the path has* is not. Answering the second through the first made a
1114 dormant in-repo sink record `artifact_scope: machine`, which reads as "written on
1115 another host" for a repo-relative path every clone can see.
1116 """
1117 path = str(sink.get("path") or DEFAULT_LEARNING_SINK_PATH)
1118 if path.startswith("~"):
1119 return False
1120 # **Anchored on *any* platform, not this one.** `Path("C:/knowledge").is_absolute()`
1121 # is False on POSIX and `Path("/srv/knowledge").is_absolute()` is False on
1122 # Windows, so each host called the other's absolute path in-repo and would have
1123 # told the adapter to `git add` it — committing a `C:` directory into the
1124 # repository, or reaching outside it. A keel config is the same text wherever it
1125 # is read; `workspace.is_root_anchored` is the question already asked that way.
1126 if workspace.is_root_anchored(path):
1127 return False
1128 # **Normalised, because `../learnings` is relative and still outside.** It is
1129 # the documented "folder next to the checkout" shape without the leading `~`,
1130 # and reported as in-repo it would send the adapter to `git add` a path git
1131 # refuses — leaving the file off `base_branch` and the next worktree empty,
1132 # which is the failure this flag exists to prevent, arriving through the flag.
1133 return Path(os.path.normpath(path)).parts[:1] != ("..",)
1136def learning_sink_errors(sink: Any) -> list[str]:
1137 """Why this `sink` block cannot be used, or `[]`.
1139 Validated where the config is read rather than where the file is written: a
1140 template naming `{repoo}` is a typo whose only symptom would otherwise be a
1141 directory by that name, created successfully, on a machine nobody is watching.
1142 """
1143 if sink in (None, {}):
1144 return []
1145 if not isinstance(sink, dict):
1146 return ["policy_pack.capture.learning.sink must be a mapping"]
1147 errors: list[str] = []
1148 kind = sink.get("kind", LEARNING_SINK_KINDS[0])
1149 if kind not in LEARNING_SINK_KINDS:
1150 errors.append(
1151 f"policy_pack.capture.learning.sink.kind must be one of "
1152 f"{', '.join(LEARNING_SINK_KINDS)} (got {kind!r})"
1153 )
1154 for field_name in ("path", "filename"):
1155 raw = sink.get(field_name)
1156 if raw is None:
1157 continue
1158 if not isinstance(raw, str) or not raw.strip():
1159 errors.append(
1160 f"policy_pack.capture.learning.sink.{field_name} must be a non-empty string"
1161 )
1162 continue
1163 # `[^}]*`, not `[a-z_]*`: a mixed-case or hyphenated typo — `{Repo}`,
1164 # `{base-branch}` — is exactly as wrong as `{repoo}` and was invisible to a
1165 # pattern that only matched the shape of a correct name.
1166 used = re.findall(r"\{([^}]*)\}", raw)
1167 for name in used:
1168 if name not in LEARNING_SINK_PLACEHOLDERS:
1169 errors.append(
1170 f"policy_pack.capture.learning.sink.{field_name} uses unknown placeholder "
1171 f"{{{name}}}; known: {', '.join(LEARNING_SINK_PLACEHOLDERS)}"
1172 )
1173 # **A filename must be able to name two lessons.** Date, PR and slug do not
1174 # distinguish them: a second `create-learning` run on the same PR the same
1175 # day is a *different* lesson with a different fingerprint, and without it in
1176 # the name the write destroys the first one — leaving its ledger record
1177 # pointing at a document that says something else. Refused here, where the
1178 # placeholder typos are refused, because the only other symptom is a file
1179 # that quietly stops existing.
1180 if field_name == "filename" and "fingerprint" not in used:
1181 errors.append(
1182 "policy_pack.capture.learning.sink.filename must contain {fingerprint}; "
1183 "without it two lessons on one pull request overwrite each other"
1184 )
1185 # **A filename is a name, not a path.** `{pr}/{fingerprint}.md` passes every
1186 # other check, `mkdir(parents=True)` creates the directory happily, and
1187 # `retrieve_relevant_learnings` — the only reader — globs one level and
1188 # skips directories, so the lesson is written where nothing will ever read
1189 # it. Nesting belongs in `path`, which is the field that names a directory.
1190 if field_name == "filename" and not raw.endswith(LEARNING_READ_SUFFIXES):
1191 errors.append(
1192 f"policy_pack.capture.learning.sink.filename must end in one of "
1193 f"{', '.join(LEARNING_READ_SUFFIXES)}; the read path opens no other "
1194 f"suffix, so anything else is written and never found"
1195 )
1196 if field_name == "filename" and ("/" in raw or "\\" in raw):
1197 errors.append(
1198 "policy_pack.capture.learning.sink.filename must not contain a path "
1199 "separator; it names a file inside `path`, and the read path does not "
1200 "descend into subdirectories"
1201 )
1202 return errors
1205#: Anything that would make a filename more than one path component, or unwritable.
1206#: `/` and `\\` split it; the control characters are the same class one field over.
1207_FILENAME_UNSAFE = re.compile(r"[/\\\x00-\x1f\x7f]+")
1210def _relative_stays_relative(template: str, values: dict[str, str]) -> str:
1211 """Expand a directory template without letting it change what it *is*.
1213 An absolute or `~` template stays what the project wrote. A relative one has
1214 to come back relative: a leading placeholder that expands to nothing otherwise
1215 turns `{repo}/learnings` into `/learnings`, which the shape-reading predicates
1216 still report as inside the repository — so the adapter is told to `git add` a
1217 path at the filesystem root.
1218 """
1219 expanded = _expand(template, values)
1220 if template.startswith("~") or workspace.is_root_anchored(template):
1221 return expanded
1222 return expanded.lstrip("/\\") or DEFAULT_LEARNING_SINK_PATH
1225def _one_component(name: str) -> str:
1226 """A filename that names exactly one file, whatever the placeholders held.
1228 Refusing a separator in the *template* is not enough: `{base_branch}` is a legal
1229 filename placeholder and `feat/sink` is a normal branch, so
1230 `{date}-pr{pr}-{base_branch}-{fingerprint}.md` expands to a name with a slash in
1231 it, `mkdir(parents=True)` makes the directory, and the lesson lands one level
1232 below where `retrieve_relevant_learnings` looks. Only `{slug}` was slugified;
1233 every other value went in raw.
1234 """
1235 return _FILENAME_UNSAFE.sub("-", name)
1238def _expand(template: str, values: dict[str, str]) -> str:
1239 out = template
1240 for name, value in values.items():
1241 out = out.replace("{" + name + "}", value)
1242 return out
1245#: A value plain YAML reads back unchanged: starts with a letter, contains only
1246#: letters, digits, space and a few punctuation marks that carry no meaning there.
1247_YAML_PLAIN = re.compile(r"[A-Za-z][A-Za-z0-9 ._/()+-]*")
1249#: **Everything `str.splitlines()` treats as a line break**, plus the rest of C0 and
1250#: DEL. Not a taste question and not only the obvious ones: a raw CR splits a line
1251#: inside quotes, a NUL makes a real parser refuse the document, and `\x85` (NEL),
1252#: `\u2028` (LINE SEPARATOR) and `\u2029` (PARAGRAPH SEPARATOR) are line breaks to
1253#: Python while looking like nothing at all — a C0-only pattern let a title open a
1254#: Markdown section of its own through the very guard written to stop it.
1255_YAML_CONTROL = re.compile("[\x00-\x1f\x7f\x85\u2028\u2029]")
1257#: Words plain YAML turns into something that is not a string.
1258_YAML_KEYWORDS = frozenset(
1259 {"y", "n", "yes", "no", "true", "false", "on", "off", "null", "none", "~"}
1260)
1263def _one_line(value: str) -> str:
1264 """A value that cannot start a second line, wherever it is written.
1266 :func:`_yaml_scalar` applies this before quoting, and the **body** needs it too:
1267 the document's `# {title}` heading took the raw string, so a title carrying a
1268 newline wrote a heading and then whatever followed it as Markdown of its own —
1269 `foo\n## injected` became `# foo` and an `## injected` section. The front matter
1270 and the body have to say the same thing about the same field.
1271 """
1272 return _YAML_CONTROL.sub(" ", value)
1275def _yaml_scalar(value: str) -> str:
1276 """A front-matter value that survives a real YAML parser.
1278 **Quote unless the value is plainly safe**, rather than quoting a list of
1279 dangerous characters. The first cut listed `:`, `#`, `"` and a newline, and a
1280 dozen other shapes went through it: a leading `-`, `*`, `&`, `!`, `@`, `%`,
1281 `|`, `>` or backtick either fails to parse or comes back as something else,
1282 `{a}` becomes a mapping, `[a]` a list, and `yes` becomes `True`. An issue title
1283 can be any of those. A deny-list has to be right about every character; an
1284 allow-list only has to be right about the ones it lets through.
1285 """
1286 # `value == value.strip()` is not tidiness: `yes ` matches the allow-list, is not
1287 # in the keyword set, and goes in bare — and a plain YAML scalar drops its
1288 # trailing space, so a parser reads `yes` and returns `True`. The same
1289 # non-string a keyword produces, through the one gap the keyword check had.
1290 if (
1291 value
1292 and value == value.strip()
1293 and _YAML_PLAIN.fullmatch(value)
1294 and value.lower() not in _YAML_KEYWORDS
1295 ):
1296 return value
1297 escaped = value.replace("\\", "\\\\").replace('"', '\\"')
1298 # **Every control character, not the newline.** A raw carriage return survives
1299 # inside quotes, and `_front_matter` splits lines on it — the reader gets a
1300 # title of `"foo` and drops the rest, while a real parser reads `foo bar`, so
1301 # the two disagree about the same file. A NUL is worse: PyYAML refuses the
1302 # document outright. They become spaces, because a title is a line.
1303 return f'"{_one_line(escaped)}"'
1306def _yaml_sequence(key: str, values: list[str] | tuple[str, ...]) -> list[str]:
1307 """A front-matter list, written so an empty one reads back as an empty list.
1309 `key:` with nothing under it is a **null** to a YAML parser, not `[]`, and the
1310 contract calls these fields sequences — a consumer that iterates them raises
1311 `TypeError` on a merge that touched nothing it recorded. `key: []` is the same
1312 field with the type it promised.
1313 """
1314 items = _strings(values)
1315 if not items:
1316 return [f"{key}: []"]
1317 return [f"{key}:", *(f" - {_yaml_scalar(item)}" for item in items)]
1320def learning_file_link_base(
1321 *,
1322 directory: str,
1323 in_repo: bool,
1324 owner: str | None,
1325 repo: str | None,
1326 head_sha: str | None,
1327) -> str | None:
1328 """Where the **Files** section's links point, or ``None`` for bare paths (#1166).
1330 Two forms, chosen by where the sink is. A sink **inside the checkout** links
1331 relative to the document's own directory — ``../..`` from the default
1332 ``.keel/learning`` — so the link resolves on disk from where the file sits, and a
1333 graph builder walking the repository makes the edge to a node it already has.
1334 The prefix is **lexical** — one ``..`` per component of the normalised directory —
1335 because a plan touches no filesystem: :func:`posixpath.relpath` would consult
1336 ``os.getcwd()`` for two relative arguments, which made the prefix depend on where
1337 the process stood and raise from a deleted directory. It is POSIX on every platform
1338 because the document is read on machines other than the one that wrote it.
1340 A sink **outside** the checkout has nothing to link to relatively, and a relative
1341 link that resolves to nothing is worse than none. It links the file on GitHub at
1342 the merged head when the owner, the repository and the head are all known, and
1343 otherwise says nothing — the caller renders the bare path. An *expanded* template
1344 that climbs out of the checkout (``{owner}/../learnings`` with ``owner`` unset) is
1345 outside it whatever the unexpanded template looked like, and takes the same forms.
1346 """
1347 if in_repo:
1348 normalised = posixpath.normpath(directory.replace("\\", "/"))
1349 parts = [part for part in normalised.split("/") if part not in ("", ".")]
1350 if not parts or parts[0] != "..":
1351 return "/".join([".."] * len(parts)) or "."
1352 if owner and repo and head_sha:
1353 return f"https://github.com/{owner}/{repo}/blob/{head_sha}"
1354 return None
1357def _file_bullet(path: str, base: str | None) -> str:
1358 """One **Files** bullet: a CommonMark link when there is a base, else the bare path.
1360 The link text is the repository's own name for the file, which also puts every
1361 path into the body — and :func:`retrieve_relevant_learnings` scores a document by
1362 its text, so a query naming a file now scores the lesson about it higher than the
1363 front-matter list alone did (the list matched once; the link matches again — in
1364 its text, or in its destination when the text carries an escape).
1365 The destination is percent-encoded: a space or a parenthesis in a path would
1366 otherwise end the link where the path continues.
1367 """
1368 name = _one_line(path)
1369 if base is None:
1370 # A code span ends at a backtick run as long as its opener, so a path carrying
1371 # one is fenced with a run one longer, padded as CommonMark allows — never
1372 # written plain, where its own Markdown would render.
1373 longest = max((len(run) for run in re.findall(r"`+", name)), default=0)
1374 if longest == 0:
1375 return f"- `{name}`"
1376 fence = "`" * (longest + 1)
1377 return f"- {fence} {name} {fence}"
1378 text = _LINK_TEXT_UNSAFE.sub(r"\\\1", name)
1379 return f"- [{text}]({base}/{quote(name, safe='/')})"
1382def _files_section(changed_files: list[str] | tuple[str, ...], base: str | None) -> list[str]:
1383 """The **Files** section's lines, from the same list the front matter is given.
1385 An empty list still renders the heading: the document shape is stable — four
1386 sections, always — and a lesson about no file says so rather than pointing at
1387 nothing.
1388 """
1389 items = _strings(changed_files)
1390 if not items:
1391 return [LEARNING_NO_FILES]
1392 return [_file_bullet(item, base) for item in items]
1395def render_learning_document(
1396 *,
1397 title: str | None,
1398 description: str | None,
1399 pr_number: int | None,
1400 issue_number: int | None,
1401 repo: str | None,
1402 date: str,
1403 labels: list[str] | tuple[str, ...] = (),
1404 changed_files: list[str] | tuple[str, ...] = (),
1405 fingerprint: str = "",
1406 what_changed: str = "",
1407 what_we_learned: str = "",
1408 do_differently: str = "",
1409 file_link_base: str | None = None,
1410) -> str:
1411 """One learning file: frontmatter the reader can rely on, then four sections.
1413 The heading is repeated below the frontmatter deliberately. `retrieve_relevant_learnings`
1414 scores a file by its own text, and a title that lives only in frontmatter is a
1415 title the search cannot weigh.
1417 ``file_link_base`` is what :func:`learning_file_link_base` decided for the sink;
1418 the **Files** section links each ``changed_files`` entry under it, and renders
1419 the bare path when it is ``None``. The front matter does not read it: the list
1420 up there is byte-for-byte what it was before the section existed (#1166).
1421 """
1422 front = [
1423 "---",
1424 f"schema: {LEARNING_SCHEMA_VERSION}",
1425 f"title: {_yaml_scalar(title or 'Learning')}",
1426 f"description: {_yaml_scalar(description or '')}",
1427 f"repo: {_yaml_scalar(repo or '')}",
1428 # `null`, not an empty value. Both read back as `None`; only one of them
1429 # says so on purpose, and a bare `issue:` looks like a field somebody forgot
1430 # to fill rather than a merge that was linked to no issue.
1431 f"pr: {pr_number if pr_number is not None else 'null'}",
1432 f"issue: {issue_number if issue_number is not None else 'null'}",
1433 # Quoted like every other scalar in the block. Left bare, `2026-09-09` is a
1434 # YAML *timestamp*: a real parser returns `datetime.date` where keel's reader
1435 # returns the string, which is the one disagreement all this quoting exists
1436 # to prevent.
1437 f"date: {_yaml_scalar(date)}",
1438 # Quoted like every other scalar: a sha256 that happens to be all digits is
1439 # an `int` to a real parser and a 64-character string to keel's reader, and
1440 # this is the field that *identifies* the lesson.
1441 f"fingerprint: {_yaml_scalar(fingerprint)}",
1442 ]
1443 front += _yaml_sequence("labels", labels)
1444 front += _yaml_sequence("changed_files", changed_files)
1445 front.append("---")
1446 body = [
1447 "",
1448 # Through `_one_line`, like the front matter above: a heading built from a
1449 # raw title let a newline open a section of its own inside the document.
1450 f"# {_one_line(title or 'Learning')}",
1451 "",
1452 _one_line(description or ""),
1453 "",
1454 LEARNING_SECTION_HEADINGS[0],
1455 "",
1456 what_changed or LEARNING_EMPTY_SECTION,
1457 "",
1458 LEARNING_SECTION_HEADINGS[1],
1459 "",
1460 what_we_learned or LEARNING_EMPTY_SECTION,
1461 "",
1462 LEARNING_SECTION_HEADINGS[2],
1463 "",
1464 do_differently or LEARNING_EMPTY_SECTION,
1465 "",
1466 LEARNING_FILES_HEADING,
1467 "",
1468 *_files_section(changed_files, file_link_base),
1469 "",
1470 ]
1471 return "\n".join(front + body)
1474def learning_sink_writes(
1475 *,
1476 config: _HasPolicyPack | None,
1477 decision: dict[str, Any] | None,
1478 capture_status: str | None,
1479) -> bool:
1480 """Whether this run writes a learning file at all.
1482 Split out so a caller can answer it **before** doing anything expensive — the
1483 writer redacts its values before rendering, and reaching for the redaction
1484 policy on a run with no sink turned an invalid `capture_redaction` pattern into
1485 an exception raised from the wrong place, past the handler `keel ship` has for
1486 exactly that. :func:`learning_sink_plan` asks the same question through this
1487 function, so the two cannot drift.
1489 Three conditions, and the third is the one that took two rounds to get right.
1490 **The decision is the gate**, not merely the dedupe: `learning_decision` already
1491 answers whether this run earns a durable artifact, and `create-learning` is the
1492 one answer whose `durable_artifact` is true. Refusing only `duplicate` let four
1493 other answers through — `learning.enabled` false, `enabled` omitted, `mode:
1494 defer`, `mode: marker-only` — each planning a write while the record beside it
1495 said the policy had decided not to keep one. A configured sink is where a
1496 project's learnings go, not permission to write one whatever the policy says.
1497 """
1498 if capture_status != "applied":
1499 return False
1500 if not capture_hook_enabled(config):
1501 return False
1502 sink = learning_sink_policy(config)
1503 if sink is None or learning_sink_errors(sink):
1504 return False
1505 return isinstance(decision, dict) and decision.get("decision") == "create-learning"
1508def learning_sink_plan(
1509 *,
1510 config: _HasPolicyPack | None,
1511 decision: dict[str, Any] | None,
1512 capture_status: str | None,
1513 owner: str | None,
1514 repo: str | None,
1515 base_branch: str | None,
1516 date: str,
1517 pr_number: int | None,
1518 title: str | None = None,
1519 description: str | None = None,
1520 labels: list[str] | tuple[str, ...] = (),
1521 changed_files: list[str] | tuple[str, ...] = (),
1522 issue_number: int | None = None,
1523 what_changed: str = "",
1524 what_we_learned: str = "",
1525 do_differently: str = "",
1526 head_sha: str | None = None,
1527) -> dict[str, Any] | None:
1528 """What to write for this run, or `None` with the reason folded into the caller.
1530 Pure: it resolves a path and renders a document and touches no filesystem and no
1531 clock — `date` is passed in for the same reason every other plan in this package
1532 takes its facts as arguments.
1534 Returns `None` when :func:`learning_sink_writes` says there is nothing to write:
1535 no sink configured, a capture that is not `applied`, or a learning decision that
1536 does not call for a durable artifact — a `duplicate`, which is the dedupe doing
1537 its job, but equally a project whose policy said `marker-only`.
1538 """
1539 if not learning_sink_writes(config=config, decision=decision, capture_status=capture_status):
1540 return None
1541 sink = learning_sink_policy(config) or {}
1542 # No `isinstance` re-check: the gate above returned for anything that is not a
1543 # `create-learning` mapping, so by here the decision is one.
1544 fingerprint = str(decision.get("fingerprint") or "")
1545 values = {
1546 "owner": owner or "",
1547 "repo": repo or "",
1548 "base_branch": base_branch or "",
1549 "date": date,
1550 "pr": str(pr_number) if pr_number is not None else "",
1551 "slug": _slugify(title),
1552 "fingerprint": fingerprint[:LEARNING_FINGERPRINT_SLICE],
1553 }
1554 # **A relative template stays relative.** `{repo}/learnings` with `repo` unset
1555 # expands to `/learnings` — absolute, at the filesystem root — while every
1556 # reader of the template's shape (`learning_sink_in_worktree`, and so
1557 # `commit_required` and the adapter's `git add`) still calls it in-repo. The
1558 # filename's expansion was already flattened; the directory's was not.
1559 directory = _relative_stays_relative(
1560 str(sink.get("path") or DEFAULT_LEARNING_SINK_PATH), values
1561 )
1562 filename = _one_component(
1563 _expand(str(sink.get("filename") or DEFAULT_LEARNING_SINK_FILENAME), values)
1564 )
1565 # Decided from the sink's *shape*, the way `learning_sink_in_worktree` decides
1566 # who commits the file: the same question, and the two answers agree about where
1567 # the document sits — except when a placeholder's expansion climbs out of the
1568 # checkout, which the link base sees and the template reader does not, and then
1569 # the outside form is the safe one.
1570 file_link_base = learning_file_link_base(
1571 directory=directory,
1572 in_repo=learning_sink_in_worktree(config),
1573 owner=owner,
1574 repo=repo,
1575 head_sha=head_sha,
1576 )
1577 return {
1578 "kind": sink.get("kind", LEARNING_SINK_KINDS[0]),
1579 "directory": directory,
1580 "filename": filename,
1581 "file_link_base": file_link_base,
1582 "content": render_learning_document(
1583 title=title,
1584 description=description,
1585 pr_number=pr_number,
1586 issue_number=issue_number,
1587 repo=repo,
1588 date=date,
1589 labels=labels,
1590 changed_files=changed_files,
1591 fingerprint=fingerprint,
1592 what_changed=what_changed,
1593 what_we_learned=what_we_learned,
1594 do_differently=do_differently,
1595 file_link_base=file_link_base,
1596 ),
1597 }
1600def duplicate_learning_artifact(
1601 *,
1602 config: _HasPolicyPack | None,
1603 decision: dict[str, Any] | None,
1604 capture_status: str | None,
1605 existing_records: list[dict[str, Any]] | tuple[dict[str, Any], ...] = (),
1606) -> str | None:
1607 """The artifact an earlier run already wrote for this same learning, or `None`.
1609 A `duplicate` decision writes no file — that is the dedupe working — but the
1610 run still records `applied`, and `applied` with no artifact is exactly what
1611 `capture-verify` reports as a finding. The lesson is not missing: it is on
1612 disk, under the run this one duplicates. Naming that file keeps the claim
1613 provable instead of letting the dedupe manufacture the gap the artifact
1614 exists to close.
1616 **Only for an `applied` capture**, and that gate is the whole point rather than
1617 a precaution: `learning_decision` answers `duplicate` on a fingerprint match
1618 before it looks at the status, so without it a `not-run` or `skipped` record
1619 was handed the earlier run's path — the exact contradiction the CLI refuses at
1620 its flag boundary, *a run that never reached capture produced no artifact*, and
1621 the one `record_marker` states for `deferred` and `skipped`.
1623 Pure: it reads recorded paths and never asks whether one still exists. The
1624 caller that can answer that is the caller that touches the filesystem.
1625 """
1626 if capture_status != "applied":
1627 return None
1628 # **The hook, not just the sink block.** `learning_decision` answers
1629 # `duplicate` on a fingerprint match *before* it reads the enabled flags, so
1630 # `duplicate` reaches here under a `capture.enabled: false` or `marker-only`
1631 # project — and the caller treats a returned path as permission to replace the
1632 # operator's own `--capture-artifact` with a stale sink file, for a project
1633 # whose contract says `extension-owned`. Fifth reader of the same question.
1634 if not capture_hook_enabled(config) or learning_sink_policy(config) is None:
1635 return None
1636 if not isinstance(decision, dict) or decision.get("decision") != "duplicate":
1637 return None
1638 fingerprint = decision.get("fingerprint")
1639 if not fingerprint:
1640 return None
1641 artifact: str | None = None
1642 for record in existing_records:
1643 if not isinstance(record, dict):
1644 continue
1645 capture_block = record.get("capture")
1646 if not isinstance(capture_block, dict):
1647 continue
1648 learning = capture_block.get("learning")
1649 if not isinstance(learning, dict) or learning.get("fingerprint") != fingerprint:
1650 continue
1651 candidate = capture_block.get("artifact")
1652 if isinstance(candidate, str) and candidate.strip():
1653 # Scope is recorded, not acted on here. Whether a `machine`-scoped path can be
1654 # read is a filesystem question, and this module is pure — the CLI wrapper
1655 # already asks it (`_duplicate_learning_artifact`), which is what keeps the
1656 # same-host case working: the file really is there, and dropping the path here
1657 # would record `applied` with no artifact on the very machine that wrote it.
1658 # Keep scanning: the ledger is append-only and the newest record
1659 # holding this fingerprint is the one whose path is current.
1660 artifact = candidate.strip()
1661 return artifact
1664def _unquote(value: str) -> str:
1665 """Undo :func:`_yaml_scalar` for the fields this reader uses.
1667 The writer quotes any value containing a colon — an issue title usually does —
1668 so a reader that took the raw text would hand back a title wrapped in quotes.
1669 """
1670 if len(value) >= 2 and value[0] == value[-1] == '"':
1671 return value[1:-1].replace('\\"', '"').replace("\\\\", "\\")
1672 return value
1675def _front_matter(content: str) -> tuple[dict[str, str], str]:
1676 """Split a leading `---` block off, as `(fields, body)`.
1678 Only the scalar fields this contract defines are read; a list value (`labels:`)
1679 is skipped rather than parsed, because the caller wants a title and a sentence
1680 and nothing here should grow into a YAML parser.
1681 """
1682 lines = content.splitlines()
1683 if not lines or lines[0].strip() != "---":
1684 return {}, content
1685 fields: dict[str, str] = {}
1686 for index, line in enumerate(lines[1:], start=1):
1687 if line.strip() == "---":
1688 return fields, "\n".join(lines[index + 1 :])
1689 key, sep, value = line.partition(":")
1690 if sep and not key.startswith(" "):
1691 fields[key.strip()] = _unquote(value.strip())
1692 # No closing delimiter: not front matter, whatever it looked like.
1693 return {}, content
1696def _learning_title_and_summary(content: str, fallback: str) -> tuple[str, str]:
1697 """A learning file's title and one-line summary.
1699 Front matter first, because that is what the writer fills. Read line-by-line
1700 instead, a file written by `render_learning_document` would be titled `---` and
1701 summarised `schema: keel.learning.v1` — measured, and the reason the reader is
1702 part of the change that added the writer.
1703 """
1704 fields, body = _front_matter(content)
1705 lines = [line.strip() for line in body.splitlines() if line.strip()]
1706 title = fields.get("title") or (lines[0].lstrip("#").strip() if lines else fallback)
1707 summary = fields.get("description") or ""
1708 if not summary:
1709 for line in lines[1:]:
1710 if not line.startswith("#"):
1711 summary = line
1712 break
1713 return title or fallback, summary
1716#: Placeholders a `source` directory may use. A subset of the sink's: `{pr}`,
1717#: `{date}` and `{slug}` name one *document*, and a directory that named a single
1718#: PR would retrieve only that PR's lesson.
1719LEARNING_SOURCE_PLACEHOLDERS = ("owner", "repo", "base_branch")
1721#: How many lessons a brief carries, and how much of it they may take. A brief
1722#: *becomes* an agent's prompt, so an unbounded retrieval is an unbounded prompt —
1723#: the same reason `fixloop` clamps a reviewer's findings.
1724DEFAULT_LEARNING_RETRIEVAL_LIMIT = 5
1725LEARNING_BRIEF_CHAR_BUDGET = 4000
1726LEARNING_BRIEF_SUMMARY_CHARS = 240
1728#: The heading both briefs use, fixed so an agent reading two of them recognises
1729#: the same section rather than inferring one from prose.
1730LEARNING_BRIEF_HEADING = "Relevant past learnings"
1731LEARNING_RETRIEVAL_SCHEMA_VERSION = "keel.learning-retrieval.v1"
1733#: What an exact front-matter match is worth beside a text hit. A file whose
1734#: `labels` or `changed_files` name this task's own label or path is *about* this
1735#: task; one that merely says "auth" eight times is worded like it. Text scoring
1736#: alone ranked the second above the first.
1737LEARNING_LABEL_MATCH_SCORE = 6
1738LEARNING_FILE_MATCH_SCORE = 8
1740#: keel's own scaffolding, emitted into **every** document this writer produces. The
1741#: scorer subtracts it before counting words: an issue titled *"What changed in the
1742#: merge window"* otherwise scored `what` and `changed` against all three headings of
1743#: every learning in the directory and cleared the floor on all of them, so three
1744#: unrelated lessons opened the brief. Named here rather than spelled twice, so the
1745#: writer and the scorer cannot drift.
1746LEARNING_SECTION_HEADINGS = (
1747 "## What changed",
1748 "## What we learned",
1749 "## What to do differently next time",
1750)
1751LEARNING_EMPTY_SECTION = "_Not recorded._"
1753#: How many **distinct** query words a file must contain to be a text match at all.
1754#: One is a coincidence — "keel" appears in every learning this repository writes —
1755#: and repetition does not make it less of one, which a point floor could not say: at
1756#: four points a single word said four times passed, and a genuine three-word match
1757#: in a short handwritten note did not. An exact front-matter match is admitted on
1758#: its own, whatever the text says.
1759LEARNING_MIN_DISTINCT_TOKENS = 2
1761#: Words too common to distinguish one learning from another.
1762_LEARNING_STOPWORDS = frozenset(
1763 {"the", "and", "for", "with", "this", "that", "issue", "feat", "fix"}
1764)
1767def learning_source_errors(source: Any) -> list[str]:
1768 """Why this `source` cannot be used, or `[]`.
1770 Checked where the config is read, like the sink's templates and for the same
1771 reason: a directory naming `{repoo}` is a typo whose only symptom is a
1772 retrieval that silently finds nothing, on every run, forever.
1773 """
1774 if source in (None, [], ()):
1775 return []
1776 entries = [source] if isinstance(source, str) else source
1777 if not isinstance(entries, (list, tuple)):
1778 return ["policy_pack.capture.learning.source must be a string or a list of strings"]
1779 errors: list[str] = []
1780 for entry in entries:
1781 if not isinstance(entry, str) or not entry.strip():
1782 errors.append("policy_pack.capture.learning.source entries must be non-empty strings")
1783 continue
1784 for name in re.findall(r"\{([^}]*)\}", entry):
1785 if name not in LEARNING_SOURCE_PLACEHOLDERS:
1786 errors.append(
1787 f"policy_pack.capture.learning.source uses unknown placeholder "
1788 f"{{{name}}}; known: {', '.join(LEARNING_SOURCE_PLACEHOLDERS)}"
1789 )
1790 return errors
1793def learning_source_entries(config: _HasPolicyPack | None) -> list[str]:
1794 """The directories a project *configured*, before any placeholder is expanded.
1796 Unset, it is **the directory the sink writes to** — a project that turned
1797 capture on has exactly one place its learnings live, and a second setting to
1798 keep in step with the first is a second setting to get wrong. With no sink
1799 either, the `.keel/learning/` convention, so retrieval works the moment a
1800 project has files however they got there.
1802 Separate from :func:`learning_source_dirs` because the pure contract has no
1803 `{repo}` to expand with: reporting the resolved list there would print an
1804 empty `sources` for a templated setting that reads fine at run time.
1805 """
1806 policy = _learning_policy(config)
1807 raw = policy.get("source")
1808 if isinstance(raw, str):
1809 entries = [raw]
1810 elif isinstance(raw, (list, tuple)):
1811 entries = [entry for entry in raw if isinstance(entry, str)]
1812 else:
1813 entries = []
1814 if not entries:
1815 sink = learning_sink_policy(config) or {}
1816 entries = [str(sink.get("path") or DEFAULT_LEARNING_SINK_PATH)]
1817 return entries
1820def learning_source_dirs(
1821 config: _HasPolicyPack | None,
1822 *,
1823 values: dict[str, str] | None = None,
1824) -> list[str]:
1825 """The directories this run reads, with the placeholders expanded.
1827 Pure. An entry still holding a placeholder after expansion is dropped rather
1828 than taken literally, which is what keeps a per-document sink path (`…/{pr}/`)
1829 from being read as a folder called `{pr}`.
1830 """
1831 resolved: list[str] = []
1832 for entry in learning_source_entries(config):
1833 expanded = _expand(entry, values or {}).strip()
1834 if not expanded or "{" in expanded:
1835 continue
1836 if expanded not in resolved:
1837 resolved.append(expanded)
1838 return resolved
1841def learning_query_text(
1842 *,
1843 title: str | None = None,
1844 labels: list[str] | tuple[str, ...] = (),
1845 changed_files: list[str] | tuple[str, ...] = (),
1846) -> str:
1847 """The text a task is matched by: its title, its labels, and its paths.
1849 The paths belong in it. A lesson about `src/keel/ledger.py` and an issue that
1850 touches `src/keel/ledger.py` share no words at all when only titles are
1851 compared, which is the pairing retrieval most needs to make.
1852 """
1853 parts = [title or "", *_strings(labels), *_strings(changed_files)]
1854 return " ".join(part for part in parts if part)
1857def _front_matter_lists(content: str) -> dict[str, list[str]]:
1858 """Read string sequences from a bounded, complete YAML front matter block.
1860 Use the same safe parser as project configuration for block and flow lists.
1861 Only direct string items are consumed; aliases cannot cause recursive walks.
1862 Invalid or oversized metadata contributes no exact matches.
1863 """
1864 lines = content.splitlines()
1865 if not lines or lines[0].strip() != "---":
1866 return {}
1867 for index, line in enumerate(lines[1:], start=1):
1868 if line.strip() == "---":
1869 front = "\n".join(lines[1:index])
1870 if len(front) > 65536:
1871 return {}
1872 try:
1873 fields = yaml.load(front)
1874 except (yaml.YAMLError, RecursionError):
1875 return {}
1876 if not isinstance(fields, dict):
1877 return {}
1878 return {
1879 key: [item for item in value if isinstance(item, str)]
1880 for key in ("labels", "changed_files")
1881 if isinstance(value := fields.get(key), list)
1882 }
1883 return {}
1886def _learning_tokens(query_text: str) -> set[str]:
1887 return {
1888 word.lower()
1889 for word in re.findall(r"[A-Za-z0-9_-]{3,}", query_text)
1890 if word.lower() not in _LEARNING_STOPWORDS
1891 }
1894_LEARNING_METADATA_KEYS = (
1895 "schema",
1896 "title",
1897 "description",
1898 "repo",
1899 "pr",
1900 "issue",
1901 "date",
1902 "fingerprint",
1903 "labels",
1904 "changed_files",
1905)
1906_LEARNING_METADATA_LINE = re.compile(
1907 r"^(?:" + "|".join(_LEARNING_METADATA_KEYS) + r")\s*[:=].*$",
1908 re.IGNORECASE,
1909)
1912def _lesson_text(body: str, suffix: str = ".md") -> str:
1913 """Score prose without treating metadata-shaped lesson sentences as fields.
1915 Markdown metadata was already removed with its front matter. JSON fields
1916 are removed structurally; plain text recognizes a leading schema header,
1917 ending at the first non-metadata line. Body sentences stay intact.
1918 """
1919 if suffix == ".json":
1920 try:
1921 record = json.loads(body)
1922 except (ValueError, RecursionError):
1923 record = None
1924 if isinstance(record, dict):
1925 body = "\n".join(
1926 value
1927 for key, value in record.items()
1928 if key.lower() not in _LEARNING_METADATA_KEYS and isinstance(value, str)
1929 )
1930 elif suffix == ".txt":
1931 lines = body.splitlines()
1932 if lines and re.fullmatch(r"schema\s*[:=]\s*keel\.learning\.v1", lines[0], re.I):
1933 index = 0
1934 while index < len(lines) and _LEARNING_METADATA_LINE.fullmatch(lines[index]):
1935 index += 1
1936 body = "\n".join(lines[index:])
1937 text = body.lower()
1938 for scaffold in (
1939 *LEARNING_SECTION_HEADINGS,
1940 LEARNING_EMPTY_SECTION,
1941 LEARNING_FILES_HEADING,
1942 LEARNING_NO_FILES,
1943 "## Changed files",
1944 ):
1945 text = text.replace(scaffold.lower(), " ")
1946 return text
1949def _text_score(content_lower: str, filename_lower: str, tokens: set[str]) -> tuple[int, int]:
1950 """`(points, distinct tokens matched)` for one file.
1952 The two answer different questions and only one of them decides admission.
1953 Points order the results; **distinct tokens** say whether this is a match at
1954 all, because repetition is not evidence — a note saying `ledger` twelve times
1955 matches a query about ledgers exactly as much as one saying it twice.
1956 """
1957 score = 0
1958 distinct = 0
1959 for token in tokens:
1960 hit = False
1961 if token in filename_lower:
1962 score += 3
1963 hit = True
1964 count = content_lower.count(token)
1965 if count > 0:
1966 score += min(count, 5)
1967 hit = True
1968 distinct += hit
1969 return score, distinct
1972def _exact_matches(content: str, labels: set[str], changed_files: set[str]) -> tuple[list, list]:
1973 """The declared `labels` / `changed_files` this file and this task share.
1975 Front matter only. A file without it is plain Markdown and falls back to text
1976 scoring, which is the whole tolerance the read path promises: learnings keel
1977 wrote and learnings a person wrote both rank.
1978 """
1979 lists = _front_matter_lists(content)
1980 matched_labels = sorted({_normalize_text(value) for value in lists.get("labels", [])} & labels)
1981 matched_files = sorted(
1982 {_normalize_path(value) for value in lists.get("changed_files", [])} & changed_files
1983 )
1984 return matched_labels, matched_files
1987def dedupe_learning_hits(
1988 hits: list[dict[str, Any]] | tuple[dict[str, Any], ...],
1989) -> list[dict[str, Any]]:
1990 """One entry per lesson, keeping the first — which is the best-ranked.
1992 Reading several directories is the point of a list, and the same lesson living
1993 in two of them is the normal way that happens: a shared knowledge folder synced
1994 into a checkout, a copy taken before a move. Left in, one lesson would take two
1995 of the five slots a brief has and push a different one out.
1997 Identity is the fingerprint, not the path: the same lesson under two names is
1998 still one lesson. A hit without one (a hand-written file) falls back to its
1999 path, which is the only thing that distinguishes it.
2000 """
2001 seen: set[str] = set()
2002 unique: list[dict[str, Any]] = []
2003 for hit in hits:
2004 key = str(hit.get("fingerprint") or hit.get("path") or "")
2005 if key in seen:
2006 continue
2007 seen.add(key)
2008 unique.append(hit)
2009 return unique
2012def _render_learning_brief(
2013 hits: list[dict[str, Any]] | tuple[dict[str, Any], ...],
2014 *,
2015 char_budget: int = LEARNING_BRIEF_CHAR_BUDGET,
2016) -> tuple[str, list[dict[str, Any]]]:
2017 """The fixed **Relevant past learnings** block and the hits that fit in budget.
2019 Empty in, empty out, and the caller renders nothing at all then — which is
2020 what keeps a project with no learnings byte-identical to the brief it got
2021 before this existed. Zero cost when unused is a requirement, not a nicety.
2022 """
2023 if not hits:
2024 return "", []
2025 lines = [f"### {LEARNING_BRIEF_HEADING}", ""]
2026 used = sum(len(line) + 1 for line in lines)
2027 rendered_hits: list[dict[str, Any]] = []
2028 for hit in hits:
2029 title = str(hit.get("title") or hit.get("file") or "Learning").strip()
2030 summary = " ".join(str(hit.get("summary") or "").split())
2031 if len(summary) > LEARNING_BRIEF_SUMMARY_CHARS:
2032 summary = summary[: LEARNING_BRIEF_SUMMARY_CHARS - 1].rstrip() + "\u2026"
2033 path = str(hit.get("path") or hit.get("file") or "")
2034 entry = f"- **{title}**" + (f" — {summary}" if summary else "") + f" (`{path}`)"
2035 if used + len(entry) + 1 > char_budget:
2036 # Stop at the budget rather than trimming the last entry into a
2037 # sentence fragment: a brief that ends mid-lesson reads as a bug.
2038 break
2039 lines.append(entry)
2040 used += len(entry) + 1
2041 rendered_hits.append(hit)
2042 if len(lines) == 2:
2043 return "", []
2044 return "\n".join(lines) + "\n", rendered_hits
2047def render_learning_brief_section(
2048 hits: list[dict[str, Any]] | tuple[dict[str, Any], ...],
2049 *,
2050 char_budget: int = LEARNING_BRIEF_CHAR_BUDGET,
2051) -> str:
2052 """The fixed **Relevant past learnings** block, or `""` when nothing matched.
2054 Empty in, empty out, and the caller renders nothing at all then — which is
2055 what keeps a project with no learnings byte-identical to the brief it got
2056 before this existed. Zero cost when unused is a requirement, not a nicety.
2057 """
2058 section, _ = _render_learning_brief(hits, char_budget=char_budget)
2059 return section
2062def learning_retrieval_as_dict(
2063 *,
2064 sources: list[str] | tuple[str, ...] = (),
2065 hits: list[dict[str, Any]] | tuple[dict[str, Any], ...] = (),
2066 limit: int = DEFAULT_LEARNING_RETRIEVAL_LIMIT,
2067 char_budget: int = LEARNING_BRIEF_CHAR_BUDGET,
2068) -> dict[str, Any]:
2069 """The retrieval block adapters read and the ledger records.
2071 `section` is rendered here rather than left to each brief's author: the
2072 implement brief and the reviewer brief must carry the *same* section, and two
2073 renderers agree only until one of them is edited.
2074 """
2075 section, rendered_hits = _render_learning_brief(hits, char_budget=char_budget)
2076 entries = [
2077 {
2078 "file": hit.get("file"),
2079 "path": hit.get("path"),
2080 "title": hit.get("title"),
2081 "summary": hit.get("summary"),
2082 "score": hit.get("score"),
2083 "fingerprint": hit.get("fingerprint"),
2084 "matched_labels": list(hit.get("matched_labels") or []),
2085 "matched_files": list(hit.get("matched_files") or []),
2086 }
2087 for hit in rendered_hits
2088 ]
2089 return {
2090 "schema_version": LEARNING_RETRIEVAL_SCHEMA_VERSION,
2091 "heading": LEARNING_BRIEF_HEADING,
2092 "policy_source": "policy_pack.capture.learning.source",
2093 "sources": list(sources),
2094 "limit": limit,
2095 "hits": entries,
2096 "fingerprints": [entry["fingerprint"] for entry in entries if entry["fingerprint"]],
2097 "section": section,
2098 }
2101def retrieve_relevant_learnings(
2102 query_text: str,
2103 learning_dir: str | Path,
2104 *,
2105 max_results: int = 3,
2106 min_score: int = 1,
2107 min_distinct_tokens: int = LEARNING_MIN_DISTINCT_TOKENS,
2108 labels: list[str] | tuple[str, ...] = (),
2109 changed_files: list[str] | tuple[str, ...] = (),
2110) -> list[dict[str, Any]]:
2111 """Retrieve relevant historical learning records for an issue or task.
2113 Stdlib-first token matching against Markdown or JSON learning files in
2114 ``learning_dir`` (e.g. ``.keel/learning/``). Returns the top matching lessons
2115 to be injected into implementation / review contexts.
2117 ``labels`` and ``changed_files`` are this task's own, and a file that
2118 *declares* one of them in its front matter is about this task rather than
2119 merely worded like it — see :data:`LEARNING_LABEL_MATCH_SCORE`. A file with no
2120 front matter is plain Markdown and scores on its text alone, which is the
2121 tolerance this reader promises: a lesson keel wrote and a lesson a person
2122 wrote both rank.
2124 The one function in this module that reads the filesystem, and it was written
2125 that way before the pure/thin-I/O split had a name for it. Everything it
2126 decides is in the pure helpers around it.
2127 """
2128 path = Path(learning_dir)
2129 if not path.is_dir():
2130 return []
2132 tokens = _learning_tokens(query_text)
2133 want_labels = {_normalize_text(label) for label in _strings(labels)}
2134 want_files = {_normalize_path(name) for name in _strings(changed_files)}
2135 if not tokens and not want_labels and not want_files:
2136 return []
2138 # **Only files that really live here.** A lesson is text an agent brief quotes, and a
2139 # link in this directory — `.keel/learning/x.md -> ~/.aws/credentials` — made whatever
2140 # it pointed at into one: `is_file` and `read_text` both follow it. Resolved on both
2141 # ends, as the landing's `_contained_real_path` does, so a link to another lesson in
2142 # the same directory still reads and a directory reached through a link still works.
2143 real_dir = path.resolve()
2144 results: list[dict[str, Any]] = []
2145 for file_path in sorted(path.glob("*")):
2146 if not file_path.is_file() or file_path.suffix not in LEARNING_READ_SUFFIXES:
2147 continue
2148 if file_path.resolve().parent != real_dir:
2149 continue
2150 try:
2151 content = file_path.read_text(encoding="utf-8", errors="replace")
2152 except OSError:
2153 continue
2155 fields, body = _front_matter(content)
2156 lists = _front_matter_lists(content)
2157 matched_labels, matched_files = _exact_matches(content, want_labels, want_files)
2158 # **The body, minus keel's own scaffolding.** Front matter is metadata with
2159 # its own exact matching, and scoring it as prose made every learning this
2160 # repository ever wrote match every task in it (`repo: keel` and `schema:`
2161 # name the project in all of them). The section headings are the same problem
2162 # one layer down: they are in every document, so a title sharing a word with
2163 # one matched the whole directory.
2164 lesson_text = _lesson_text(body, file_path.suffix)
2165 changed_paths = " ".join(lists.get("changed_files", []))
2166 if changed_paths:
2167 lesson_text = f"{lesson_text} {changed_paths.lower()}"
2168 text_score, distinct = _text_score(lesson_text, file_path.name.lower(), tokens)
2169 declared = len(matched_labels) + len(matched_files)
2170 score = (
2171 text_score
2172 + LEARNING_LABEL_MATCH_SCORE * len(matched_labels)
2173 + LEARNING_FILE_MATCH_SCORE * len(matched_files)
2174 )
2176 # **A declaration beats prose, whatever the prose says.** Added together, a
2177 # file repeating the query's words a dozen times outscored one that *declared*
2178 # the path the task touches — the ranking this whole scheme exists to get
2179 # right — because `min(count, 5)` per token compounds and the bonus does not.
2180 # It is a lexicographic key now: declared matches first, text only to break
2181 # the tie among files that declare the same number.
2182 if declared or (distinct >= min_distinct_tokens and text_score >= min_score):
2183 title, summary = _learning_title_and_summary(content, file_path.name)
2184 results.append(
2185 {
2186 "file": file_path.name,
2187 "path": str(file_path),
2188 "title": title,
2189 "summary": summary,
2190 "score": score,
2191 "declared": declared,
2192 # The writer's own fingerprint when there is one, so a later
2193 # run can tell that this exact lesson was surfaced. A file
2194 # written by hand has none; its content identifies it.
2195 "fingerprint": fields.get("fingerprint") or _content_fingerprint(content),
2196 "matched_labels": matched_labels,
2197 "matched_files": matched_files,
2198 }
2199 )
2201 results.sort(key=lambda r: (-r["declared"], -r["score"], r["file"]))
2202 return results[:max_results]
2205#: Statuses :func:`learning_land_plan` and ``keel capture-land`` speak. ``landed``
2206#: and ``already-landed`` are both success — the second is what a retry of a run
2207#: that already pushed reports, which is what makes the command idempotent.
2208LEARNING_LAND_STATUSES = (
2209 "landed",
2210 "already-landed",
2211 "not-required",
2212 "no-artifact",
2213 "would-land",
2214 "contended",
2215 "failed",
2216)
2218#: How many times the landing rebuilds its commit when a concurrent ship pushed
2219#: first. Each attempt re-reads ``origin/<base>``, so the rebuild is against the
2220#: branch as it is *now* rather than the one the run started from; three is the
2221#: same budget s6 gives CI fixes and bounds a live lock-step between two ships.
2222LEARNING_LAND_ATTEMPTS = 3
2224#: Marker on the landing commit, so the commit that carried a lesson onto the base
2225#: branch can be found by the pull request it came from without parsing prose.
2226LEARNING_LAND_SCHEMA_VERSION = "keel.capture-land.v1"
2228#: The line the landing commit carries, and it is the **schema version**.
2229#:
2230#: This commit is the one thing keel pushes to a base branch outside a pull request,
2231#: so a history reader — `git log`, `keel verify-merge`'s drift read, a person asking
2232#: what this commit is — has to be able to tell it from a stray push. Naming it after
2233#: the schema rather than inventing a second literal means the marker and the record
2234#: cannot drift apart, and the name is greppable against the contract that defines it.
2235LEARNING_LAND_MARKER = LEARNING_LAND_SCHEMA_VERSION
2238def learning_land_message(
2239 *, pr_number: int | None, path: str, issue_number: int | None = None
2240) -> str:
2241 """The landing commit's message — deterministic, so two runs agree byte for byte.
2243 Consumer-neutral on purpose: it carries no vendor trailer. The commit is keel's,
2244 made on behalf of whatever agent ran the ship, and a core command that stamped one
2245 vendor's co-authorship onto every consumer's base branch would be asserting an
2246 attribution it cannot know. The run ledger already records who implemented.
2247 """
2248 subject = (
2249 f"chore(learning): record the lesson from PR #{pr_number}"
2250 if pr_number is not None
2251 else "chore(learning): record the lesson from this run"
2252 )
2253 fields = " ".join(
2254 f"{name}={value if value is not None else '-'}"
2255 for name, value in (("pr", pr_number), ("issue", issue_number), ("path", path))
2256 )
2257 return f"{subject}\n\n{LEARNING_LAND_MARKER}: {fields}\n"
2260def learning_land_plan(
2261 config: _HasPolicyPack | None,
2262 *,
2263 artifact: str | None,
2264 pr_number: int | None = None,
2265 issue_number: int | None = None,
2266 remote: str = "origin",
2267 base_branch: str | None = None,
2268 onto: str | None = None,
2269 attempts: int = LEARNING_LAND_ATTEMPTS,
2270) -> dict[str, Any]:
2271 """Plan the landing of one learning artifact onto a branch.
2273 ``onto`` names it — the pull request's own under `/keel:ship` (#1203) — and without it
2274 the target is ``origin/<base_branch>`` (#1163).
2276 Pure: it reads the config's *shape* and the artifact's path and answers what the
2277 I/O layer should do, the way :func:`learning_sink_plan` answers what the writer
2278 should write. Nothing here touches git or the filesystem, so the decision that
2279 keel pushes to a base branch at all is unit-tested offline.
2281 The plan deliberately describes **plumbing, not a checkout**. keel runs its own
2282 s0–s12 inside a worktree while the primary checkout holds the base branch, so
2283 every recipe built on ``git switch <base>`` exits 128 with *'<base>' is already
2284 used by worktree* — the defect #1163 was opened for. Building the commit with
2285 ``hash-object`` / ``ls-tree`` / ``mktree`` / ``commit-tree`` against
2286 ``<remote>/<base>`` needs no checkout of the base branch at all, so it runs the
2287 same from a worktree, from the primary checkout, and from a bare CI clone.
2289 ``status`` is ``not-required`` when the sink is outside the repository (git never
2290 sees it, so there is nothing to land), ``no-artifact`` when no path was recorded,
2291 and ``planned`` when the caller should go ahead. ``errors`` is non-empty only for
2292 a path this command must refuse to write to the base branch.
2293 """
2294 resolved_base = base_branch or getattr(config, "base_branch", None) or ""
2295 # **The branch the commit goes to, kept apart from the base branch.** #1203 lands the
2296 # lesson on the pull request's own branch, as its last commit, so it merges with the
2297 # work it describes. The sink template still expands against the *base* branch —
2298 # that is the value the writer used — so the two cannot share one variable.
2299 target = onto or resolved_base
2300 if not learning_sink_in_worktree(config):
2301 return {
2302 "schema_version": LEARNING_LAND_SCHEMA_VERSION,
2303 "status": "not-required",
2304 "reason": (
2305 "the learning sink is outside the checkout, so git never sees it and "
2306 "nothing has to reach the base branch"
2307 ),
2308 "path": None,
2309 "remote": remote,
2310 "base_branch": resolved_base,
2311 "onto": None,
2312 "ref": None,
2313 "remote_ref": None,
2314 "message": None,
2315 "attempts": attempts,
2316 "errors": [],
2317 }
2318 normalized = _land_path(artifact)
2319 sink_dir = _land_sink_root(config, pr_number=pr_number, base_branch=resolved_base)
2320 errors: list[str] = []
2321 if not is_remote_name(remote) or (target and not is_branch_name(target)):
2322 status = "failed"
2323 reason = "the remote or the target branch is not a name git can take as one"
2324 # **Refused before any git call sees it, by git's own rules.** Both reach
2325 # `git fetch <remote> <ref>` as positional arguments. A leading `-` is read there as
2326 # an option — `--upload-pack=<program>` runs a program — and a `:` makes the branch
2327 # a two-sided refspec: `foo:refs/heads/main` moves the local `main` to `foo`'s tip.
2328 # Refusing only the `-` shipped first and was not enough. Answered here, in the one
2329 # place every landing is planned, rather than trusted to each wrapper downstream.
2330 errors.append(f"remote {remote!r} or target {target!r} is not a valid name")
2331 elif artifact is None or not str(artifact).strip():
2332 status = "no-artifact"
2333 reason = "no capture artifact was recorded for this run, so there is nothing to land"
2334 elif normalized is None:
2335 status = "failed"
2336 reason = "the capture artifact is not a path inside the repository"
2337 # Refused rather than normalised. The command's whole job is to push one file
2338 # to a shared base branch, so an artifact that climbs out of the checkout —
2339 # `../x`, `/etc/x`, `C:\x` — is the one input that must never be resolved
2340 # helpfully. `learning_sink_in_worktree` says the *sink* is in-repo; this says
2341 # the recorded path is too, and they are answered from different values.
2342 errors.append(f"capture artifact {artifact!r} is absolute or escapes the repository root")
2343 elif sink_dir is None:
2344 status = "failed"
2345 reason = (
2346 "the learning sink path does not name a directory this command can resolve, "
2347 "so there is nothing to confine the landing to"
2348 )
2349 # A sink of `.`, or one whose directory still holds `{date}`/`{slug}`/
2350 # `{fingerprint}` after `{owner}`/`{repo}`/`{base_branch}`/`{pr}` are filled in.
2351 # Both describe a boundary that would accept any path in the repository, which is
2352 # the boundary this check exists to replace.
2353 errors.append("the learning sink path cannot be resolved to a directory")
2354 elif not path_under_sink(normalized, sink_dir):
2355 status = "failed"
2356 reason = f"the capture artifact is not inside the learning sink ({sink_dir})"
2357 # **Inside the repository is not the containment this command needs.** Every
2358 # path test above answers "could git address this?", and the answer is yes for
2359 # `config/private.env` and for `src/keel/cli.py` alike — so a ledger record
2360 # naming one of those fast-forwarded the shared base branch with it, one file
2361 # at a time, and the exactly-one-file check downstream agreed because it was
2362 # exactly one file. The sink is the only directory this command has any
2363 # business writing to, and it is the one value that says which.
2364 errors.append(
2365 f"capture artifact {artifact!r} is not under the configured learning sink {sink_dir!r}"
2366 )
2367 elif not resolved_base:
2368 status = "failed"
2369 reason = "the project declares no base branch to land the lesson on"
2370 errors.append("base_branch is not configured")
2371 else:
2372 status = "planned"
2373 reason = f"land {normalized} on {remote}/{target}"
2374 return {
2375 "schema_version": LEARNING_LAND_SCHEMA_VERSION,
2376 "status": status,
2377 "reason": reason,
2378 "path": normalized,
2379 # Published so the I/O layer can resolve the *real* artifact against the sink
2380 # rather than against the checkout: a link inside the sink pointing at an
2381 # untracked `.env` beside the code is in the repository, and a checkout-wide
2382 # containment test says yes to it.
2383 "sink": sink_dir,
2384 "remote": remote,
2385 "base_branch": resolved_base,
2386 "onto": target or None,
2387 "ref": f"refs/heads/{target}" if target else None,
2388 # Spelled in full, as `git.remote_tracking_ref` spells it and `git.fetch` writes it:
2389 # the short `<remote>/<target>` resolves a tag or local branch of that name first,
2390 # and the landing builds its commit on whatever this resolves to.
2391 "remote_ref": f"refs/remotes/{remote}/{target}" if target else None,
2392 "message": (
2393 learning_land_message(pr_number=pr_number, path=normalized, issue_number=issue_number)
2394 if status == "planned"
2395 else None
2396 ),
2397 "attempts": attempts,
2398 "errors": errors,
2399 }
2402def _land_sink_root(
2403 config: _HasPolicyPack | None, *, pr_number: int | None, base_branch: str
2404) -> str | None:
2405 """The directory the landing is allowed to write inside, as a repo-relative path.
2407 The sink's ``path`` is a **template**, and `learning_sink_plan` writes the expanded
2408 form — so comparing a recorded artifact against the literal `.keel/{repo}/learning`
2409 refuses every lesson a project with placeholders ever writes. It is expanded here
2410 with the same values, through the same `_relative_stays_relative` that keeps a
2411 relative template relative when a leading placeholder expands to nothing.
2413 Only the **directory-shaped** placeholders are resolvable here, and this command has
2414 all four of them. One that still holds ``{date}``, ``{slug}`` or ``{fingerprint}``
2415 afterwards names a place nobody can point at, and the answer is ``None`` — the caller
2416 refuses rather than guessing. Two earlier shapes were tried and are recorded because
2417 both looked reasonable: confining to the known *prefix* accepts anything under
2418 ``.keel/``, and matching an unresolved component as a *wildcard* is not a boundary at
2419 all — a sink of ``{date}`` then makes the first component match anything, so
2420 ``config/private.env`` is "inside" it.
2421 """
2422 # `or {}` rather than a guard: the one caller reaches this only after
2423 # `learning_sink_in_worktree` said there *is* a sink, and an empty block takes the
2424 # documented default anyway — a branch no input can take is a claim the tests
2425 # cannot check.
2426 sink = learning_sink_policy(config) or {}
2427 values = {
2428 "owner": str(getattr(config, "owner", "") or ""),
2429 "repo": str(getattr(config, "repo", "") or ""),
2430 "base_branch": base_branch,
2431 "pr": str(pr_number) if pr_number is not None else "",
2432 }
2433 expanded = _relative_stays_relative(str(sink.get("path") or DEFAULT_LEARNING_SINK_PATH), values)
2434 # **Every placeholder resolved, or no sink root at all.** `{owner}`, `{repo}`,
2435 # `{base_branch}` and `{pr}` are the directory-shaped ones and this command has all
2436 # four. `{date}`, `{slug}` and `{fingerprint}` vary per run and belong in `filename`;
2437 # left in a directory they name a place nobody can point at, and the landing says so
2438 # rather than guessing — a wildcard component is not a boundary, it is a hole.
2439 return None if "{" in expanded else _land_path(expanded)
2442def path_under_sink(path: str, directory: str) -> bool:
2443 """Is POSIX ``path`` inside ``directory``? Compared component by component.
2445 Not a prefix test: a plain ``startswith`` says `.keel/learning-notes/x.md` is inside
2446 `.keel/learning`, which is a different directory whose name merely begins the same way.
2448 Every component of ``directory`` is a literal. A wildcard component was tried and is
2449 not a boundary at all — a sink of `{date}` makes the first component match anything,
2450 so `config/private.env` is "inside" it, and `{date}/learning` accepts
2451 `config/learning/secrets.md`. The landing refuses a sink it cannot resolve instead,
2452 so nothing here has to guess.
2453 """
2454 wanted, have = directory.split("/"), path.split("/")
2455 # Strictly deeper: the sink directory is not a file inside itself.
2456 if len(have) <= len(wanted):
2457 return False
2458 return all(want == got for want, got in zip(wanted, have, strict=False))
2461#: The line a landing commit's message carries, as the prefix a reader matches on.
2462LEARNING_LAND_MARKER_LINE = f"{LEARNING_LAND_MARKER}:"
2465def land_sink_root(
2466 config: _HasPolicyPack | None, *, pr_number: int | None, base_branch: str
2467) -> str | None:
2468 """The public name for :func:`_land_sink_root`, for the readers outside this module.
2470 The evidence gate and the merge gate ask the same containment question the landing
2471 asks, and a second copy of the answer is how the three would come to disagree.
2472 """
2473 return _land_sink_root(config, pr_number=pr_number, base_branch=base_branch)
2476def lesson_changed_files(
2477 config: _HasPolicyPack | None,
2478 changed_files: list[str] | tuple[str, ...],
2479 *,
2480 pr_number: int | None,
2481 base_branch: str,
2482) -> list[str]:
2483 """The files a lesson is about: the pull request's, less the lessons in the sink (#1203).
2485 The landing puts the lesson on the pull request itself, so once it has landed the host
2486 lists the lesson among the files that pull request changed. The document is written
2487 before the landing and the s11 record after the merge, and both read the host's list:
2488 without this the record fingerprinted one more path than the document it names — the
2489 mismatch the shared list exists to prevent — and a retried s10 rendered a lesson whose
2490 **Files** section named the lesson.
2492 Only an in-repo sink is subtracted, and only for a pull request. A sink outside the
2493 checkout never appears in a pull request's files, and one whose path cannot be resolved
2494 has no boundary to subtract by. Without a pull request there is no landing to subtract —
2495 and a ``{pr}`` sink resolved with none reads as its parent (``docs/{pr}`` as ``docs``),
2496 which would take the run's real work in ``docs/`` out of its own lesson.
2497 """
2498 paths = list(changed_files)
2499 if pr_number is None or not learning_sink_in_worktree(config):
2500 return paths
2501 sink = _land_sink_root(config, pr_number=pr_number, base_branch=base_branch)
2502 if sink is None:
2503 return paths
2504 return [path for path in paths if not path_under_sink(path, sink)]
2507#: Characters `git check-ref-format` refuses anywhere in a ref name.
2508_REF_FORBIDDEN = re.compile(r"[\x00-\x20\x7f~^:?*\[\\]")
2510#: A remote *name* — what `--remote` documents — rather than a URL or a refspec.
2511_REMOTE_NAME = re.compile(r"\A[A-Za-z0-9][A-Za-z0-9._-]*\Z")
2514def is_branch_name(name: object) -> bool:
2515 """Would ``git check-ref-format --branch`` accept ``name`` as a literal branch? (#1203)
2517 Pure, so the landing plan can refuse a bad target before any git call sees it. The
2518 rules are git's own, and the test holds this function to git's verdict case by case
2519 rather than to a reading of the man page:
2521 - not empty, not beginning with ``-`` (git reads that as an option) or ``/``;
2522 - no ``:`` — ``foo:refs/heads/main`` is a two-sided refspec, and handed to
2523 ``git fetch`` it moves the local ``main`` to another branch's tip (measured);
2524 - no control character, space, ``~``, ``^``, ``?``, ``*``, ``[`` or backslash;
2525 - no ``..``, no ``@{``, no ``//``, not ending in ``/`` or ``.``;
2526 - no path component beginning with ``.`` or ending in ``.lock``.
2528 ``HEAD`` is refused, as git refuses it. ``@`` alone is refused although git accepts it:
2529 ``--branch`` expands it to the current branch, which is not a name a caller can mean as a
2530 literal target.
2531 """
2532 if not isinstance(name, str) or not name or name in ("@", "HEAD"):
2533 return False
2534 if name[0] in "-/" or name.endswith(("/", ".")):
2535 return False
2536 if _REF_FORBIDDEN.search(name) or ".." in name or "@{" in name or "//" in name:
2537 return False
2538 return not any(part.startswith(".") or part.endswith(".lock") for part in name.split("/"))
2541def is_remote_name(name: object) -> bool:
2542 """Is ``name`` a plain remote name — letters, digits, ``.``, ``_``, ``-``, no leading ``-``?"""
2543 return isinstance(name, str) and bool(_REMOTE_NAME.match(name))
2546#: What a landing commit may do to its one path: write a new lesson, or rewrite one at the
2547#: same path. Never rename, copy or remove — those are other changes wearing the marker.
2548LANDING_FILE_STATUSES = ("added", "modified")
2551def carries_landing_marker(message: object) -> bool:
2552 """Does a commit message carry the ``keel.capture-land.v1:`` marker line?
2554 One predicate for both of its readers: the head-pin exemption below, and the search for
2555 a lesson already riding a pull request (``keel capture-land --write``). Two copies of the
2556 match would drift apart, and the search would then find landings the exemption refuses.
2557 """
2558 return isinstance(message, str) and any(
2559 line.strip().startswith(LEARNING_LAND_MARKER_LINE) for line in message.splitlines()
2560 )
2563def capture_only_descent(
2564 base: str,
2565 head: str,
2566 commits: list[dict[str, Any]],
2567 *,
2568 sink: str | None,
2569) -> bool:
2570 """Does ``head`` descend from ``base`` by **capture commits and nothing else**? (#1203)
2572 The question the head-pin exemption rests on. Every review verdict and every
2573 gates-pass is pinned to the head it was recorded against, and #1203 puts the
2574 learning on the pull request's own branch as its last commit — after review, before
2575 the merge — which moves that head. A verdict for ``base`` still answers for ``head``
2576 exactly when nothing between them could change what was reviewed:
2578 - every commit has **exactly one parent**, and they form an unbroken chain from
2579 ``base`` to ``head`` — a merge commit could carry anything;
2580 - every commit's message carries the ``keel.capture-land.v1:`` marker line — the
2581 exemption is for commits that *say* they are a landing, not for any edit that
2582 happens to touch the sink;
2583 - every commit differs from its parent by **exactly one path**, and that path is
2584 inside the configured sink — the same containment the landing enforces before it
2585 pushes, asked again here of commits it did not necessarily build;
2586 - and that one path was **added or modified** — not renamed, copied or removed.
2588 ``commits`` is ordered oldest to newest, each a mapping with ``sha``, ``parents``
2589 (a list of SHAs), ``message``, ``files`` (every path the commit touches, a rename's
2590 source included) and ``statuses`` (one per changed entry). Anything malformed or
2591 absent is ``False``: this answer removes a requirement, so it fails closed.
2593 A sink that cannot be resolved is ``False`` too. There is then no boundary to hold a
2594 commit to, and "inside the sink" would silently mean "anywhere".
2595 """
2596 if not base or not head or base == head or sink is None or not commits:
2597 return False
2598 previous = base
2599 for commit in commits:
2600 if not isinstance(commit, dict):
2601 return False
2602 parents = commit.get("parents")
2603 if not isinstance(parents, list) or parents != [previous]:
2604 return False
2605 if not carries_landing_marker(commit.get("message")):
2606 return False
2607 files = commit.get("files")
2608 if not isinstance(files, list) or len(files) != 1 or not isinstance(files[0], str):
2609 return False
2610 if not path_under_sink(files[0], sink):
2611 return False
2612 # **What happened to that path, not only which path it was.** A rename or a copy
2613 # into the sink arrives as one entry naming a path inside it; `files` counts the
2614 # path it came from too, and this refuses the status outright, so neither a moved
2615 # reviewed file nor a deleted lesson reads as a landing. Absent is refused as well:
2616 # a reader that cannot say what the commit did has not shown it only *added*.
2617 statuses = commit.get("statuses")
2618 if not isinstance(statuses, list) or len(statuses) != 1:
2619 return False
2620 if statuses[0] not in LANDING_FILE_STATUSES:
2621 return False
2622 sha = commit.get("sha")
2623 if not isinstance(sha, str) or not sha:
2624 return False
2625 previous = sha
2626 return previous == head
2629def _land_path(artifact: str | None) -> str | None:
2630 """``artifact`` as a repo-relative POSIX path, or ``None`` when it is not one.
2632 Anchored on *any* platform for the same reason
2633 :func:`learning_sink_in_worktree` is: a keel config and a keel ledger are the
2634 same text wherever they are read, and a path this host calls relative is the
2635 one the next host would push to its base branch.
2636 """
2637 if artifact is None:
2638 return None
2639 raw = str(artifact).strip()
2640 if not raw or raw.startswith("~"):
2641 return None
2642 normalized = posixpath.normpath(raw.replace("\\", "/"))
2643 # **Both forms are tested, and the rewrite is why.** `is_root_anchored` asks each
2644 # flavour of path whether it is absolute, and `PureWindowsPath("\\etc\\hostname")`
2645 # says no — a rooted path with no drive letter is *drive-relative*, not absolute.
2646 # The backslash rewrite then turned that same string into `/etc/hostname`, which
2647 # `os.path.join(root, …)` resolves by discarding root entirely: `git hash-object -w`
2648 # stored a file from outside the checkout in the object database, and the tree
2649 # composition split it into an entry with an empty name. Asking the question after
2650 # the rewrite as well as before is what closes it.
2651 if workspace.is_root_anchored(raw) or workspace.is_root_anchored(normalized):
2652 return None
2653 if normalized in (".", "") or normalized.split("/")[0] == "..":
2654 return None
2655 return normalized
2658#: git's own words for "the ref moved under you" — the only rejection worth retrying.
2659#: Printed by the client when its remote-tracking ref is behind, and by the server in
2660#: the reason it returns; both reach us through the same combined output.
2661PUSH_CONTENTION_MARKERS = ("fetch first", "non-fast-forward", "non-fast forward")
2664def push_rejection_is_contention(output: str | None) -> bool:
2665 """Did this push fail because the ref **moved**, or because it is **refused**?
2667 The retry exists for one case: another ship landed its lesson between this run's
2668 read of ``<remote>/<base>`` and its push. Rebuilding on the branch as it now is
2669 will then succeed, which is why that case retries.
2671 Every other rejection will refuse again, and a protected base branch is the
2672 ordinary one — ``! [remote rejected] … (protected branch hook declined)``, or a
2673 ``pre-receive`` hook's own sentence. Treating it as contention burned three pushes
2674 on something that cannot succeed and then reported that the branch *"moved under
2675 every one of 3 attempt(s)"* — naming a cause that did not happen and hiding the
2676 server's actual reason, which is the one thing the operator needs.
2678 Judged from git's text because the exit code is 1 either way. An unrecognised
2679 failure is **not** contention: this pushes to the shared base branch, so an
2680 unexplained refusal stops after one attempt rather than being retried on a guess.
2681 """
2682 text = (output or "").lower()
2683 return any(marker in text for marker in PUSH_CONTENTION_MARKERS)
2686#: One ``git ls-tree`` / ``git mktree`` line: ``<mode> SP <type> SP <sha> TAB <name>``.
2687#: ``DOTALL`` because a name may contain a newline, which ``-z`` returns raw.
2688_TREE_ENTRY_RE = re.compile(r"\A(\d{6}) (blob|tree|commit) ([0-9a-f]{40,64})\t(.+)\Z", re.DOTALL)
2690#: git's own mode for a regular, non-executable file and for a subdirectory.
2691TREE_MODE_BLOB = "100644"
2692TREE_MODE_TREE = "040000"
2695@dataclass(frozen=True)
2696class TreeEntry:
2697 """One entry of a git tree, as ``ls-tree`` prints it and ``mktree`` reads it."""
2699 mode: str
2700 kind: str
2701 sha: str
2702 name: str
2704 def render(self) -> str:
2705 return f"{self.mode} {self.kind} {self.sha}\t{self.name}"
2708def parse_tree_listing(listing: str | None) -> list[TreeEntry]:
2709 """Parse ``git ls-tree -z`` output; unreadable records are dropped, not guessed at.
2711 A missing directory is an empty listing rather than an error: landing a lesson
2712 into a sink directory that does not exist on the base branch yet is the ordinary
2713 first run, not a failure.
2715 Records are NUL-separated (see :func:`keel.git.ls_tree`). A listing with no NUL in it
2716 at all is read one record per line, so one that reached this from a LF-terminated
2717 source still parses — the reader is permissive, the *writer* is the side that has to
2718 be exact.
2720 **Never both.** git allows a newline inside a name and ``-z`` returns it raw, so
2721 splitting a ``-z`` listing on newlines as well cut that record in two; neither half
2722 parsed, the entry was dropped, and the tree rebuilt from what was left no longer had
2723 it — a landing commit that deleted a file it was never asked to touch. Without ``-z``
2724 git C-quotes such a name, so the line-per-record form has no raw newline to cut.
2725 """
2726 text = listing or ""
2727 entries: list[TreeEntry] = []
2728 for line in text.split("\0") if "\0" in text else text.split("\n"):
2729 match = _TREE_ENTRY_RE.match(line)
2730 if match is not None:
2731 entries.append(
2732 TreeEntry(match.group(1), match.group(2), match.group(3), match.group(4))
2733 )
2734 return entries
2737def _tree_sort_key(entry: TreeEntry) -> str:
2738 # git orders tree entries as if every directory name ended in "/", which is why
2739 # "learning" and "learning.md" sort the way they do. `mktree` normalises the order
2740 # itself; composing it correctly here keeps the pure result byte-stable so a test
2741 # can assert the exact input the plumbing is handed.
2742 return entry.name + "/" if entry.kind == "tree" else entry.name
2745def upsert_tree_entry(listing: str | None, entry: TreeEntry) -> str:
2746 """``listing`` with ``entry`` added or replacing the one of the same name.
2748 Pure, and the whole reason the landing can be asserted offline: this is where a
2749 lesson is grafted onto the base branch's tree, so "the commit differs from its
2750 parent by exactly this one path" is a property of a string function rather than
2751 of a live push nobody can re-run.
2753 **NUL-terminated**, because the result is written to a subprocess's stdin and a
2754 text-mode pipe rewrites ``\n`` as CRLF on Windows. ``git mktree`` accepts the
2755 corrupted listing without complaint and writes a tree whose entry is named
2756 ``<name>\r`` — a different SHA, exit 0, no error (measured). There is no newline
2757 in this output to translate.
2758 """
2759 kept = [existing for existing in parse_tree_listing(listing) if existing.name != entry.name]
2760 kept.append(entry)
2761 kept.sort(key=_tree_sort_key)
2762 return "".join(f"{item.render()}\x00" for item in kept)
2765def _content_fingerprint(content: str) -> str:
2766 import hashlib
2768 return hashlib.sha256(content.encode("utf-8")).hexdigest()