Coverage for src/keel/capture.py: 100%

814 statements  

« prev     ^ index     » next       coverage.py v7.16.2, created at 2026-10-02 20:26 +0000

1"""Consumer-neutral post-merge capture contract and verification helpers.""" 

2 

3from __future__ import annotations 

4 

5import json 

6import os.path 

7import posixpath 

8import re 

9from dataclasses import dataclass 

10from pathlib import Path 

11from typing import Any, Protocol 

12from urllib.parse import quote 

13 

14# `workspace` imports nothing from this package's config layer, so naming it here 

15# keeps the import graph acyclic — see `_HasPolicyPack` for why that matters. 

16from . import workspace 

17from . import yaml_helper as yaml 

18 

19 

20class _HasPolicyPack(Protocol): 

21 """Duck type for a loaded ``ProjectConfig``. 

22 

23 This module only reads ``policy_pack``. Naming ``config.ProjectConfig`` here 

24 would import ``config``, and ``config`` already imports this module to 

25 validate ``policy_pack.capture.learning``. CodeQL counts even a 

26 ``TYPE_CHECKING`` import as that reverse edge — the cycle it reports as 

27 ``py/cyclic-import``, because ``ProjectConfig`` is defined *after* 

28 ``config``'s import of ``capture``. 

29 """ 

30 

31 policy_pack: Any 

32 

33 

34CAPTURE_SCHEMA_VERSION = "keel.capture.v1" 

35RECONCILE_SCHEMA_VERSION = "keel.capture-reconcile.v1" 

36LEARNING_DECISION_SCHEMA_VERSION = "keel.capture-learning.v1" 

37MARKER_PREFIX = "compound-learning" 

38STATUSES = ("applied", "deferred", "skipped") 

39SKIP_REASONS = ( 

40 "dry-run", 

41 "deferred", 

42 "merge-failed", 

43 "recursion-guard", 

44 "capability-unavailable", 

45 "no-policy", 

46) 

47LEARNING_DECISIONS = ("create-learning", "marker-only", "defer", "duplicate") 

48 

49_MARKER_RE = re.compile( 

50 r"^compound-learning:\s+pr=(?P<pr>[1-9][0-9]*)\s+status=" 

51 r"(?P<status>applied|deferred|skipped(?::[a-z0-9-]+)?)$" 

52) 

53 

54 

55class CaptureError(ValueError): 

56 """Raised when a capture marker or capture record is invalid.""" 

57 

58 

59@dataclass(frozen=True) 

60class CaptureMarker: 

61 """One stable capture marker emitted after a merged PR.""" 

62 

63 pr_number: int 

64 status: str 

65 reason: str | None = None 

66 

67 def as_text(self) -> str: 

68 return marker_text( 

69 pr_number=self.pr_number, 

70 status=self.status, 

71 reason=self.reason, 

72 ) 

73 

74 def as_dict(self) -> dict[str, Any]: 

75 return { 

76 "schema_version": CAPTURE_SCHEMA_VERSION, 

77 "prefix": MARKER_PREFIX, 

78 "pr": self.pr_number, 

79 "status": self.status, 

80 "reason": self.reason, 

81 "text": self.as_text(), 

82 } 

83 

84 

85def contract_as_dict(config: _HasPolicyPack | None = None) -> dict[str, Any]: 

86 """Return the stable capture contract consumed by adapters and verifiers.""" 

87 capture_policy = _capture_policy(config) 

88 # **The sink core will actually use**, not merely one written down. A dormant 

89 # `sink:` block under `capture.enabled: false` or `mode: marker-only` published 

90 # `project_destination: "sink"` — telling an adapter core would write — while 

91 # `learning_sink_writes` refused, so neither wrote and the contract had promised 

92 # one of them would. Third reader of the same question; they all ask it here. 

93 sink_policy = learning_sink_policy(config) if capture_hook_enabled(config) else None 

94 return { 

95 "schema_version": CAPTURE_SCHEMA_VERSION, 

96 "marker": { 

97 "prefix": MARKER_PREFIX, 

98 "format": "compound-learning: pr=<N> status=<applied|deferred|skipped:reason>", 

99 "statuses": list(STATUSES), 

100 "skip_reasons": list(SKIP_REASONS), 

101 "required_after_merged_pr": True, 

102 }, 

103 "extension_slots": ["capture", "post-merge"], 

104 "policy_source": "policy_pack.capture + capture/post-merge extensions", 

105 "policy_enabled": bool(capture_policy.get("enabled", False)), 

106 "policy_mode": capture_policy.get("mode", "extension"), 

107 "recursion_guard": { 

108 "enabled": True, 

109 "reason": "recursion-guard", 

110 "never_capture_capture_work": True, 

111 }, 

112 "fail_soft": { 

113 "enabled": True, 

114 "merge_revert_on_capture_failure": False, 

115 "failure_marker": "skipped:capability-unavailable", 

116 }, 

117 "durable_artifacts": { 

118 "requires_redaction": True, 

119 "redaction_contract": "run_ledger.capture_redaction", 

120 "core_destination": "run-ledger", 

121 # `extension-owned` was the whole truth until keel shipped a writer. 

122 # A project that configures `learning.sink` has **core** writing the 

123 # file and filling `capture.artifact`, and an adapter that read this 

124 # block and wrote its own would have had it overwritten (#1154). 

125 "project_destination": "sink" if sink_policy is not None else "extension-owned", 

126 # Defaulted the way `learning_sink_plan` defaults it. Read without the 

127 # fallback, a sink that did not spell out its `kind` — which the schema, 

128 # the docs and the validator all allow — published 

129 # `{project_destination: sink, sink: None}`: a contract disagreeing with 

130 # itself about the writer it had just named. 

131 "sink": ( 

132 sink_policy.get("kind", LEARNING_SINK_KINDS[0]) if sink_policy is not None else None 

133 ), 

134 # **Whether the file lands in the working tree, and so has to be 

135 # committed.** keel writes it and stops there. A relative sink path is 

136 # inside the repository, where an uncommitted file is one the next 

137 # worktree — cut from `origin/<base>` — and every CI runner never see: 

138 # keel would be writing a learning and then throwing it away, which is 

139 # the failure classifying `.keel/learning` as committed was for. An 

140 # absolute or `~` path is outside the checkout and git never sees it. 

141 "commit_required": learning_sink_in_worktree(config), 

142 # **How a `commit_required` artifact reaches the base branch (#1163).** 

143 # `commit_required` answered *whether* the file has to be committed and 

144 # nothing answered *how*, so keel wrote a learning on every merge and 

145 # threw it away: s2 cuts the next worktree from `origin/<base>`, every CI 

146 # runner clones fresh, and s10's pre-clean deletes the worktree outright. 

147 # The recipe cannot be "switch to the base branch and commit" — s2, 

148 # `overnight` and `swarm` all run inside a worktree while the primary 

149 # checkout holds the base branch, so `git switch <base>` there exits 128 

150 # with *'<base>' is already used by worktree*. `keel capture-land` builds 

151 # the commit with plumbing against `origin/<base>` and never checks it 

152 # out, which is why it is topology-independent. 

153 "land_command": "keel capture-land" if learning_sink_in_worktree(config) else None, 

154 }, 

155 "learning_quality": learning_quality_contract_as_dict(config), 

156 "learning_retrieval": learning_retrieval_contract_as_dict(config), 

157 "session_end_verifier": { 

158 "primitive": "capture.verify_session", 

159 "cli": "keel capture-verify", 

160 "missing_marker_status": "missing", 

161 "invalid_marker_status": "invalid", 

162 }, 

163 "reconcile": { 

164 "schema_version": RECONCILE_SCHEMA_VERSION, 

165 "primitive": "capture.reconcile_session", 

166 "cli": "keel capture-reconcile", 

167 "idempotent": True, 

168 "never_reopens_implementation": True, 

169 "never_pushes_code": True, 

170 "never_merges_prs": True, 

171 "actions": [ 

172 "emit-capture-marker", 

173 "run-capture-extension", 

174 "post-closure-summary", 

175 "close-linked-issue", 

176 "record-skip", 

177 ], 

178 }, 

179 } 

180 

181 

182def learning_retrieval_contract_as_dict( 

183 config: _HasPolicyPack | None = None, 

184) -> dict[str, Any]: 

185 """The read side of capture, declared so an adapter knows the section exists. 

186 

187 Policy only — where this project reads learnings from and how many a brief 

188 carries. What was actually *found* is measured per run and travels on the ship 

189 contract, because reading a directory is I/O and this contract is pure. 

190 """ 

191 return { 

192 "schema_version": LEARNING_RETRIEVAL_SCHEMA_VERSION, 

193 "policy_source": "policy_pack.capture.learning.source", 

194 # As **configured**, not as resolved: this contract is pure and has no 

195 # `{repo}` to expand with, so a resolved list would report an empty 

196 # `sources` for a templated setting that reads perfectly well at run time. 

197 "sources": learning_source_entries(config), 

198 "limit": DEFAULT_LEARNING_RETRIEVAL_LIMIT, 

199 "heading": LEARNING_BRIEF_HEADING, 

200 "briefs": ["implement", "review"], 

201 "reader": "capture.retrieve_relevant_learnings", 

202 "ledger_field": "capture.retrieved", 

203 "silent_when_empty": True, 

204 } 

205 

206 

207def learning_quality_contract_as_dict(config: _HasPolicyPack | None = None) -> dict[str, Any]: 

208 """Return the consumer-neutral durable-learning quality contract.""" 

209 policy = _learning_policy(config) 

210 dedupe = policy.get("dedupe") if isinstance(policy.get("dedupe"), dict) else {} 

211 return { 

212 "schema_version": LEARNING_DECISION_SCHEMA_VERSION, 

213 "decisions": list(LEARNING_DECISIONS), 

214 "policy_source": "policy_pack.capture.learning", 

215 "policy_enabled": bool(policy.get("enabled", False)), 

216 "policy_mode": policy.get("mode", "policy-unavailable"), 

217 "default_decision": "marker-only", 

218 "default_reason": "policy-unavailable", 

219 "marker_required_for_every_merge": True, 

220 "durable_learning_optional": True, 

221 "dedupe": { 

222 "enabled": bool(dedupe.get("enabled", True)), 

223 "fingerprint": "sha256(normalized title + labels + changed files)", 

224 "matching": "stable fingerprint plus configured matching rules", 

225 }, 

226 "ledger_field": "capture.learning", 

227 "closure_summary_field": "Capture", 

228 } 

229 

230 

231def marker_text(*, pr_number: int, status: str, reason: str | None = None) -> str: 

232 """Render one stable capture marker.""" 

233 marker = build_marker(pr_number=pr_number, status=status, reason=reason) 

234 suffix = marker.status if marker.reason is None else f"{marker.status}:{marker.reason}" 

235 return f"{MARKER_PREFIX}: pr={marker.pr_number} status={suffix}" 

236 

237 

238def build_marker(*, pr_number: int, status: str, reason: str | None = None) -> CaptureMarker: 

239 """Validate and build a capture marker.""" 

240 if pr_number <= 0: 

241 raise CaptureError("capture marker requires a positive PR number") 

242 status, reason = normalize_status(status, reason) 

243 return CaptureMarker(pr_number=pr_number, status=status, reason=reason) 

244 

245 

246def normalize_status(status: str | None, reason: str | None = None) -> tuple[str, str | None]: 

247 """Normalize ``skipped:<reason>`` into a structured status and reason.""" 

248 if not status: 

249 raise CaptureError("capture status is required") 

250 raw = status.strip() 

251 if raw.startswith("skipped:"): 

252 raw, embedded_reason = raw.split(":", 1) 

253 reason = embedded_reason 

254 if raw not in STATUSES: 

255 raise CaptureError(f"unsupported capture status: {status}") 

256 clean_reason = reason.strip() if isinstance(reason, str) and reason.strip() else None 

257 if raw == "skipped": 

258 if clean_reason not in SKIP_REASONS: 

259 raise CaptureError("skipped capture requires an allowed skip reason") 

260 else: 

261 clean_reason = None 

262 return raw, clean_reason 

263 

264 

265def parse_marker(text: str) -> CaptureMarker: 

266 """Parse a stable marker string into structured data.""" 

267 match = _MARKER_RE.match(text.strip()) 

268 if not match: 

269 raise CaptureError("invalid capture marker") 

270 status_text = match.group("status") 

271 status, reason = normalize_status(status_text) 

272 return CaptureMarker( 

273 pr_number=int(match.group("pr")), 

274 status=status, 

275 reason=reason, 

276 ) 

277 

278 

279def record_marker( 

280 *, 

281 pr_number: int | None, 

282 status: str | None, 

283 reason: str | None = None, 

284 artifact: str | None = None, 

285 title: str | None = None, 

286 labels: list[str] | tuple[str, ...] = (), 

287 changed_files: list[str] | tuple[str, ...] = (), 

288 existing_records: list[dict[str, Any]] | tuple[dict[str, Any], ...] = (), 

289 config: _HasPolicyPack | None = None, 

290 not_run: bool = False, 

291 retrieved: list[str] | tuple[str, ...] = (), 

292) -> dict[str, Any]: 

293 """Build the capture block stored in a ship run ledger record. 

294 

295 ``artifact`` is an optional reference (path or content hash) to the durable 

296 capture artifact. It is the proof that an ``applied`` capture actually 

297 produced something; capture reconcile treats ``applied`` with no artifact as 

298 a finding. ``deferred``/``skipped`` need no artifact. 

299 

300 ``retrieved`` is the fingerprints of the learnings this run put in front of the 

301 implementer and the reviewers (#1155). Recording them is what lets a later run 

302 tell a lesson nobody had from one that was surfaced and still not applied — 

303 the difference between a retrieval gap and a discipline gap. 

304 

305 ``not_run`` marks a record whose run never reached capture, so a reader can 

306 tell it apart from one that reached capture and lost its marker. Both carry 

307 ``marker: None``, and without the flag a capture-health reader can only 

308 assume the second and report a gap that is not there (#945). It is a 

309 *declaration* by the operator, deliberately: inferring it from the null would 

310 reclassify every genuinely missing marker as "never attempted", which is the 

311 fail-open this field exists to avoid. 

312 """ 

313 clean_artifact = artifact.strip() if isinstance(artifact, str) and artifact.strip() else None 

314 clean_retrieved = _strings(retrieved) 

315 if status is None: 

316 return { 

317 "schema_version": CAPTURE_SCHEMA_VERSION, 

318 "status": None, 

319 "reason": reason, 

320 "marker_reason": None, 

321 "marker": None, 

322 "not_run": not_run, 

323 "artifact": clean_artifact, 

324 "artifact_scope": artifact_scope(clean_artifact, config), 

325 "retrieved": clean_retrieved, 

326 "fail_soft": True, 

327 "learning": learning_decision( 

328 title=title, 

329 labels=labels, 

330 changed_files=changed_files, 

331 capture_status=None, 

332 capture_reason=reason, 

333 existing_records=existing_records, 

334 config=config, 

335 ), 

336 } 

337 learning = learning_decision( 

338 title=title, 

339 labels=labels, 

340 changed_files=changed_files, 

341 capture_status=status, 

342 capture_reason=reason, 

343 existing_records=existing_records, 

344 config=config, 

345 ) 

346 marker_reason = _marker_reason(status, reason) 

347 if pr_number is None: 

348 clean_status, clean_marker_reason = normalize_status(status, marker_reason) 

349 return { 

350 "schema_version": CAPTURE_SCHEMA_VERSION, 

351 "status": clean_status, 

352 "reason": reason, 

353 "marker_reason": clean_marker_reason, 

354 "marker": None, 

355 "artifact": clean_artifact, 

356 "artifact_scope": artifact_scope(clean_artifact, config), 

357 "retrieved": clean_retrieved, 

358 "fail_soft": True, 

359 "learning": learning, 

360 } 

361 marker = build_marker(pr_number=pr_number, status=status, reason=marker_reason) 

362 return { 

363 "schema_version": CAPTURE_SCHEMA_VERSION, 

364 "status": marker.status, 

365 "reason": reason, 

366 "marker_reason": marker.reason, 

367 "marker": marker.as_text(), 

368 "artifact": clean_artifact, 

369 "artifact_scope": artifact_scope(clean_artifact, config), 

370 "retrieved": clean_retrieved, 

371 "fail_soft": True, 

372 "learning": learning, 

373 } 

374 

375 

376def learning_decision( 

377 *, 

378 title: str | None = None, 

379 labels: list[str] | tuple[str, ...] = (), 

380 changed_files: list[str] | tuple[str, ...] = (), 

381 capture_status: str | None = None, 

382 capture_reason: str | None = None, 

383 existing_records: list[dict[str, Any]] | tuple[dict[str, Any], ...] = (), 

384 config: _HasPolicyPack | None = None, 

385) -> dict[str, Any]: 

386 """Classify whether a merged PR deserves a durable learning artifact. 

387 

388 The marker is mandatory and independent from this decision. Durable learning is 

389 optional, policy-driven, and deduped by a stable fingerprint so routine merges can stay 

390 marker-only without losing auditability. 

391 """ 

392 policy = _learning_policy(config) 

393 fingerprint = learning_fingerprint( 

394 title=title, 

395 labels=labels, 

396 changed_files=changed_files, 

397 ) 

398 if _learning_dedupe_enabled(policy): 

399 duplicate_of = _duplicate_learning_fingerprint(fingerprint, existing_records) 

400 if duplicate_of is not None: 

401 return _learning_result( 

402 "duplicate", 

403 reason="duplicate-learning", 

404 fingerprint=fingerprint, 

405 duplicate_of=duplicate_of, 

406 policy=policy, 

407 ) 

408 # **The same parent pair the writer consults.** Gating only the write left the 

409 # record claiming `create-learning` with `durable_artifact: true` for a run that 

410 # produced nothing — the write/record disagreement this feature already treats 

411 # as load-bearing one level down, inverted. One predicate answers both. 

412 if not policy.get("enabled") or not capture_hook_enabled(config): 

413 return _learning_result( 

414 "marker-only", 

415 reason="policy-unavailable", 

416 fingerprint=fingerprint, 

417 policy=policy, 

418 ) 

419 mode = policy.get("mode", "marker-only") 

420 if mode == "create-learning": 

421 # **Anything but `applied`**, not just `skipped:*`. `deferred` fell through 

422 # and answered `create-learning` with `durable_artifact: true`, while 

423 # `learning_sink_writes` refuses every status but `applied` — the same 

424 # write/record disagreement, one status over. 

425 if capture_status != "applied": 

426 return _learning_result( 

427 "marker-only", 

428 reason=( 

429 "capture-skipped" 

430 if (capture_status or "").startswith("skipped") 

431 else "capture-not-applied" 

432 ), 

433 fingerprint=fingerprint, 

434 policy=policy, 

435 ) 

436 return _learning_result( 

437 "create-learning", 

438 reason=_policy_reason(policy, "policy-requested-learning"), 

439 fingerprint=fingerprint, 

440 policy=policy, 

441 ) 

442 if mode == "defer": 

443 return _learning_result( 

444 "defer", 

445 reason=_policy_reason(policy, "policy-deferred"), 

446 fingerprint=fingerprint, 

447 policy=policy, 

448 ) 

449 return _learning_result( 

450 "marker-only", 

451 reason=_policy_reason(policy, "marker-only-policy"), 

452 fingerprint=fingerprint, 

453 policy=policy, 

454 ) 

455 

456 

457def capture_hook_enabled(config: _HasPolicyPack | None) -> bool: 

458 """Whether this project runs a post-merge **content hook** at all. 

459 

460 `policy_pack.capture.enabled` is the project saying it intends to; `mode: 

461 marker-only` records the core marker *without* one, which is the schema's own 

462 wording. The learning sink is that hook, and so is the decision that says a 

463 durable artifact is wanted — both ask here, because a writer and a record that 

464 answer this differently is the disagreement this feature keeps producing. 

465 `_reconcile_marker_decision` reads the same pair for `skipped:no-policy`. 

466 """ 

467 policy = _capture_policy(config) 

468 return bool(policy.get("enabled")) and policy.get("mode", "extension") == "extension" 

469 

470 

471def learning_fingerprint( 

472 *, 

473 title: str | None = None, 

474 labels: list[str] | tuple[str, ...] = (), 

475 changed_files: list[str] | tuple[str, ...] = (), 

476) -> str: 

477 """Return a stable, consumer-neutral dedupe fingerprint for learning candidates.""" 

478 import hashlib 

479 

480 payload = { 

481 "title": _normalize_text(title), 

482 "labels": sorted(_normalize_text(label) for label in _strings(labels)), 

483 "changed_files": sorted(_normalize_path(path) for path in _strings(changed_files)), 

484 } 

485 encoded = json.dumps(payload, sort_keys=True, separators=(",", ":")) 

486 return hashlib.sha256(encoded.encode("utf-8")).hexdigest() 

487 

488 

489def verify_session( 

490 records: list[dict[str, Any]], 

491 merged_prs: list[int] | tuple[int, ...], 

492) -> dict[str, Any]: 

493 """Verify that each merged PR has an applied/deferred/allowed-skip capture marker.""" 

494 results = [_verify_pr(records, pr) for pr in merged_prs] 

495 missing = [item for item in results if item["status"] == "missing"] 

496 invalid = [item for item in results if item["status"] == "invalid"] 

497 status = "complete" if not missing and not invalid else "incomplete" 

498 return { 

499 "schema_version": CAPTURE_SCHEMA_VERSION, 

500 "status": status, 

501 "expected_prs": list(merged_prs), 

502 "results": results, 

503 "summary": { 

504 "ok": sum(1 for item in results if item["ok"]), 

505 "missing": len(missing), 

506 "invalid": len(invalid), 

507 }, 

508 } 

509 

510 

511def reconcile_session( 

512 records: list[dict[str, Any]], 

513 merged_prs: list[int | dict[str, Any]] | tuple[int | dict[str, Any], ...], 

514 *, 

515 config: _HasPolicyPack | None = None, 

516 capture_capability_available: bool = False, 

517) -> dict[str, Any]: 

518 """Plan idempotent post-merge reconciliation actions for capture gaps. 

519 

520 The returned plan is pure data. It never writes ledger records, comments, issues, git 

521 state, or PR state; adapters may apply the listed actions after their own transport and 

522 consent checks. This keeps reconcile recovery deterministic and safe to run repeatedly. 

523 """ 

524 items = [_merged_pr_info(item) for item in merged_prs] 

525 results = [ 

526 _reconcile_pr( 

527 records, 

528 item, 

529 config=config, 

530 capture_capability_available=capture_capability_available, 

531 ) 

532 for item in items 

533 ] 

534 actionable = [item for item in results if item["actions"]] 

535 blocked = [item for item in results if item["status"] in {"invalid", "ambiguous"}] 

536 complete = [item for item in results if item["status"] == "complete"] 

537 status = "blocked" if blocked else "actionable" if actionable else "complete" 

538 return { 

539 "schema_version": RECONCILE_SCHEMA_VERSION, 

540 "status": status, 

541 "dry_run_safe": True, 

542 "idempotent": True, 

543 "no_code_mutations": True, 

544 "expected_prs": [item["number"] for item in items], 

545 "results": results, 

546 "summary": { 

547 "complete": len(complete), 

548 "actionable": len(actionable), 

549 "blocked": len(blocked), 

550 }, 

551 } 

552 

553 

554def recursion_guard( 

555 *, 

556 title: str | None = None, 

557 labels: list[str] | tuple[str, ...] = (), 

558 changed_files: list[str] | tuple[str, ...] = (), 

559) -> bool: 

560 """Return true when capture should skip to avoid capture-on-capture recursion.""" 

561 # ⚡ Bolt: Return early if title matches to avoid expensive loops on labels and files 

562 if title and "capture" in title.lower(): 

563 return True 

564 

565 # ⚡ Bolt: Return early if label matches to avoid expensive loops on files 

566 for label in labels: 

567 if label.lower() == "capture": 

568 return True 

569 

570 # ⚡ Bolt: Avoid generator overhead with explicit loop for paths (~90x speedup when early match) 

571 for path in changed_files: 

572 p = path.lower() 

573 if "/capture" in p or p.endswith("capture.py"): 

574 return True 

575 

576 return False 

577 

578 

579def _merged_pr_info(item: int | dict[str, Any]) -> dict[str, Any]: 

580 if isinstance(item, int): 

581 return { 

582 "number": item, 

583 "title": None, 

584 "labels": [], 

585 "changed_files": [], 

586 "issue_numbers": [], 

587 } 

588 number = item.get("number") 

589 if not isinstance(number, int) or number <= 0: 

590 raise CaptureError("merged PR entry requires a positive number") 

591 return { 

592 "number": number, 

593 "title": item.get("title") if isinstance(item.get("title"), str) else None, 

594 "labels": _strings(item.get("labels")), 

595 "changed_files": _strings(item.get("changed_files")), 

596 "issue_numbers": _positive_ints(item.get("issue_numbers")), 

597 } 

598 

599 

600def _reconcile_pr( 

601 records: list[dict[str, Any]], 

602 item: dict[str, Any], 

603 *, 

604 config: _HasPolicyPack | None, 

605 capture_capability_available: bool, 

606) -> dict[str, Any]: 

607 pr_number = item["number"] 

608 verification = _verify_pr(records, pr_number) 

609 issue_numbers = _linked_issue_numbers(records, item) 

610 if len(issue_numbers) > 1: 

611 return _reconcile_result( 

612 pr_number, 

613 status="ambiguous", 

614 reason="multiple linked issues found for merged PR", 

615 verification=verification, 

616 issue_numbers=issue_numbers, 

617 blocked=True, 

618 ) 

619 if verification["ok"]: 

620 if len(issue_numbers) == 1: 

621 return _reconcile_result( 

622 pr_number, 

623 status="actionable", 

624 reason="capture marker already present; linked issue closeout can be reconciled", 

625 verification=verification, 

626 issue_numbers=issue_numbers, 

627 marker=verification["marker"], 

628 actions=[ 

629 _action( 

630 "close-linked-issue", pr_number=pr_number, issue_number=issue_numbers[0] 

631 ), 

632 ], 

633 ) 

634 return _reconcile_result( 

635 pr_number, 

636 status="complete", 

637 reason="capture marker already present", 

638 verification=verification, 

639 ) 

640 if verification["status"] == "invalid": 

641 return _reconcile_result( 

642 pr_number, 

643 status="invalid", 

644 reason=verification["reason"], 

645 verification=verification, 

646 issue_numbers=issue_numbers, 

647 blocked=True, 

648 ) 

649 marker_status, marker_reason, reason = _reconcile_marker_decision( 

650 item, 

651 config=config, 

652 capture_capability_available=capture_capability_available, 

653 ) 

654 marker = marker_text( 

655 pr_number=pr_number, 

656 status=marker_status, 

657 reason=marker_reason, 

658 ) 

659 actions = [ 

660 _action( 

661 "emit-capture-marker", 

662 pr_number=pr_number, 

663 marker=marker, 

664 status=marker_status, 

665 reason=marker_reason, 

666 ), 

667 _action("post-closure-summary", pr_number=pr_number), 

668 ] 

669 if marker_status == "deferred": 

670 actions.insert(0, _action("run-capture-extension", pr_number=pr_number)) 

671 if marker_status == "skipped": 

672 actions.append(_action("record-skip", pr_number=pr_number, reason=marker_reason)) 

673 if len(issue_numbers) == 1: 

674 actions.append( 

675 _action("close-linked-issue", pr_number=pr_number, issue_number=issue_numbers[0]) 

676 ) 

677 return _reconcile_result( 

678 pr_number, 

679 status="actionable", 

680 reason=reason, 

681 verification=verification, 

682 issue_numbers=issue_numbers, 

683 marker=marker, 

684 actions=actions, 

685 ) 

686 

687 

688def _marker_of(record: dict[str, Any]) -> str | None: 

689 """This record's capture marker, or ``None`` when it carries none.""" 

690 capture_block = record.get("capture") 

691 marker = capture_block.get("marker") if isinstance(capture_block, dict) else None 

692 return marker if marker else None 

693 

694 

695def _record_head(record: dict[str, Any]) -> str | None: 

696 """The head a ship_run record was written for, or ``None`` when it names none.""" 

697 git = record.get("git") 

698 head = git.get("head_sha") if isinstance(git, dict) else None 

699 return head.strip() if isinstance(head, str) and head.strip() else None 

700 

701 

702def _verify_pr(records: list[dict[str, Any]], pr_number: int) -> dict[str, Any]: 

703 candidates = [ 

704 record 

705 for record in records 

706 if record.get("record_type") == "ship_run" 

707 and (record.get("pull_request") or {}).get("number") == pr_number 

708 ] 

709 # Markers are counted on one head, not across every head this pull request 

710 # ever had (#1157). `existing_capture_marker` refuses a second marker per 

711 # (pull request, head); counting per pull request here would move the same 

712 # deadlock one step later — a pull request whose superseded head left a 

713 # marker would fail verification for carrying two, and the only exit would 

714 # again be editing an append-only ledger by hand. 

715 # 

716 # The head is taken from the **last record that carries a marker**, not from 

717 # the last record. `ledger.latest_ship_run_for_pr` documents why the latter 

718 # is the wrong proxy, and the concrete case is #945's: a run that never 

719 # reached capture re-records gates on a new head and writes `not_run` with no 

720 # marker. Reading the head off that row would look past a real applied marker 

721 # on the merged head and report the capture missing, and `reconcile_session` 

722 # would then plan to emit a marker that already exists. A row with no marker 

723 # cannot move the head the markers are counted on, because it is not evidence 

724 # about capture at all. 

725 marked = [record for record in candidates if _marker_of(record)] 

726 merged_head = _record_head(marked[-1]) if marked else None 

727 markers = [ 

728 marker 

729 for record in marked 

730 if _record_head(record) == merged_head and (marker := _marker_of(record)) 

731 ] 

732 if len(markers) > 1: 

733 return { 

734 "pr": pr_number, 

735 "ok": False, 

736 "status": "invalid", 

737 "reason": "multiple capture markers found for merged PR", 

738 "marker": markers[-1], 

739 "marker_count": len(markers), 

740 } 

741 for marker in markers: 

742 try: 

743 parsed = parse_marker(marker) 

744 except CaptureError as exc: 

745 return { 

746 "pr": pr_number, 

747 "ok": False, 

748 "status": "invalid", 

749 "reason": str(exc), 

750 "marker": marker, 

751 } 

752 if parsed.pr_number != pr_number: 

753 return { 

754 "pr": pr_number, 

755 "ok": False, 

756 "status": "invalid", 

757 "reason": "marker PR does not match ledger PR", 

758 "marker": marker, 

759 } 

760 return { 

761 "pr": pr_number, 

762 "ok": True, 

763 "status": parsed.status, 

764 "reason": parsed.reason, 

765 "marker": marker, 

766 } 

767 return { 

768 "pr": pr_number, 

769 "ok": False, 

770 "status": "missing", 

771 "reason": "no capture marker found for merged PR", 

772 "marker": None, 

773 } 

774 

775 

776def _reconcile_result( 

777 pr_number: int, 

778 *, 

779 status: str, 

780 reason: str, 

781 verification: dict[str, Any], 

782 issue_numbers: list[int] | None = None, 

783 marker: str | None = None, 

784 actions: list[dict[str, Any]] | None = None, 

785 blocked: bool = False, 

786) -> dict[str, Any]: 

787 return { 

788 "pr": pr_number, 

789 "status": status, 

790 "reason": reason, 

791 "verification_status": verification["status"], 

792 "blocked": blocked, 

793 "issue_numbers": list(issue_numbers or ()), 

794 "marker": marker, 

795 "actions": list(actions or ()), 

796 } 

797 

798 

799def _reconcile_marker_decision( 

800 item: dict[str, Any], 

801 *, 

802 config: _HasPolicyPack | None, 

803 capture_capability_available: bool, 

804) -> tuple[str, str | None, str]: 

805 if recursion_guard( 

806 title=item["title"], 

807 labels=item["labels"], 

808 changed_files=item["changed_files"], 

809 ): 

810 return "skipped", "recursion-guard", "capture recursion guard matched" 

811 policy = _capture_policy(config) 

812 if policy.get("enabled") and policy.get("mode", "extension") == "marker-only": 

813 return "applied", None, "marker-only capture policy configured" 

814 if policy.get("enabled") and policy.get("mode", "extension") == "extension": 

815 if capture_capability_available: 

816 return "deferred", None, "capture extension can be rerun" 

817 return "skipped", "capability-unavailable", "capture extension capability unavailable" 

818 return "skipped", "no-policy", "no capture policy configured" 

819 

820 

821def _linked_issue_numbers(records: list[dict[str, Any]], item: dict[str, Any]) -> list[int]: 

822 numbers = set(item["issue_numbers"]) 

823 pr_number = item["number"] 

824 for record in records: 

825 if record.get("record_type") != "ship_run": 

826 continue 

827 if (record.get("pull_request") or {}).get("number") != pr_number: 

828 continue 

829 issue_number = (record.get("issue") or {}).get("number") 

830 if isinstance(issue_number, int) and issue_number > 0: 

831 numbers.add(issue_number) 

832 return sorted(numbers) 

833 

834 

835def _action( 

836 action_type: str, 

837 *, 

838 pr_number: int, 

839 marker: str | None = None, 

840 status: str | None = None, 

841 reason: str | None = None, 

842 issue_number: int | None = None, 

843) -> dict[str, Any]: 

844 action = { 

845 "type": action_type, 

846 "pr": pr_number, 

847 "idempotency_key": f"{action_type}:pr-{pr_number}", 

848 } 

849 if marker is not None: 

850 action["marker"] = marker 

851 if status is not None: 

852 action["status"] = status 

853 if reason is not None: 

854 action["reason"] = reason 

855 if issue_number is not None: 

856 action["issue"] = issue_number 

857 action["idempotency_key"] = f"{action_type}:issue-{issue_number}:pr-{pr_number}" 

858 return action 

859 

860 

861def _capture_policy(config: _HasPolicyPack | None) -> dict[str, Any]: 

862 if config is None or not isinstance(config.policy_pack, dict): 

863 return {} 

864 policy = config.policy_pack.get("capture") 

865 return policy if isinstance(policy, dict) else {} 

866 

867 

868def _learning_policy(config: _HasPolicyPack | None) -> dict[str, Any]: 

869 policy = _capture_policy(config) 

870 learning = policy.get("learning") if isinstance(policy, dict) else None 

871 return learning if isinstance(learning, dict) else {} 

872 

873 

874def _learning_dedupe_enabled(policy: dict[str, Any]) -> bool: 

875 dedupe = policy.get("dedupe") 

876 if not isinstance(dedupe, dict): 

877 return True 

878 return bool(dedupe.get("enabled", True)) 

879 

880 

881def _marker_reason(status: str, reason: str | None) -> str | None: 

882 raw = status.strip() 

883 if raw.startswith("skipped:"): 

884 return None 

885 if raw != "skipped": 

886 return None 

887 if reason in SKIP_REASONS: 

888 return reason 

889 return "no-policy" 

890 

891 

892def _strings(value: Any) -> list[str]: 

893 if not isinstance(value, list | tuple): 

894 return [] 

895 return [item for item in value if isinstance(item, str)] 

896 

897 

898def _positive_ints(value: Any) -> list[int]: 

899 if not isinstance(value, list | tuple): 

900 return [] 

901 return [item for item in value if isinstance(item, int) and item > 0] 

902 

903 

904def _duplicate_learning_fingerprint( 

905 fingerprint: str, 

906 records: list[dict[str, Any]] | tuple[dict[str, Any], ...], 

907) -> str | None: 

908 for record in records: 

909 if not isinstance(record, dict): 

910 continue 

911 capture_block = record.get("capture") 

912 learning = capture_block.get("learning") if isinstance(capture_block, dict) else None 

913 if not isinstance(learning, dict): 

914 continue 

915 if learning.get("fingerprint") != fingerprint: 

916 continue 

917 decision = learning.get("decision") 

918 if decision in {"create-learning", "duplicate"}: 

919 return str(record.get("run_id") or (record.get("pull_request") or {}).get("number")) 

920 return None 

921 

922 

923def _learning_result( 

924 decision: str, 

925 *, 

926 reason: str, 

927 fingerprint: str, 

928 policy: dict[str, Any], 

929 duplicate_of: str | None = None, 

930) -> dict[str, Any]: 

931 if decision not in LEARNING_DECISIONS: 

932 raise CaptureError(f"unsupported learning decision: {decision}") 

933 result = { 

934 "schema_version": LEARNING_DECISION_SCHEMA_VERSION, 

935 "decision": decision, 

936 "reason": reason, 

937 "fingerprint": fingerprint, 

938 "policy_source": "policy_pack.capture.learning", 

939 "policy_mode": policy.get("mode", "policy-unavailable"), 

940 "durable_artifact": decision == "create-learning", 

941 } 

942 if duplicate_of is not None: 

943 result["duplicate_of"] = duplicate_of 

944 return result 

945 

946 

947def _policy_reason(policy: dict[str, Any], default: str) -> str: 

948 reason = policy.get("reason") 

949 return reason.strip() if isinstance(reason, str) and reason.strip() else default 

950 

951 

952def _normalize_text(value: str | None) -> str: 

953 return " ".join(value.lower().split()) if isinstance(value, str) else "" 

954 

955 

956def _normalize_path(value: str) -> str: 

957 return "/".join(value.strip().lower().replace("\\", "/").split("/")) 

958 

959 

960#: The one sink kind this issue ships. A directory of Markdown with stable 

961#: frontmatter is the whole contract — no vault format, no wikilinks, no plugin 

962#: API — so a project can point it at whatever reads Markdown and keel never 

963#: learns what that is. 

964LEARNING_SINK_KINDS = ("markdown-dir",) 

965 

966#: Where a sink writes when a project configures capture but names no path. The 

967#: existing `.keel/learning/` convention, so turning the sink on changes where 

968#: files appear only for a project that asked it to. 

969DEFAULT_LEARNING_SINK_PATH = ".keel/learning" 

970 

971#: **The fingerprint is in the name because the fingerprint is the identity.** Date, 

972#: PR and slug do not distinguish two lessons: a second `create-learning` run on the 

973#: same PR the same day — different labels, different files, a different lesson — 

974#: resolved to the same path and `os.replace` destroyed the first, leaving the 

975#: earlier ledger record pointing at a document that says something else. The dedupe 

976#: cannot help; it suppresses *identical* fingerprints, and these differ. 

977DEFAULT_LEARNING_SINK_FILENAME = "{date}-pr{pr}-{slug}-{fingerprint}.md" 

978 

979#: The suffixes the reader opens. Named because the *writer* is validated against 

980#: them: a sink filename ending `.markdown`, or in nothing at all, was written 

981#: successfully into the sink and then skipped by the only thing that reads it — 

982#: the writer/reader disagreement this feature exists inside, arriving through a 

983#: template the validator accepted. 

984LEARNING_READ_SUFFIXES = (".md", ".json", ".txt") 

985 

986#: How much of the fingerprint a filename carries. A sha256 prefix this long 

987#: distinguishes every learning a project will ever write without making the name 

988#: unreadable. 

989LEARNING_FINGERPRINT_SLICE = 12 

990 

991#: The document's fourth section (#1166): one entry per ``changed_files`` path, written 

992#: as a Markdown link so a **link-following** reader — a knowledge-graph builder, a 

993#: wiki — gets the file ↔ lesson edge. The front-matter list is a string to such a 

994#: reader; keel's own reader keeps matching on the list, which is why it stays. 

995LEARNING_FILES_HEADING = "## Files" 

996LEARNING_NO_FILES = "_No files recorded._" 

997 

998#: Characters that end, escape, or open something inside a CommonMark link *text*: 

999#: the bracket pair and the backslash; the angle brackets that would start an autolink 

1000#: or raw HTML from a file name (a merged PR chooses those bytes); the emphasis and 

1001#: strikethrough delimiters that turned ``__init__.py`` — the most common Python file 

1002#: name — into bold ``init``; the ampersand that would decode an entity reference; and 

1003#: the backtick, because a code span binds more tightly than the link's brackets and 

1004#: swallows the characters between a pair. Every one is ASCII punctuation, which 

1005#: CommonMark lets a backslash escape. The destination is percent-encoded instead, so 

1006#: the two halves of a link never disagree about where a path ends. 

1007_LINK_TEXT_UNSAFE = re.compile(r"([\\\[\]<>_*~&`])") 

1008 

1009#: The frontmatter contract the reader depends on. Fixed and small on purpose: 

1010#: `retrieve_relevant_learnings` reads `title` and `description` out of it, so a 

1011#: field added here is a field that side has to be taught. 

1012LEARNING_SCHEMA_VERSION = "keel.learning.v1" 

1013 

1014#: Every placeholder a `path` or `filename` template may use. Named rather than 

1015#: open-ended: an unknown placeholder is a typo that would otherwise write a 

1016#: directory called `{repoo}` and look like it worked. 

1017LEARNING_SINK_PLACEHOLDERS = ( 

1018 "owner", 

1019 "repo", 

1020 "base_branch", 

1021 "date", 

1022 "pr", 

1023 "slug", 

1024 "fingerprint", 

1025) 

1026 

1027_SLUG_STRIP = re.compile(r"[^a-z0-9]+") 

1028 

1029 

1030def _slugify(text: str | None, *, limit: int = 48) -> str: 

1031 """A filename-safe slug, or `learning` when the title reduces to nothing.""" 

1032 slug = _SLUG_STRIP.sub("-", (text or "").lower()).strip("-") 

1033 if not slug: 

1034 return "learning" 

1035 return slug[:limit].rstrip("-") 

1036 

1037 

1038def learning_sink_policy(config: _HasPolicyPack | None) -> dict[str, Any] | None: 

1039 """The `policy_pack.capture.learning.sink` block, or `None` when unset. 

1040 

1041 `None` and `{}` are different answers and the difference is load-bearing: every 

1042 field of a sink is optional, so `sink: {}` is a project taking the documented 

1043 defaults — `markdown-dir` into `.keel/learning`. Collapsed to `{}`, that 

1044 declaration read as *no sink at all* and the project got the pre-#1154 

1045 behaviour of writing nothing, while the same block naming its `kind` wrote. 

1046 """ 

1047 sink = _learning_policy(config).get("sink") 

1048 return sink if isinstance(sink, dict) else None 

1049 

1050 

1051#: Where a recorded `capture.artifact` can be read from. 

1052ARTIFACT_SCOPE_REPOSITORY = "repository" 

1053ARTIFACT_SCOPE_MACHINE = "machine" 

1054 

1055 

1056def artifact_scope(artifact: str | None, config: _HasPolicyPack | None = None) -> str | None: 

1057 """Where this project's sink writes: `repository`, `machine`, or ``None`` for neither. 

1058 

1059 An in-repo sink's path is recorded relative to ``--root``, so it means the same thing 

1060 in every clone. A sink outside the checkout — a shared `~/knowledge` folder — is 

1061 recorded absolute, which names nothing on any other host. The run ledger is gitignored 

1062 per-host state by default (`.keel/state/run-ledger.jsonl`), but 

1063 `policy_pack.reports.run_ledger` may point it at a tracked file, and then that path 

1064 travels to teammates and CI runners with it. Saying so in the record is what lets 

1065 `keel capture-verify` tell "written somewhere this host cannot see" from "never 

1066 written", which are the same absence and very different facts (#1185). 

1067 

1068 It describes the **sink**, not the artifact, so it is recorded even when this run 

1069 wrote nothing: a cross-host duplicate drops the unreadable path it would have reused, 

1070 and without the scope beside it that record is indistinguishable from a capture that 

1071 produced no file at all. 

1072 

1073 ``None`` when the project has no sink, which is not the same as a sink somewhere else 

1074 — `learning_sink_in_worktree` answers False for "no sink" and for "capture disabled" 

1075 as well as for an outside one, so the sink has to be looked for separately. With no 

1076 config, the path's own shape decides: an anchored path was always an outside sink. 

1077 """ 

1078 if config is not None: 

1079 sink = learning_sink_policy(config) 

1080 if sink is None: 

1081 return None 

1082 return ( 

1083 ARTIFACT_SCOPE_REPOSITORY if _sink_writes_in_worktree(sink) else ARTIFACT_SCOPE_MACHINE 

1084 ) 

1085 if not isinstance(artifact, str) or not artifact.strip(): 

1086 return None 

1087 path = artifact.strip() 

1088 anchored = path.startswith("~") or workspace.is_root_anchored(path) 

1089 return ARTIFACT_SCOPE_MACHINE if anchored else ARTIFACT_SCOPE_REPOSITORY 

1090 

1091 

1092def learning_sink_in_worktree(config: _HasPolicyPack | None) -> bool: 

1093 """Does this project's sink write **inside the repository**? 

1094 

1095 Pure, and answered from the path's shape rather than the filesystem: a relative 

1096 path resolves against `--root`, which is the checkout, while an absolute or `~` 

1097 path is a folder somewhere else. It decides who has to commit the file — keel 

1098 writes it and does not, so one inside the working tree is lost to the next 

1099 worktree and to every CI runner unless the run commits it. 

1100 """ 

1101 # Gated on the hook too: a dormant `sink:` under a disabled capture writes 

1102 # nothing, so nothing needs committing. Fourth reader of the same question. 

1103 sink = learning_sink_policy(config) if capture_hook_enabled(config) else None 

1104 return sink is not None and _sink_writes_in_worktree(sink) 

1105 

1106 

1107def _sink_writes_in_worktree(sink: dict[str, Any]) -> bool: 

1108 """Does this sink block's **path** name somewhere inside the checkout? 

1109 

1110 Split out from :func:`learning_sink_in_worktree` because the two callers ask 

1111 different questions of the same block. *Who commits the file* is gated on the 

1112 capture hook — a dormant sink writes nothing, so nothing needs committing — but 

1113 *what shape the path has* is not. Answering the second through the first made a 

1114 dormant in-repo sink record `artifact_scope: machine`, which reads as "written on 

1115 another host" for a repo-relative path every clone can see. 

1116 """ 

1117 path = str(sink.get("path") or DEFAULT_LEARNING_SINK_PATH) 

1118 if path.startswith("~"): 

1119 return False 

1120 # **Anchored on *any* platform, not this one.** `Path("C:/knowledge").is_absolute()` 

1121 # is False on POSIX and `Path("/srv/knowledge").is_absolute()` is False on 

1122 # Windows, so each host called the other's absolute path in-repo and would have 

1123 # told the adapter to `git add` it — committing a `C:` directory into the 

1124 # repository, or reaching outside it. A keel config is the same text wherever it 

1125 # is read; `workspace.is_root_anchored` is the question already asked that way. 

1126 if workspace.is_root_anchored(path): 

1127 return False 

1128 # **Normalised, because `../learnings` is relative and still outside.** It is 

1129 # the documented "folder next to the checkout" shape without the leading `~`, 

1130 # and reported as in-repo it would send the adapter to `git add` a path git 

1131 # refuses — leaving the file off `base_branch` and the next worktree empty, 

1132 # which is the failure this flag exists to prevent, arriving through the flag. 

1133 return Path(os.path.normpath(path)).parts[:1] != ("..",) 

1134 

1135 

1136def learning_sink_errors(sink: Any) -> list[str]: 

1137 """Why this `sink` block cannot be used, or `[]`. 

1138 

1139 Validated where the config is read rather than where the file is written: a 

1140 template naming `{repoo}` is a typo whose only symptom would otherwise be a 

1141 directory by that name, created successfully, on a machine nobody is watching. 

1142 """ 

1143 if sink in (None, {}): 

1144 return [] 

1145 if not isinstance(sink, dict): 

1146 return ["policy_pack.capture.learning.sink must be a mapping"] 

1147 errors: list[str] = [] 

1148 kind = sink.get("kind", LEARNING_SINK_KINDS[0]) 

1149 if kind not in LEARNING_SINK_KINDS: 

1150 errors.append( 

1151 f"policy_pack.capture.learning.sink.kind must be one of " 

1152 f"{', '.join(LEARNING_SINK_KINDS)} (got {kind!r})" 

1153 ) 

1154 for field_name in ("path", "filename"): 

1155 raw = sink.get(field_name) 

1156 if raw is None: 

1157 continue 

1158 if not isinstance(raw, str) or not raw.strip(): 

1159 errors.append( 

1160 f"policy_pack.capture.learning.sink.{field_name} must be a non-empty string" 

1161 ) 

1162 continue 

1163 # `[^}]*`, not `[a-z_]*`: a mixed-case or hyphenated typo — `{Repo}`, 

1164 # `{base-branch}` — is exactly as wrong as `{repoo}` and was invisible to a 

1165 # pattern that only matched the shape of a correct name. 

1166 used = re.findall(r"\{([^}]*)\}", raw) 

1167 for name in used: 

1168 if name not in LEARNING_SINK_PLACEHOLDERS: 

1169 errors.append( 

1170 f"policy_pack.capture.learning.sink.{field_name} uses unknown placeholder " 

1171 f"{{{name}}}; known: {', '.join(LEARNING_SINK_PLACEHOLDERS)}" 

1172 ) 

1173 # **A filename must be able to name two lessons.** Date, PR and slug do not 

1174 # distinguish them: a second `create-learning` run on the same PR the same 

1175 # day is a *different* lesson with a different fingerprint, and without it in 

1176 # the name the write destroys the first one — leaving its ledger record 

1177 # pointing at a document that says something else. Refused here, where the 

1178 # placeholder typos are refused, because the only other symptom is a file 

1179 # that quietly stops existing. 

1180 if field_name == "filename" and "fingerprint" not in used: 

1181 errors.append( 

1182 "policy_pack.capture.learning.sink.filename must contain {fingerprint}; " 

1183 "without it two lessons on one pull request overwrite each other" 

1184 ) 

1185 # **A filename is a name, not a path.** `{pr}/{fingerprint}.md` passes every 

1186 # other check, `mkdir(parents=True)` creates the directory happily, and 

1187 # `retrieve_relevant_learnings` — the only reader — globs one level and 

1188 # skips directories, so the lesson is written where nothing will ever read 

1189 # it. Nesting belongs in `path`, which is the field that names a directory. 

1190 if field_name == "filename" and not raw.endswith(LEARNING_READ_SUFFIXES): 

1191 errors.append( 

1192 f"policy_pack.capture.learning.sink.filename must end in one of " 

1193 f"{', '.join(LEARNING_READ_SUFFIXES)}; the read path opens no other " 

1194 f"suffix, so anything else is written and never found" 

1195 ) 

1196 if field_name == "filename" and ("/" in raw or "\\" in raw): 

1197 errors.append( 

1198 "policy_pack.capture.learning.sink.filename must not contain a path " 

1199 "separator; it names a file inside `path`, and the read path does not " 

1200 "descend into subdirectories" 

1201 ) 

1202 return errors 

1203 

1204 

1205#: Anything that would make a filename more than one path component, or unwritable. 

1206#: `/` and `\\` split it; the control characters are the same class one field over. 

1207_FILENAME_UNSAFE = re.compile(r"[/\\\x00-\x1f\x7f]+") 

1208 

1209 

1210def _relative_stays_relative(template: str, values: dict[str, str]) -> str: 

1211 """Expand a directory template without letting it change what it *is*. 

1212 

1213 An absolute or `~` template stays what the project wrote. A relative one has 

1214 to come back relative: a leading placeholder that expands to nothing otherwise 

1215 turns `{repo}/learnings` into `/learnings`, which the shape-reading predicates 

1216 still report as inside the repository — so the adapter is told to `git add` a 

1217 path at the filesystem root. 

1218 """ 

1219 expanded = _expand(template, values) 

1220 if template.startswith("~") or workspace.is_root_anchored(template): 

1221 return expanded 

1222 return expanded.lstrip("/\\") or DEFAULT_LEARNING_SINK_PATH 

1223 

1224 

1225def _one_component(name: str) -> str: 

1226 """A filename that names exactly one file, whatever the placeholders held. 

1227 

1228 Refusing a separator in the *template* is not enough: `{base_branch}` is a legal 

1229 filename placeholder and `feat/sink` is a normal branch, so 

1230 `{date}-pr{pr}-{base_branch}-{fingerprint}.md` expands to a name with a slash in 

1231 it, `mkdir(parents=True)` makes the directory, and the lesson lands one level 

1232 below where `retrieve_relevant_learnings` looks. Only `{slug}` was slugified; 

1233 every other value went in raw. 

1234 """ 

1235 return _FILENAME_UNSAFE.sub("-", name) 

1236 

1237 

1238def _expand(template: str, values: dict[str, str]) -> str: 

1239 out = template 

1240 for name, value in values.items(): 

1241 out = out.replace("{" + name + "}", value) 

1242 return out 

1243 

1244 

1245#: A value plain YAML reads back unchanged: starts with a letter, contains only 

1246#: letters, digits, space and a few punctuation marks that carry no meaning there. 

1247_YAML_PLAIN = re.compile(r"[A-Za-z][A-Za-z0-9 ._/()+-]*") 

1248 

1249#: **Everything `str.splitlines()` treats as a line break**, plus the rest of C0 and 

1250#: DEL. Not a taste question and not only the obvious ones: a raw CR splits a line 

1251#: inside quotes, a NUL makes a real parser refuse the document, and `\x85` (NEL), 

1252#: `\u2028` (LINE SEPARATOR) and `\u2029` (PARAGRAPH SEPARATOR) are line breaks to 

1253#: Python while looking like nothing at all — a C0-only pattern let a title open a 

1254#: Markdown section of its own through the very guard written to stop it. 

1255_YAML_CONTROL = re.compile("[\x00-\x1f\x7f\x85\u2028\u2029]") 

1256 

1257#: Words plain YAML turns into something that is not a string. 

1258_YAML_KEYWORDS = frozenset( 

1259 {"y", "n", "yes", "no", "true", "false", "on", "off", "null", "none", "~"} 

1260) 

1261 

1262 

1263def _one_line(value: str) -> str: 

1264 """A value that cannot start a second line, wherever it is written. 

1265 

1266 :func:`_yaml_scalar` applies this before quoting, and the **body** needs it too: 

1267 the document's `# {title}` heading took the raw string, so a title carrying a 

1268 newline wrote a heading and then whatever followed it as Markdown of its own — 

1269 `foo\n## injected` became `# foo` and an `## injected` section. The front matter 

1270 and the body have to say the same thing about the same field. 

1271 """ 

1272 return _YAML_CONTROL.sub(" ", value) 

1273 

1274 

1275def _yaml_scalar(value: str) -> str: 

1276 """A front-matter value that survives a real YAML parser. 

1277 

1278 **Quote unless the value is plainly safe**, rather than quoting a list of 

1279 dangerous characters. The first cut listed `:`, `#`, `"` and a newline, and a 

1280 dozen other shapes went through it: a leading `-`, `*`, `&`, `!`, `@`, `%`, 

1281 `|`, `>` or backtick either fails to parse or comes back as something else, 

1282 `{a}` becomes a mapping, `[a]` a list, and `yes` becomes `True`. An issue title 

1283 can be any of those. A deny-list has to be right about every character; an 

1284 allow-list only has to be right about the ones it lets through. 

1285 """ 

1286 # `value == value.strip()` is not tidiness: `yes ` matches the allow-list, is not 

1287 # in the keyword set, and goes in bare — and a plain YAML scalar drops its 

1288 # trailing space, so a parser reads `yes` and returns `True`. The same 

1289 # non-string a keyword produces, through the one gap the keyword check had. 

1290 if ( 

1291 value 

1292 and value == value.strip() 

1293 and _YAML_PLAIN.fullmatch(value) 

1294 and value.lower() not in _YAML_KEYWORDS 

1295 ): 

1296 return value 

1297 escaped = value.replace("\\", "\\\\").replace('"', '\\"') 

1298 # **Every control character, not the newline.** A raw carriage return survives 

1299 # inside quotes, and `_front_matter` splits lines on it — the reader gets a 

1300 # title of `"foo` and drops the rest, while a real parser reads `foo bar`, so 

1301 # the two disagree about the same file. A NUL is worse: PyYAML refuses the 

1302 # document outright. They become spaces, because a title is a line. 

1303 return f'"{_one_line(escaped)}"' 

1304 

1305 

1306def _yaml_sequence(key: str, values: list[str] | tuple[str, ...]) -> list[str]: 

1307 """A front-matter list, written so an empty one reads back as an empty list. 

1308 

1309 `key:` with nothing under it is a **null** to a YAML parser, not `[]`, and the 

1310 contract calls these fields sequences — a consumer that iterates them raises 

1311 `TypeError` on a merge that touched nothing it recorded. `key: []` is the same 

1312 field with the type it promised. 

1313 """ 

1314 items = _strings(values) 

1315 if not items: 

1316 return [f"{key}: []"] 

1317 return [f"{key}:", *(f" - {_yaml_scalar(item)}" for item in items)] 

1318 

1319 

1320def learning_file_link_base( 

1321 *, 

1322 directory: str, 

1323 in_repo: bool, 

1324 owner: str | None, 

1325 repo: str | None, 

1326 head_sha: str | None, 

1327) -> str | None: 

1328 """Where the **Files** section's links point, or ``None`` for bare paths (#1166). 

1329 

1330 Two forms, chosen by where the sink is. A sink **inside the checkout** links 

1331 relative to the document's own directory — ``../..`` from the default 

1332 ``.keel/learning`` — so the link resolves on disk from where the file sits, and a 

1333 graph builder walking the repository makes the edge to a node it already has. 

1334 The prefix is **lexical** — one ``..`` per component of the normalised directory — 

1335 because a plan touches no filesystem: :func:`posixpath.relpath` would consult 

1336 ``os.getcwd()`` for two relative arguments, which made the prefix depend on where 

1337 the process stood and raise from a deleted directory. It is POSIX on every platform 

1338 because the document is read on machines other than the one that wrote it. 

1339 

1340 A sink **outside** the checkout has nothing to link to relatively, and a relative 

1341 link that resolves to nothing is worse than none. It links the file on GitHub at 

1342 the merged head when the owner, the repository and the head are all known, and 

1343 otherwise says nothing — the caller renders the bare path. An *expanded* template 

1344 that climbs out of the checkout (``{owner}/../learnings`` with ``owner`` unset) is 

1345 outside it whatever the unexpanded template looked like, and takes the same forms. 

1346 """ 

1347 if in_repo: 

1348 normalised = posixpath.normpath(directory.replace("\\", "/")) 

1349 parts = [part for part in normalised.split("/") if part not in ("", ".")] 

1350 if not parts or parts[0] != "..": 

1351 return "/".join([".."] * len(parts)) or "." 

1352 if owner and repo and head_sha: 

1353 return f"https://github.com/{owner}/{repo}/blob/{head_sha}" 

1354 return None 

1355 

1356 

1357def _file_bullet(path: str, base: str | None) -> str: 

1358 """One **Files** bullet: a CommonMark link when there is a base, else the bare path. 

1359 

1360 The link text is the repository's own name for the file, which also puts every 

1361 path into the body — and :func:`retrieve_relevant_learnings` scores a document by 

1362 its text, so a query naming a file now scores the lesson about it higher than the 

1363 front-matter list alone did (the list matched once; the link matches again — in 

1364 its text, or in its destination when the text carries an escape). 

1365 The destination is percent-encoded: a space or a parenthesis in a path would 

1366 otherwise end the link where the path continues. 

1367 """ 

1368 name = _one_line(path) 

1369 if base is None: 

1370 # A code span ends at a backtick run as long as its opener, so a path carrying 

1371 # one is fenced with a run one longer, padded as CommonMark allows — never 

1372 # written plain, where its own Markdown would render. 

1373 longest = max((len(run) for run in re.findall(r"`+", name)), default=0) 

1374 if longest == 0: 

1375 return f"- `{name}`" 

1376 fence = "`" * (longest + 1) 

1377 return f"- {fence} {name} {fence}" 

1378 text = _LINK_TEXT_UNSAFE.sub(r"\\\1", name) 

1379 return f"- [{text}]({base}/{quote(name, safe='/')})" 

1380 

1381 

1382def _files_section(changed_files: list[str] | tuple[str, ...], base: str | None) -> list[str]: 

1383 """The **Files** section's lines, from the same list the front matter is given. 

1384 

1385 An empty list still renders the heading: the document shape is stable — four 

1386 sections, always — and a lesson about no file says so rather than pointing at 

1387 nothing. 

1388 """ 

1389 items = _strings(changed_files) 

1390 if not items: 

1391 return [LEARNING_NO_FILES] 

1392 return [_file_bullet(item, base) for item in items] 

1393 

1394 

1395def render_learning_document( 

1396 *, 

1397 title: str | None, 

1398 description: str | None, 

1399 pr_number: int | None, 

1400 issue_number: int | None, 

1401 repo: str | None, 

1402 date: str, 

1403 labels: list[str] | tuple[str, ...] = (), 

1404 changed_files: list[str] | tuple[str, ...] = (), 

1405 fingerprint: str = "", 

1406 what_changed: str = "", 

1407 what_we_learned: str = "", 

1408 do_differently: str = "", 

1409 file_link_base: str | None = None, 

1410) -> str: 

1411 """One learning file: frontmatter the reader can rely on, then four sections. 

1412 

1413 The heading is repeated below the frontmatter deliberately. `retrieve_relevant_learnings` 

1414 scores a file by its own text, and a title that lives only in frontmatter is a 

1415 title the search cannot weigh. 

1416 

1417 ``file_link_base`` is what :func:`learning_file_link_base` decided for the sink; 

1418 the **Files** section links each ``changed_files`` entry under it, and renders 

1419 the bare path when it is ``None``. The front matter does not read it: the list 

1420 up there is byte-for-byte what it was before the section existed (#1166). 

1421 """ 

1422 front = [ 

1423 "---", 

1424 f"schema: {LEARNING_SCHEMA_VERSION}", 

1425 f"title: {_yaml_scalar(title or 'Learning')}", 

1426 f"description: {_yaml_scalar(description or '')}", 

1427 f"repo: {_yaml_scalar(repo or '')}", 

1428 # `null`, not an empty value. Both read back as `None`; only one of them 

1429 # says so on purpose, and a bare `issue:` looks like a field somebody forgot 

1430 # to fill rather than a merge that was linked to no issue. 

1431 f"pr: {pr_number if pr_number is not None else 'null'}", 

1432 f"issue: {issue_number if issue_number is not None else 'null'}", 

1433 # Quoted like every other scalar in the block. Left bare, `2026-09-09` is a 

1434 # YAML *timestamp*: a real parser returns `datetime.date` where keel's reader 

1435 # returns the string, which is the one disagreement all this quoting exists 

1436 # to prevent. 

1437 f"date: {_yaml_scalar(date)}", 

1438 # Quoted like every other scalar: a sha256 that happens to be all digits is 

1439 # an `int` to a real parser and a 64-character string to keel's reader, and 

1440 # this is the field that *identifies* the lesson. 

1441 f"fingerprint: {_yaml_scalar(fingerprint)}", 

1442 ] 

1443 front += _yaml_sequence("labels", labels) 

1444 front += _yaml_sequence("changed_files", changed_files) 

1445 front.append("---") 

1446 body = [ 

1447 "", 

1448 # Through `_one_line`, like the front matter above: a heading built from a 

1449 # raw title let a newline open a section of its own inside the document. 

1450 f"# {_one_line(title or 'Learning')}", 

1451 "", 

1452 _one_line(description or ""), 

1453 "", 

1454 LEARNING_SECTION_HEADINGS[0], 

1455 "", 

1456 what_changed or LEARNING_EMPTY_SECTION, 

1457 "", 

1458 LEARNING_SECTION_HEADINGS[1], 

1459 "", 

1460 what_we_learned or LEARNING_EMPTY_SECTION, 

1461 "", 

1462 LEARNING_SECTION_HEADINGS[2], 

1463 "", 

1464 do_differently or LEARNING_EMPTY_SECTION, 

1465 "", 

1466 LEARNING_FILES_HEADING, 

1467 "", 

1468 *_files_section(changed_files, file_link_base), 

1469 "", 

1470 ] 

1471 return "\n".join(front + body) 

1472 

1473 

1474def learning_sink_writes( 

1475 *, 

1476 config: _HasPolicyPack | None, 

1477 decision: dict[str, Any] | None, 

1478 capture_status: str | None, 

1479) -> bool: 

1480 """Whether this run writes a learning file at all. 

1481 

1482 Split out so a caller can answer it **before** doing anything expensive — the 

1483 writer redacts its values before rendering, and reaching for the redaction 

1484 policy on a run with no sink turned an invalid `capture_redaction` pattern into 

1485 an exception raised from the wrong place, past the handler `keel ship` has for 

1486 exactly that. :func:`learning_sink_plan` asks the same question through this 

1487 function, so the two cannot drift. 

1488 

1489 Three conditions, and the third is the one that took two rounds to get right. 

1490 **The decision is the gate**, not merely the dedupe: `learning_decision` already 

1491 answers whether this run earns a durable artifact, and `create-learning` is the 

1492 one answer whose `durable_artifact` is true. Refusing only `duplicate` let four 

1493 other answers through — `learning.enabled` false, `enabled` omitted, `mode: 

1494 defer`, `mode: marker-only` — each planning a write while the record beside it 

1495 said the policy had decided not to keep one. A configured sink is where a 

1496 project's learnings go, not permission to write one whatever the policy says. 

1497 """ 

1498 if capture_status != "applied": 

1499 return False 

1500 if not capture_hook_enabled(config): 

1501 return False 

1502 sink = learning_sink_policy(config) 

1503 if sink is None or learning_sink_errors(sink): 

1504 return False 

1505 return isinstance(decision, dict) and decision.get("decision") == "create-learning" 

1506 

1507 

1508def learning_sink_plan( 

1509 *, 

1510 config: _HasPolicyPack | None, 

1511 decision: dict[str, Any] | None, 

1512 capture_status: str | None, 

1513 owner: str | None, 

1514 repo: str | None, 

1515 base_branch: str | None, 

1516 date: str, 

1517 pr_number: int | None, 

1518 title: str | None = None, 

1519 description: str | None = None, 

1520 labels: list[str] | tuple[str, ...] = (), 

1521 changed_files: list[str] | tuple[str, ...] = (), 

1522 issue_number: int | None = None, 

1523 what_changed: str = "", 

1524 what_we_learned: str = "", 

1525 do_differently: str = "", 

1526 head_sha: str | None = None, 

1527) -> dict[str, Any] | None: 

1528 """What to write for this run, or `None` with the reason folded into the caller. 

1529 

1530 Pure: it resolves a path and renders a document and touches no filesystem and no 

1531 clock — `date` is passed in for the same reason every other plan in this package 

1532 takes its facts as arguments. 

1533 

1534 Returns `None` when :func:`learning_sink_writes` says there is nothing to write: 

1535 no sink configured, a capture that is not `applied`, or a learning decision that 

1536 does not call for a durable artifact — a `duplicate`, which is the dedupe doing 

1537 its job, but equally a project whose policy said `marker-only`. 

1538 """ 

1539 if not learning_sink_writes(config=config, decision=decision, capture_status=capture_status): 

1540 return None 

1541 sink = learning_sink_policy(config) or {} 

1542 # No `isinstance` re-check: the gate above returned for anything that is not a 

1543 # `create-learning` mapping, so by here the decision is one. 

1544 fingerprint = str(decision.get("fingerprint") or "") 

1545 values = { 

1546 "owner": owner or "", 

1547 "repo": repo or "", 

1548 "base_branch": base_branch or "", 

1549 "date": date, 

1550 "pr": str(pr_number) if pr_number is not None else "", 

1551 "slug": _slugify(title), 

1552 "fingerprint": fingerprint[:LEARNING_FINGERPRINT_SLICE], 

1553 } 

1554 # **A relative template stays relative.** `{repo}/learnings` with `repo` unset 

1555 # expands to `/learnings` — absolute, at the filesystem root — while every 

1556 # reader of the template's shape (`learning_sink_in_worktree`, and so 

1557 # `commit_required` and the adapter's `git add`) still calls it in-repo. The 

1558 # filename's expansion was already flattened; the directory's was not. 

1559 directory = _relative_stays_relative( 

1560 str(sink.get("path") or DEFAULT_LEARNING_SINK_PATH), values 

1561 ) 

1562 filename = _one_component( 

1563 _expand(str(sink.get("filename") or DEFAULT_LEARNING_SINK_FILENAME), values) 

1564 ) 

1565 # Decided from the sink's *shape*, the way `learning_sink_in_worktree` decides 

1566 # who commits the file: the same question, and the two answers agree about where 

1567 # the document sits — except when a placeholder's expansion climbs out of the 

1568 # checkout, which the link base sees and the template reader does not, and then 

1569 # the outside form is the safe one. 

1570 file_link_base = learning_file_link_base( 

1571 directory=directory, 

1572 in_repo=learning_sink_in_worktree(config), 

1573 owner=owner, 

1574 repo=repo, 

1575 head_sha=head_sha, 

1576 ) 

1577 return { 

1578 "kind": sink.get("kind", LEARNING_SINK_KINDS[0]), 

1579 "directory": directory, 

1580 "filename": filename, 

1581 "file_link_base": file_link_base, 

1582 "content": render_learning_document( 

1583 title=title, 

1584 description=description, 

1585 pr_number=pr_number, 

1586 issue_number=issue_number, 

1587 repo=repo, 

1588 date=date, 

1589 labels=labels, 

1590 changed_files=changed_files, 

1591 fingerprint=fingerprint, 

1592 what_changed=what_changed, 

1593 what_we_learned=what_we_learned, 

1594 do_differently=do_differently, 

1595 file_link_base=file_link_base, 

1596 ), 

1597 } 

1598 

1599 

1600def duplicate_learning_artifact( 

1601 *, 

1602 config: _HasPolicyPack | None, 

1603 decision: dict[str, Any] | None, 

1604 capture_status: str | None, 

1605 existing_records: list[dict[str, Any]] | tuple[dict[str, Any], ...] = (), 

1606) -> str | None: 

1607 """The artifact an earlier run already wrote for this same learning, or `None`. 

1608 

1609 A `duplicate` decision writes no file — that is the dedupe working — but the 

1610 run still records `applied`, and `applied` with no artifact is exactly what 

1611 `capture-verify` reports as a finding. The lesson is not missing: it is on 

1612 disk, under the run this one duplicates. Naming that file keeps the claim 

1613 provable instead of letting the dedupe manufacture the gap the artifact 

1614 exists to close. 

1615 

1616 **Only for an `applied` capture**, and that gate is the whole point rather than 

1617 a precaution: `learning_decision` answers `duplicate` on a fingerprint match 

1618 before it looks at the status, so without it a `not-run` or `skipped` record 

1619 was handed the earlier run's path — the exact contradiction the CLI refuses at 

1620 its flag boundary, *a run that never reached capture produced no artifact*, and 

1621 the one `record_marker` states for `deferred` and `skipped`. 

1622 

1623 Pure: it reads recorded paths and never asks whether one still exists. The 

1624 caller that can answer that is the caller that touches the filesystem. 

1625 """ 

1626 if capture_status != "applied": 

1627 return None 

1628 # **The hook, not just the sink block.** `learning_decision` answers 

1629 # `duplicate` on a fingerprint match *before* it reads the enabled flags, so 

1630 # `duplicate` reaches here under a `capture.enabled: false` or `marker-only` 

1631 # project — and the caller treats a returned path as permission to replace the 

1632 # operator's own `--capture-artifact` with a stale sink file, for a project 

1633 # whose contract says `extension-owned`. Fifth reader of the same question. 

1634 if not capture_hook_enabled(config) or learning_sink_policy(config) is None: 

1635 return None 

1636 if not isinstance(decision, dict) or decision.get("decision") != "duplicate": 

1637 return None 

1638 fingerprint = decision.get("fingerprint") 

1639 if not fingerprint: 

1640 return None 

1641 artifact: str | None = None 

1642 for record in existing_records: 

1643 if not isinstance(record, dict): 

1644 continue 

1645 capture_block = record.get("capture") 

1646 if not isinstance(capture_block, dict): 

1647 continue 

1648 learning = capture_block.get("learning") 

1649 if not isinstance(learning, dict) or learning.get("fingerprint") != fingerprint: 

1650 continue 

1651 candidate = capture_block.get("artifact") 

1652 if isinstance(candidate, str) and candidate.strip(): 

1653 # Scope is recorded, not acted on here. Whether a `machine`-scoped path can be 

1654 # read is a filesystem question, and this module is pure — the CLI wrapper 

1655 # already asks it (`_duplicate_learning_artifact`), which is what keeps the 

1656 # same-host case working: the file really is there, and dropping the path here 

1657 # would record `applied` with no artifact on the very machine that wrote it. 

1658 # Keep scanning: the ledger is append-only and the newest record 

1659 # holding this fingerprint is the one whose path is current. 

1660 artifact = candidate.strip() 

1661 return artifact 

1662 

1663 

1664def _unquote(value: str) -> str: 

1665 """Undo :func:`_yaml_scalar` for the fields this reader uses. 

1666 

1667 The writer quotes any value containing a colon — an issue title usually does — 

1668 so a reader that took the raw text would hand back a title wrapped in quotes. 

1669 """ 

1670 if len(value) >= 2 and value[0] == value[-1] == '"': 

1671 return value[1:-1].replace('\\"', '"').replace("\\\\", "\\") 

1672 return value 

1673 

1674 

1675def _front_matter(content: str) -> tuple[dict[str, str], str]: 

1676 """Split a leading `---` block off, as `(fields, body)`. 

1677 

1678 Only the scalar fields this contract defines are read; a list value (`labels:`) 

1679 is skipped rather than parsed, because the caller wants a title and a sentence 

1680 and nothing here should grow into a YAML parser. 

1681 """ 

1682 lines = content.splitlines() 

1683 if not lines or lines[0].strip() != "---": 

1684 return {}, content 

1685 fields: dict[str, str] = {} 

1686 for index, line in enumerate(lines[1:], start=1): 

1687 if line.strip() == "---": 

1688 return fields, "\n".join(lines[index + 1 :]) 

1689 key, sep, value = line.partition(":") 

1690 if sep and not key.startswith(" "): 

1691 fields[key.strip()] = _unquote(value.strip()) 

1692 # No closing delimiter: not front matter, whatever it looked like. 

1693 return {}, content 

1694 

1695 

1696def _learning_title_and_summary(content: str, fallback: str) -> tuple[str, str]: 

1697 """A learning file's title and one-line summary. 

1698 

1699 Front matter first, because that is what the writer fills. Read line-by-line 

1700 instead, a file written by `render_learning_document` would be titled `---` and 

1701 summarised `schema: keel.learning.v1` — measured, and the reason the reader is 

1702 part of the change that added the writer. 

1703 """ 

1704 fields, body = _front_matter(content) 

1705 lines = [line.strip() for line in body.splitlines() if line.strip()] 

1706 title = fields.get("title") or (lines[0].lstrip("#").strip() if lines else fallback) 

1707 summary = fields.get("description") or "" 

1708 if not summary: 

1709 for line in lines[1:]: 

1710 if not line.startswith("#"): 

1711 summary = line 

1712 break 

1713 return title or fallback, summary 

1714 

1715 

1716#: Placeholders a `source` directory may use. A subset of the sink's: `{pr}`, 

1717#: `{date}` and `{slug}` name one *document*, and a directory that named a single 

1718#: PR would retrieve only that PR's lesson. 

1719LEARNING_SOURCE_PLACEHOLDERS = ("owner", "repo", "base_branch") 

1720 

1721#: How many lessons a brief carries, and how much of it they may take. A brief 

1722#: *becomes* an agent's prompt, so an unbounded retrieval is an unbounded prompt — 

1723#: the same reason `fixloop` clamps a reviewer's findings. 

1724DEFAULT_LEARNING_RETRIEVAL_LIMIT = 5 

1725LEARNING_BRIEF_CHAR_BUDGET = 4000 

1726LEARNING_BRIEF_SUMMARY_CHARS = 240 

1727 

1728#: The heading both briefs use, fixed so an agent reading two of them recognises 

1729#: the same section rather than inferring one from prose. 

1730LEARNING_BRIEF_HEADING = "Relevant past learnings" 

1731LEARNING_RETRIEVAL_SCHEMA_VERSION = "keel.learning-retrieval.v1" 

1732 

1733#: What an exact front-matter match is worth beside a text hit. A file whose 

1734#: `labels` or `changed_files` name this task's own label or path is *about* this 

1735#: task; one that merely says "auth" eight times is worded like it. Text scoring 

1736#: alone ranked the second above the first. 

1737LEARNING_LABEL_MATCH_SCORE = 6 

1738LEARNING_FILE_MATCH_SCORE = 8 

1739 

1740#: keel's own scaffolding, emitted into **every** document this writer produces. The 

1741#: scorer subtracts it before counting words: an issue titled *"What changed in the 

1742#: merge window"* otherwise scored `what` and `changed` against all three headings of 

1743#: every learning in the directory and cleared the floor on all of them, so three 

1744#: unrelated lessons opened the brief. Named here rather than spelled twice, so the 

1745#: writer and the scorer cannot drift. 

1746LEARNING_SECTION_HEADINGS = ( 

1747 "## What changed", 

1748 "## What we learned", 

1749 "## What to do differently next time", 

1750) 

1751LEARNING_EMPTY_SECTION = "_Not recorded._" 

1752 

1753#: How many **distinct** query words a file must contain to be a text match at all. 

1754#: One is a coincidence — "keel" appears in every learning this repository writes — 

1755#: and repetition does not make it less of one, which a point floor could not say: at 

1756#: four points a single word said four times passed, and a genuine three-word match 

1757#: in a short handwritten note did not. An exact front-matter match is admitted on 

1758#: its own, whatever the text says. 

1759LEARNING_MIN_DISTINCT_TOKENS = 2 

1760 

1761#: Words too common to distinguish one learning from another. 

1762_LEARNING_STOPWORDS = frozenset( 

1763 {"the", "and", "for", "with", "this", "that", "issue", "feat", "fix"} 

1764) 

1765 

1766 

1767def learning_source_errors(source: Any) -> list[str]: 

1768 """Why this `source` cannot be used, or `[]`. 

1769 

1770 Checked where the config is read, like the sink's templates and for the same 

1771 reason: a directory naming `{repoo}` is a typo whose only symptom is a 

1772 retrieval that silently finds nothing, on every run, forever. 

1773 """ 

1774 if source in (None, [], ()): 

1775 return [] 

1776 entries = [source] if isinstance(source, str) else source 

1777 if not isinstance(entries, (list, tuple)): 

1778 return ["policy_pack.capture.learning.source must be a string or a list of strings"] 

1779 errors: list[str] = [] 

1780 for entry in entries: 

1781 if not isinstance(entry, str) or not entry.strip(): 

1782 errors.append("policy_pack.capture.learning.source entries must be non-empty strings") 

1783 continue 

1784 for name in re.findall(r"\{([^}]*)\}", entry): 

1785 if name not in LEARNING_SOURCE_PLACEHOLDERS: 

1786 errors.append( 

1787 f"policy_pack.capture.learning.source uses unknown placeholder " 

1788 f"{{{name}}}; known: {', '.join(LEARNING_SOURCE_PLACEHOLDERS)}" 

1789 ) 

1790 return errors 

1791 

1792 

1793def learning_source_entries(config: _HasPolicyPack | None) -> list[str]: 

1794 """The directories a project *configured*, before any placeholder is expanded. 

1795 

1796 Unset, it is **the directory the sink writes to** — a project that turned 

1797 capture on has exactly one place its learnings live, and a second setting to 

1798 keep in step with the first is a second setting to get wrong. With no sink 

1799 either, the `.keel/learning/` convention, so retrieval works the moment a 

1800 project has files however they got there. 

1801 

1802 Separate from :func:`learning_source_dirs` because the pure contract has no 

1803 `{repo}` to expand with: reporting the resolved list there would print an 

1804 empty `sources` for a templated setting that reads fine at run time. 

1805 """ 

1806 policy = _learning_policy(config) 

1807 raw = policy.get("source") 

1808 if isinstance(raw, str): 

1809 entries = [raw] 

1810 elif isinstance(raw, (list, tuple)): 

1811 entries = [entry for entry in raw if isinstance(entry, str)] 

1812 else: 

1813 entries = [] 

1814 if not entries: 

1815 sink = learning_sink_policy(config) or {} 

1816 entries = [str(sink.get("path") or DEFAULT_LEARNING_SINK_PATH)] 

1817 return entries 

1818 

1819 

1820def learning_source_dirs( 

1821 config: _HasPolicyPack | None, 

1822 *, 

1823 values: dict[str, str] | None = None, 

1824) -> list[str]: 

1825 """The directories this run reads, with the placeholders expanded. 

1826 

1827 Pure. An entry still holding a placeholder after expansion is dropped rather 

1828 than taken literally, which is what keeps a per-document sink path (`…/{pr}/`) 

1829 from being read as a folder called `{pr}`. 

1830 """ 

1831 resolved: list[str] = [] 

1832 for entry in learning_source_entries(config): 

1833 expanded = _expand(entry, values or {}).strip() 

1834 if not expanded or "{" in expanded: 

1835 continue 

1836 if expanded not in resolved: 

1837 resolved.append(expanded) 

1838 return resolved 

1839 

1840 

1841def learning_query_text( 

1842 *, 

1843 title: str | None = None, 

1844 labels: list[str] | tuple[str, ...] = (), 

1845 changed_files: list[str] | tuple[str, ...] = (), 

1846) -> str: 

1847 """The text a task is matched by: its title, its labels, and its paths. 

1848 

1849 The paths belong in it. A lesson about `src/keel/ledger.py` and an issue that 

1850 touches `src/keel/ledger.py` share no words at all when only titles are 

1851 compared, which is the pairing retrieval most needs to make. 

1852 """ 

1853 parts = [title or "", *_strings(labels), *_strings(changed_files)] 

1854 return " ".join(part for part in parts if part) 

1855 

1856 

1857def _front_matter_lists(content: str) -> dict[str, list[str]]: 

1858 """Read string sequences from a bounded, complete YAML front matter block. 

1859 

1860 Use the same safe parser as project configuration for block and flow lists. 

1861 Only direct string items are consumed; aliases cannot cause recursive walks. 

1862 Invalid or oversized metadata contributes no exact matches. 

1863 """ 

1864 lines = content.splitlines() 

1865 if not lines or lines[0].strip() != "---": 

1866 return {} 

1867 for index, line in enumerate(lines[1:], start=1): 

1868 if line.strip() == "---": 

1869 front = "\n".join(lines[1:index]) 

1870 if len(front) > 65536: 

1871 return {} 

1872 try: 

1873 fields = yaml.load(front) 

1874 except (yaml.YAMLError, RecursionError): 

1875 return {} 

1876 if not isinstance(fields, dict): 

1877 return {} 

1878 return { 

1879 key: [item for item in value if isinstance(item, str)] 

1880 for key in ("labels", "changed_files") 

1881 if isinstance(value := fields.get(key), list) 

1882 } 

1883 return {} 

1884 

1885 

1886def _learning_tokens(query_text: str) -> set[str]: 

1887 return { 

1888 word.lower() 

1889 for word in re.findall(r"[A-Za-z0-9_-]{3,}", query_text) 

1890 if word.lower() not in _LEARNING_STOPWORDS 

1891 } 

1892 

1893 

1894_LEARNING_METADATA_KEYS = ( 

1895 "schema", 

1896 "title", 

1897 "description", 

1898 "repo", 

1899 "pr", 

1900 "issue", 

1901 "date", 

1902 "fingerprint", 

1903 "labels", 

1904 "changed_files", 

1905) 

1906_LEARNING_METADATA_LINE = re.compile( 

1907 r"^(?:" + "|".join(_LEARNING_METADATA_KEYS) + r")\s*[:=].*$", 

1908 re.IGNORECASE, 

1909) 

1910 

1911 

1912def _lesson_text(body: str, suffix: str = ".md") -> str: 

1913 """Score prose without treating metadata-shaped lesson sentences as fields. 

1914 

1915 Markdown metadata was already removed with its front matter. JSON fields 

1916 are removed structurally; plain text recognizes a leading schema header, 

1917 ending at the first non-metadata line. Body sentences stay intact. 

1918 """ 

1919 if suffix == ".json": 

1920 try: 

1921 record = json.loads(body) 

1922 except (ValueError, RecursionError): 

1923 record = None 

1924 if isinstance(record, dict): 

1925 body = "\n".join( 

1926 value 

1927 for key, value in record.items() 

1928 if key.lower() not in _LEARNING_METADATA_KEYS and isinstance(value, str) 

1929 ) 

1930 elif suffix == ".txt": 

1931 lines = body.splitlines() 

1932 if lines and re.fullmatch(r"schema\s*[:=]\s*keel\.learning\.v1", lines[0], re.I): 

1933 index = 0 

1934 while index < len(lines) and _LEARNING_METADATA_LINE.fullmatch(lines[index]): 

1935 index += 1 

1936 body = "\n".join(lines[index:]) 

1937 text = body.lower() 

1938 for scaffold in ( 

1939 *LEARNING_SECTION_HEADINGS, 

1940 LEARNING_EMPTY_SECTION, 

1941 LEARNING_FILES_HEADING, 

1942 LEARNING_NO_FILES, 

1943 "## Changed files", 

1944 ): 

1945 text = text.replace(scaffold.lower(), " ") 

1946 return text 

1947 

1948 

1949def _text_score(content_lower: str, filename_lower: str, tokens: set[str]) -> tuple[int, int]: 

1950 """`(points, distinct tokens matched)` for one file. 

1951 

1952 The two answer different questions and only one of them decides admission. 

1953 Points order the results; **distinct tokens** say whether this is a match at 

1954 all, because repetition is not evidence — a note saying `ledger` twelve times 

1955 matches a query about ledgers exactly as much as one saying it twice. 

1956 """ 

1957 score = 0 

1958 distinct = 0 

1959 for token in tokens: 

1960 hit = False 

1961 if token in filename_lower: 

1962 score += 3 

1963 hit = True 

1964 count = content_lower.count(token) 

1965 if count > 0: 

1966 score += min(count, 5) 

1967 hit = True 

1968 distinct += hit 

1969 return score, distinct 

1970 

1971 

1972def _exact_matches(content: str, labels: set[str], changed_files: set[str]) -> tuple[list, list]: 

1973 """The declared `labels` / `changed_files` this file and this task share. 

1974 

1975 Front matter only. A file without it is plain Markdown and falls back to text 

1976 scoring, which is the whole tolerance the read path promises: learnings keel 

1977 wrote and learnings a person wrote both rank. 

1978 """ 

1979 lists = _front_matter_lists(content) 

1980 matched_labels = sorted({_normalize_text(value) for value in lists.get("labels", [])} & labels) 

1981 matched_files = sorted( 

1982 {_normalize_path(value) for value in lists.get("changed_files", [])} & changed_files 

1983 ) 

1984 return matched_labels, matched_files 

1985 

1986 

1987def dedupe_learning_hits( 

1988 hits: list[dict[str, Any]] | tuple[dict[str, Any], ...], 

1989) -> list[dict[str, Any]]: 

1990 """One entry per lesson, keeping the first — which is the best-ranked. 

1991 

1992 Reading several directories is the point of a list, and the same lesson living 

1993 in two of them is the normal way that happens: a shared knowledge folder synced 

1994 into a checkout, a copy taken before a move. Left in, one lesson would take two 

1995 of the five slots a brief has and push a different one out. 

1996 

1997 Identity is the fingerprint, not the path: the same lesson under two names is 

1998 still one lesson. A hit without one (a hand-written file) falls back to its 

1999 path, which is the only thing that distinguishes it. 

2000 """ 

2001 seen: set[str] = set() 

2002 unique: list[dict[str, Any]] = [] 

2003 for hit in hits: 

2004 key = str(hit.get("fingerprint") or hit.get("path") or "") 

2005 if key in seen: 

2006 continue 

2007 seen.add(key) 

2008 unique.append(hit) 

2009 return unique 

2010 

2011 

2012def _render_learning_brief( 

2013 hits: list[dict[str, Any]] | tuple[dict[str, Any], ...], 

2014 *, 

2015 char_budget: int = LEARNING_BRIEF_CHAR_BUDGET, 

2016) -> tuple[str, list[dict[str, Any]]]: 

2017 """The fixed **Relevant past learnings** block and the hits that fit in budget. 

2018 

2019 Empty in, empty out, and the caller renders nothing at all then — which is 

2020 what keeps a project with no learnings byte-identical to the brief it got 

2021 before this existed. Zero cost when unused is a requirement, not a nicety. 

2022 """ 

2023 if not hits: 

2024 return "", [] 

2025 lines = [f"### {LEARNING_BRIEF_HEADING}", ""] 

2026 used = sum(len(line) + 1 for line in lines) 

2027 rendered_hits: list[dict[str, Any]] = [] 

2028 for hit in hits: 

2029 title = str(hit.get("title") or hit.get("file") or "Learning").strip() 

2030 summary = " ".join(str(hit.get("summary") or "").split()) 

2031 if len(summary) > LEARNING_BRIEF_SUMMARY_CHARS: 

2032 summary = summary[: LEARNING_BRIEF_SUMMARY_CHARS - 1].rstrip() + "\u2026" 

2033 path = str(hit.get("path") or hit.get("file") or "") 

2034 entry = f"- **{title}**" + (f" — {summary}" if summary else "") + f" (`{path}`)" 

2035 if used + len(entry) + 1 > char_budget: 

2036 # Stop at the budget rather than trimming the last entry into a 

2037 # sentence fragment: a brief that ends mid-lesson reads as a bug. 

2038 break 

2039 lines.append(entry) 

2040 used += len(entry) + 1 

2041 rendered_hits.append(hit) 

2042 if len(lines) == 2: 

2043 return "", [] 

2044 return "\n".join(lines) + "\n", rendered_hits 

2045 

2046 

2047def render_learning_brief_section( 

2048 hits: list[dict[str, Any]] | tuple[dict[str, Any], ...], 

2049 *, 

2050 char_budget: int = LEARNING_BRIEF_CHAR_BUDGET, 

2051) -> str: 

2052 """The fixed **Relevant past learnings** block, or `""` when nothing matched. 

2053 

2054 Empty in, empty out, and the caller renders nothing at all then — which is 

2055 what keeps a project with no learnings byte-identical to the brief it got 

2056 before this existed. Zero cost when unused is a requirement, not a nicety. 

2057 """ 

2058 section, _ = _render_learning_brief(hits, char_budget=char_budget) 

2059 return section 

2060 

2061 

2062def learning_retrieval_as_dict( 

2063 *, 

2064 sources: list[str] | tuple[str, ...] = (), 

2065 hits: list[dict[str, Any]] | tuple[dict[str, Any], ...] = (), 

2066 limit: int = DEFAULT_LEARNING_RETRIEVAL_LIMIT, 

2067 char_budget: int = LEARNING_BRIEF_CHAR_BUDGET, 

2068) -> dict[str, Any]: 

2069 """The retrieval block adapters read and the ledger records. 

2070 

2071 `section` is rendered here rather than left to each brief's author: the 

2072 implement brief and the reviewer brief must carry the *same* section, and two 

2073 renderers agree only until one of them is edited. 

2074 """ 

2075 section, rendered_hits = _render_learning_brief(hits, char_budget=char_budget) 

2076 entries = [ 

2077 { 

2078 "file": hit.get("file"), 

2079 "path": hit.get("path"), 

2080 "title": hit.get("title"), 

2081 "summary": hit.get("summary"), 

2082 "score": hit.get("score"), 

2083 "fingerprint": hit.get("fingerprint"), 

2084 "matched_labels": list(hit.get("matched_labels") or []), 

2085 "matched_files": list(hit.get("matched_files") or []), 

2086 } 

2087 for hit in rendered_hits 

2088 ] 

2089 return { 

2090 "schema_version": LEARNING_RETRIEVAL_SCHEMA_VERSION, 

2091 "heading": LEARNING_BRIEF_HEADING, 

2092 "policy_source": "policy_pack.capture.learning.source", 

2093 "sources": list(sources), 

2094 "limit": limit, 

2095 "hits": entries, 

2096 "fingerprints": [entry["fingerprint"] for entry in entries if entry["fingerprint"]], 

2097 "section": section, 

2098 } 

2099 

2100 

2101def retrieve_relevant_learnings( 

2102 query_text: str, 

2103 learning_dir: str | Path, 

2104 *, 

2105 max_results: int = 3, 

2106 min_score: int = 1, 

2107 min_distinct_tokens: int = LEARNING_MIN_DISTINCT_TOKENS, 

2108 labels: list[str] | tuple[str, ...] = (), 

2109 changed_files: list[str] | tuple[str, ...] = (), 

2110) -> list[dict[str, Any]]: 

2111 """Retrieve relevant historical learning records for an issue or task. 

2112 

2113 Stdlib-first token matching against Markdown or JSON learning files in 

2114 ``learning_dir`` (e.g. ``.keel/learning/``). Returns the top matching lessons 

2115 to be injected into implementation / review contexts. 

2116 

2117 ``labels`` and ``changed_files`` are this task's own, and a file that 

2118 *declares* one of them in its front matter is about this task rather than 

2119 merely worded like it — see :data:`LEARNING_LABEL_MATCH_SCORE`. A file with no 

2120 front matter is plain Markdown and scores on its text alone, which is the 

2121 tolerance this reader promises: a lesson keel wrote and a lesson a person 

2122 wrote both rank. 

2123 

2124 The one function in this module that reads the filesystem, and it was written 

2125 that way before the pure/thin-I/O split had a name for it. Everything it 

2126 decides is in the pure helpers around it. 

2127 """ 

2128 path = Path(learning_dir) 

2129 if not path.is_dir(): 

2130 return [] 

2131 

2132 tokens = _learning_tokens(query_text) 

2133 want_labels = {_normalize_text(label) for label in _strings(labels)} 

2134 want_files = {_normalize_path(name) for name in _strings(changed_files)} 

2135 if not tokens and not want_labels and not want_files: 

2136 return [] 

2137 

2138 # **Only files that really live here.** A lesson is text an agent brief quotes, and a 

2139 # link in this directory — `.keel/learning/x.md -> ~/.aws/credentials` — made whatever 

2140 # it pointed at into one: `is_file` and `read_text` both follow it. Resolved on both 

2141 # ends, as the landing's `_contained_real_path` does, so a link to another lesson in 

2142 # the same directory still reads and a directory reached through a link still works. 

2143 real_dir = path.resolve() 

2144 results: list[dict[str, Any]] = [] 

2145 for file_path in sorted(path.glob("*")): 

2146 if not file_path.is_file() or file_path.suffix not in LEARNING_READ_SUFFIXES: 

2147 continue 

2148 if file_path.resolve().parent != real_dir: 

2149 continue 

2150 try: 

2151 content = file_path.read_text(encoding="utf-8", errors="replace") 

2152 except OSError: 

2153 continue 

2154 

2155 fields, body = _front_matter(content) 

2156 lists = _front_matter_lists(content) 

2157 matched_labels, matched_files = _exact_matches(content, want_labels, want_files) 

2158 # **The body, minus keel's own scaffolding.** Front matter is metadata with 

2159 # its own exact matching, and scoring it as prose made every learning this 

2160 # repository ever wrote match every task in it (`repo: keel` and `schema:` 

2161 # name the project in all of them). The section headings are the same problem 

2162 # one layer down: they are in every document, so a title sharing a word with 

2163 # one matched the whole directory. 

2164 lesson_text = _lesson_text(body, file_path.suffix) 

2165 changed_paths = " ".join(lists.get("changed_files", [])) 

2166 if changed_paths: 

2167 lesson_text = f"{lesson_text} {changed_paths.lower()}" 

2168 text_score, distinct = _text_score(lesson_text, file_path.name.lower(), tokens) 

2169 declared = len(matched_labels) + len(matched_files) 

2170 score = ( 

2171 text_score 

2172 + LEARNING_LABEL_MATCH_SCORE * len(matched_labels) 

2173 + LEARNING_FILE_MATCH_SCORE * len(matched_files) 

2174 ) 

2175 

2176 # **A declaration beats prose, whatever the prose says.** Added together, a 

2177 # file repeating the query's words a dozen times outscored one that *declared* 

2178 # the path the task touches — the ranking this whole scheme exists to get 

2179 # right — because `min(count, 5)` per token compounds and the bonus does not. 

2180 # It is a lexicographic key now: declared matches first, text only to break 

2181 # the tie among files that declare the same number. 

2182 if declared or (distinct >= min_distinct_tokens and text_score >= min_score): 

2183 title, summary = _learning_title_and_summary(content, file_path.name) 

2184 results.append( 

2185 { 

2186 "file": file_path.name, 

2187 "path": str(file_path), 

2188 "title": title, 

2189 "summary": summary, 

2190 "score": score, 

2191 "declared": declared, 

2192 # The writer's own fingerprint when there is one, so a later 

2193 # run can tell that this exact lesson was surfaced. A file 

2194 # written by hand has none; its content identifies it. 

2195 "fingerprint": fields.get("fingerprint") or _content_fingerprint(content), 

2196 "matched_labels": matched_labels, 

2197 "matched_files": matched_files, 

2198 } 

2199 ) 

2200 

2201 results.sort(key=lambda r: (-r["declared"], -r["score"], r["file"])) 

2202 return results[:max_results] 

2203 

2204 

2205#: Statuses :func:`learning_land_plan` and ``keel capture-land`` speak. ``landed`` 

2206#: and ``already-landed`` are both success — the second is what a retry of a run 

2207#: that already pushed reports, which is what makes the command idempotent. 

2208LEARNING_LAND_STATUSES = ( 

2209 "landed", 

2210 "already-landed", 

2211 "not-required", 

2212 "no-artifact", 

2213 "would-land", 

2214 "contended", 

2215 "failed", 

2216) 

2217 

2218#: How many times the landing rebuilds its commit when a concurrent ship pushed 

2219#: first. Each attempt re-reads ``origin/<base>``, so the rebuild is against the 

2220#: branch as it is *now* rather than the one the run started from; three is the 

2221#: same budget s6 gives CI fixes and bounds a live lock-step between two ships. 

2222LEARNING_LAND_ATTEMPTS = 3 

2223 

2224#: Marker on the landing commit, so the commit that carried a lesson onto the base 

2225#: branch can be found by the pull request it came from without parsing prose. 

2226LEARNING_LAND_SCHEMA_VERSION = "keel.capture-land.v1" 

2227 

2228#: The line the landing commit carries, and it is the **schema version**. 

2229#: 

2230#: This commit is the one thing keel pushes to a base branch outside a pull request, 

2231#: so a history reader — `git log`, `keel verify-merge`'s drift read, a person asking 

2232#: what this commit is — has to be able to tell it from a stray push. Naming it after 

2233#: the schema rather than inventing a second literal means the marker and the record 

2234#: cannot drift apart, and the name is greppable against the contract that defines it. 

2235LEARNING_LAND_MARKER = LEARNING_LAND_SCHEMA_VERSION 

2236 

2237 

2238def learning_land_message( 

2239 *, pr_number: int | None, path: str, issue_number: int | None = None 

2240) -> str: 

2241 """The landing commit's message — deterministic, so two runs agree byte for byte. 

2242 

2243 Consumer-neutral on purpose: it carries no vendor trailer. The commit is keel's, 

2244 made on behalf of whatever agent ran the ship, and a core command that stamped one 

2245 vendor's co-authorship onto every consumer's base branch would be asserting an 

2246 attribution it cannot know. The run ledger already records who implemented. 

2247 """ 

2248 subject = ( 

2249 f"chore(learning): record the lesson from PR #{pr_number}" 

2250 if pr_number is not None 

2251 else "chore(learning): record the lesson from this run" 

2252 ) 

2253 fields = " ".join( 

2254 f"{name}={value if value is not None else '-'}" 

2255 for name, value in (("pr", pr_number), ("issue", issue_number), ("path", path)) 

2256 ) 

2257 return f"{subject}\n\n{LEARNING_LAND_MARKER}: {fields}\n" 

2258 

2259 

2260def learning_land_plan( 

2261 config: _HasPolicyPack | None, 

2262 *, 

2263 artifact: str | None, 

2264 pr_number: int | None = None, 

2265 issue_number: int | None = None, 

2266 remote: str = "origin", 

2267 base_branch: str | None = None, 

2268 onto: str | None = None, 

2269 attempts: int = LEARNING_LAND_ATTEMPTS, 

2270) -> dict[str, Any]: 

2271 """Plan the landing of one learning artifact onto a branch. 

2272 

2273 ``onto`` names it — the pull request's own under `/keel:ship` (#1203) — and without it 

2274 the target is ``origin/<base_branch>`` (#1163). 

2275 

2276 Pure: it reads the config's *shape* and the artifact's path and answers what the 

2277 I/O layer should do, the way :func:`learning_sink_plan` answers what the writer 

2278 should write. Nothing here touches git or the filesystem, so the decision that 

2279 keel pushes to a base branch at all is unit-tested offline. 

2280 

2281 The plan deliberately describes **plumbing, not a checkout**. keel runs its own 

2282 s0–s12 inside a worktree while the primary checkout holds the base branch, so 

2283 every recipe built on ``git switch <base>`` exits 128 with *'<base>' is already 

2284 used by worktree* — the defect #1163 was opened for. Building the commit with 

2285 ``hash-object`` / ``ls-tree`` / ``mktree`` / ``commit-tree`` against 

2286 ``<remote>/<base>`` needs no checkout of the base branch at all, so it runs the 

2287 same from a worktree, from the primary checkout, and from a bare CI clone. 

2288 

2289 ``status`` is ``not-required`` when the sink is outside the repository (git never 

2290 sees it, so there is nothing to land), ``no-artifact`` when no path was recorded, 

2291 and ``planned`` when the caller should go ahead. ``errors`` is non-empty only for 

2292 a path this command must refuse to write to the base branch. 

2293 """ 

2294 resolved_base = base_branch or getattr(config, "base_branch", None) or "" 

2295 # **The branch the commit goes to, kept apart from the base branch.** #1203 lands the 

2296 # lesson on the pull request's own branch, as its last commit, so it merges with the 

2297 # work it describes. The sink template still expands against the *base* branch — 

2298 # that is the value the writer used — so the two cannot share one variable. 

2299 target = onto or resolved_base 

2300 if not learning_sink_in_worktree(config): 

2301 return { 

2302 "schema_version": LEARNING_LAND_SCHEMA_VERSION, 

2303 "status": "not-required", 

2304 "reason": ( 

2305 "the learning sink is outside the checkout, so git never sees it and " 

2306 "nothing has to reach the base branch" 

2307 ), 

2308 "path": None, 

2309 "remote": remote, 

2310 "base_branch": resolved_base, 

2311 "onto": None, 

2312 "ref": None, 

2313 "remote_ref": None, 

2314 "message": None, 

2315 "attempts": attempts, 

2316 "errors": [], 

2317 } 

2318 normalized = _land_path(artifact) 

2319 sink_dir = _land_sink_root(config, pr_number=pr_number, base_branch=resolved_base) 

2320 errors: list[str] = [] 

2321 if not is_remote_name(remote) or (target and not is_branch_name(target)): 

2322 status = "failed" 

2323 reason = "the remote or the target branch is not a name git can take as one" 

2324 # **Refused before any git call sees it, by git's own rules.** Both reach 

2325 # `git fetch <remote> <ref>` as positional arguments. A leading `-` is read there as 

2326 # an option — `--upload-pack=<program>` runs a program — and a `:` makes the branch 

2327 # a two-sided refspec: `foo:refs/heads/main` moves the local `main` to `foo`'s tip. 

2328 # Refusing only the `-` shipped first and was not enough. Answered here, in the one 

2329 # place every landing is planned, rather than trusted to each wrapper downstream. 

2330 errors.append(f"remote {remote!r} or target {target!r} is not a valid name") 

2331 elif artifact is None or not str(artifact).strip(): 

2332 status = "no-artifact" 

2333 reason = "no capture artifact was recorded for this run, so there is nothing to land" 

2334 elif normalized is None: 

2335 status = "failed" 

2336 reason = "the capture artifact is not a path inside the repository" 

2337 # Refused rather than normalised. The command's whole job is to push one file 

2338 # to a shared base branch, so an artifact that climbs out of the checkout — 

2339 # `../x`, `/etc/x`, `C:\x` — is the one input that must never be resolved 

2340 # helpfully. `learning_sink_in_worktree` says the *sink* is in-repo; this says 

2341 # the recorded path is too, and they are answered from different values. 

2342 errors.append(f"capture artifact {artifact!r} is absolute or escapes the repository root") 

2343 elif sink_dir is None: 

2344 status = "failed" 

2345 reason = ( 

2346 "the learning sink path does not name a directory this command can resolve, " 

2347 "so there is nothing to confine the landing to" 

2348 ) 

2349 # A sink of `.`, or one whose directory still holds `{date}`/`{slug}`/ 

2350 # `{fingerprint}` after `{owner}`/`{repo}`/`{base_branch}`/`{pr}` are filled in. 

2351 # Both describe a boundary that would accept any path in the repository, which is 

2352 # the boundary this check exists to replace. 

2353 errors.append("the learning sink path cannot be resolved to a directory") 

2354 elif not path_under_sink(normalized, sink_dir): 

2355 status = "failed" 

2356 reason = f"the capture artifact is not inside the learning sink ({sink_dir})" 

2357 # **Inside the repository is not the containment this command needs.** Every 

2358 # path test above answers "could git address this?", and the answer is yes for 

2359 # `config/private.env` and for `src/keel/cli.py` alike — so a ledger record 

2360 # naming one of those fast-forwarded the shared base branch with it, one file 

2361 # at a time, and the exactly-one-file check downstream agreed because it was 

2362 # exactly one file. The sink is the only directory this command has any 

2363 # business writing to, and it is the one value that says which. 

2364 errors.append( 

2365 f"capture artifact {artifact!r} is not under the configured learning sink {sink_dir!r}" 

2366 ) 

2367 elif not resolved_base: 

2368 status = "failed" 

2369 reason = "the project declares no base branch to land the lesson on" 

2370 errors.append("base_branch is not configured") 

2371 else: 

2372 status = "planned" 

2373 reason = f"land {normalized} on {remote}/{target}" 

2374 return { 

2375 "schema_version": LEARNING_LAND_SCHEMA_VERSION, 

2376 "status": status, 

2377 "reason": reason, 

2378 "path": normalized, 

2379 # Published so the I/O layer can resolve the *real* artifact against the sink 

2380 # rather than against the checkout: a link inside the sink pointing at an 

2381 # untracked `.env` beside the code is in the repository, and a checkout-wide 

2382 # containment test says yes to it. 

2383 "sink": sink_dir, 

2384 "remote": remote, 

2385 "base_branch": resolved_base, 

2386 "onto": target or None, 

2387 "ref": f"refs/heads/{target}" if target else None, 

2388 # Spelled in full, as `git.remote_tracking_ref` spells it and `git.fetch` writes it: 

2389 # the short `<remote>/<target>` resolves a tag or local branch of that name first, 

2390 # and the landing builds its commit on whatever this resolves to. 

2391 "remote_ref": f"refs/remotes/{remote}/{target}" if target else None, 

2392 "message": ( 

2393 learning_land_message(pr_number=pr_number, path=normalized, issue_number=issue_number) 

2394 if status == "planned" 

2395 else None 

2396 ), 

2397 "attempts": attempts, 

2398 "errors": errors, 

2399 } 

2400 

2401 

2402def _land_sink_root( 

2403 config: _HasPolicyPack | None, *, pr_number: int | None, base_branch: str 

2404) -> str | None: 

2405 """The directory the landing is allowed to write inside, as a repo-relative path. 

2406 

2407 The sink's ``path`` is a **template**, and `learning_sink_plan` writes the expanded 

2408 form — so comparing a recorded artifact against the literal `.keel/{repo}/learning` 

2409 refuses every lesson a project with placeholders ever writes. It is expanded here 

2410 with the same values, through the same `_relative_stays_relative` that keeps a 

2411 relative template relative when a leading placeholder expands to nothing. 

2412 

2413 Only the **directory-shaped** placeholders are resolvable here, and this command has 

2414 all four of them. One that still holds ``{date}``, ``{slug}`` or ``{fingerprint}`` 

2415 afterwards names a place nobody can point at, and the answer is ``None`` — the caller 

2416 refuses rather than guessing. Two earlier shapes were tried and are recorded because 

2417 both looked reasonable: confining to the known *prefix* accepts anything under 

2418 ``.keel/``, and matching an unresolved component as a *wildcard* is not a boundary at 

2419 all — a sink of ``{date}`` then makes the first component match anything, so 

2420 ``config/private.env`` is "inside" it. 

2421 """ 

2422 # `or {}` rather than a guard: the one caller reaches this only after 

2423 # `learning_sink_in_worktree` said there *is* a sink, and an empty block takes the 

2424 # documented default anyway — a branch no input can take is a claim the tests 

2425 # cannot check. 

2426 sink = learning_sink_policy(config) or {} 

2427 values = { 

2428 "owner": str(getattr(config, "owner", "") or ""), 

2429 "repo": str(getattr(config, "repo", "") or ""), 

2430 "base_branch": base_branch, 

2431 "pr": str(pr_number) if pr_number is not None else "", 

2432 } 

2433 expanded = _relative_stays_relative(str(sink.get("path") or DEFAULT_LEARNING_SINK_PATH), values) 

2434 # **Every placeholder resolved, or no sink root at all.** `{owner}`, `{repo}`, 

2435 # `{base_branch}` and `{pr}` are the directory-shaped ones and this command has all 

2436 # four. `{date}`, `{slug}` and `{fingerprint}` vary per run and belong in `filename`; 

2437 # left in a directory they name a place nobody can point at, and the landing says so 

2438 # rather than guessing — a wildcard component is not a boundary, it is a hole. 

2439 return None if "{" in expanded else _land_path(expanded) 

2440 

2441 

2442def path_under_sink(path: str, directory: str) -> bool: 

2443 """Is POSIX ``path`` inside ``directory``? Compared component by component. 

2444 

2445 Not a prefix test: a plain ``startswith`` says `.keel/learning-notes/x.md` is inside 

2446 `.keel/learning`, which is a different directory whose name merely begins the same way. 

2447 

2448 Every component of ``directory`` is a literal. A wildcard component was tried and is 

2449 not a boundary at all — a sink of `{date}` makes the first component match anything, 

2450 so `config/private.env` is "inside" it, and `{date}/learning` accepts 

2451 `config/learning/secrets.md`. The landing refuses a sink it cannot resolve instead, 

2452 so nothing here has to guess. 

2453 """ 

2454 wanted, have = directory.split("/"), path.split("/") 

2455 # Strictly deeper: the sink directory is not a file inside itself. 

2456 if len(have) <= len(wanted): 

2457 return False 

2458 return all(want == got for want, got in zip(wanted, have, strict=False)) 

2459 

2460 

2461#: The line a landing commit's message carries, as the prefix a reader matches on. 

2462LEARNING_LAND_MARKER_LINE = f"{LEARNING_LAND_MARKER}:" 

2463 

2464 

2465def land_sink_root( 

2466 config: _HasPolicyPack | None, *, pr_number: int | None, base_branch: str 

2467) -> str | None: 

2468 """The public name for :func:`_land_sink_root`, for the readers outside this module. 

2469 

2470 The evidence gate and the merge gate ask the same containment question the landing 

2471 asks, and a second copy of the answer is how the three would come to disagree. 

2472 """ 

2473 return _land_sink_root(config, pr_number=pr_number, base_branch=base_branch) 

2474 

2475 

2476def lesson_changed_files( 

2477 config: _HasPolicyPack | None, 

2478 changed_files: list[str] | tuple[str, ...], 

2479 *, 

2480 pr_number: int | None, 

2481 base_branch: str, 

2482) -> list[str]: 

2483 """The files a lesson is about: the pull request's, less the lessons in the sink (#1203). 

2484 

2485 The landing puts the lesson on the pull request itself, so once it has landed the host 

2486 lists the lesson among the files that pull request changed. The document is written 

2487 before the landing and the s11 record after the merge, and both read the host's list: 

2488 without this the record fingerprinted one more path than the document it names — the 

2489 mismatch the shared list exists to prevent — and a retried s10 rendered a lesson whose 

2490 **Files** section named the lesson. 

2491 

2492 Only an in-repo sink is subtracted, and only for a pull request. A sink outside the 

2493 checkout never appears in a pull request's files, and one whose path cannot be resolved 

2494 has no boundary to subtract by. Without a pull request there is no landing to subtract — 

2495 and a ``{pr}`` sink resolved with none reads as its parent (``docs/{pr}`` as ``docs``), 

2496 which would take the run's real work in ``docs/`` out of its own lesson. 

2497 """ 

2498 paths = list(changed_files) 

2499 if pr_number is None or not learning_sink_in_worktree(config): 

2500 return paths 

2501 sink = _land_sink_root(config, pr_number=pr_number, base_branch=base_branch) 

2502 if sink is None: 

2503 return paths 

2504 return [path for path in paths if not path_under_sink(path, sink)] 

2505 

2506 

2507#: Characters `git check-ref-format` refuses anywhere in a ref name. 

2508_REF_FORBIDDEN = re.compile(r"[\x00-\x20\x7f~^:?*\[\\]") 

2509 

2510#: A remote *name* — what `--remote` documents — rather than a URL or a refspec. 

2511_REMOTE_NAME = re.compile(r"\A[A-Za-z0-9][A-Za-z0-9._-]*\Z") 

2512 

2513 

2514def is_branch_name(name: object) -> bool: 

2515 """Would ``git check-ref-format --branch`` accept ``name`` as a literal branch? (#1203) 

2516 

2517 Pure, so the landing plan can refuse a bad target before any git call sees it. The 

2518 rules are git's own, and the test holds this function to git's verdict case by case 

2519 rather than to a reading of the man page: 

2520 

2521 - not empty, not beginning with ``-`` (git reads that as an option) or ``/``; 

2522 - no ``:`` — ``foo:refs/heads/main`` is a two-sided refspec, and handed to 

2523 ``git fetch`` it moves the local ``main`` to another branch's tip (measured); 

2524 - no control character, space, ``~``, ``^``, ``?``, ``*``, ``[`` or backslash; 

2525 - no ``..``, no ``@{``, no ``//``, not ending in ``/`` or ``.``; 

2526 - no path component beginning with ``.`` or ending in ``.lock``. 

2527 

2528 ``HEAD`` is refused, as git refuses it. ``@`` alone is refused although git accepts it: 

2529 ``--branch`` expands it to the current branch, which is not a name a caller can mean as a 

2530 literal target. 

2531 """ 

2532 if not isinstance(name, str) or not name or name in ("@", "HEAD"): 

2533 return False 

2534 if name[0] in "-/" or name.endswith(("/", ".")): 

2535 return False 

2536 if _REF_FORBIDDEN.search(name) or ".." in name or "@{" in name or "//" in name: 

2537 return False 

2538 return not any(part.startswith(".") or part.endswith(".lock") for part in name.split("/")) 

2539 

2540 

2541def is_remote_name(name: object) -> bool: 

2542 """Is ``name`` a plain remote name — letters, digits, ``.``, ``_``, ``-``, no leading ``-``?""" 

2543 return isinstance(name, str) and bool(_REMOTE_NAME.match(name)) 

2544 

2545 

2546#: What a landing commit may do to its one path: write a new lesson, or rewrite one at the 

2547#: same path. Never rename, copy or remove — those are other changes wearing the marker. 

2548LANDING_FILE_STATUSES = ("added", "modified") 

2549 

2550 

2551def carries_landing_marker(message: object) -> bool: 

2552 """Does a commit message carry the ``keel.capture-land.v1:`` marker line? 

2553 

2554 One predicate for both of its readers: the head-pin exemption below, and the search for 

2555 a lesson already riding a pull request (``keel capture-land --write``). Two copies of the 

2556 match would drift apart, and the search would then find landings the exemption refuses. 

2557 """ 

2558 return isinstance(message, str) and any( 

2559 line.strip().startswith(LEARNING_LAND_MARKER_LINE) for line in message.splitlines() 

2560 ) 

2561 

2562 

2563def capture_only_descent( 

2564 base: str, 

2565 head: str, 

2566 commits: list[dict[str, Any]], 

2567 *, 

2568 sink: str | None, 

2569) -> bool: 

2570 """Does ``head`` descend from ``base`` by **capture commits and nothing else**? (#1203) 

2571 

2572 The question the head-pin exemption rests on. Every review verdict and every 

2573 gates-pass is pinned to the head it was recorded against, and #1203 puts the 

2574 learning on the pull request's own branch as its last commit — after review, before 

2575 the merge — which moves that head. A verdict for ``base`` still answers for ``head`` 

2576 exactly when nothing between them could change what was reviewed: 

2577 

2578 - every commit has **exactly one parent**, and they form an unbroken chain from 

2579 ``base`` to ``head`` — a merge commit could carry anything; 

2580 - every commit's message carries the ``keel.capture-land.v1:`` marker line — the 

2581 exemption is for commits that *say* they are a landing, not for any edit that 

2582 happens to touch the sink; 

2583 - every commit differs from its parent by **exactly one path**, and that path is 

2584 inside the configured sink — the same containment the landing enforces before it 

2585 pushes, asked again here of commits it did not necessarily build; 

2586 - and that one path was **added or modified** — not renamed, copied or removed. 

2587 

2588 ``commits`` is ordered oldest to newest, each a mapping with ``sha``, ``parents`` 

2589 (a list of SHAs), ``message``, ``files`` (every path the commit touches, a rename's 

2590 source included) and ``statuses`` (one per changed entry). Anything malformed or 

2591 absent is ``False``: this answer removes a requirement, so it fails closed. 

2592 

2593 A sink that cannot be resolved is ``False`` too. There is then no boundary to hold a 

2594 commit to, and "inside the sink" would silently mean "anywhere". 

2595 """ 

2596 if not base or not head or base == head or sink is None or not commits: 

2597 return False 

2598 previous = base 

2599 for commit in commits: 

2600 if not isinstance(commit, dict): 

2601 return False 

2602 parents = commit.get("parents") 

2603 if not isinstance(parents, list) or parents != [previous]: 

2604 return False 

2605 if not carries_landing_marker(commit.get("message")): 

2606 return False 

2607 files = commit.get("files") 

2608 if not isinstance(files, list) or len(files) != 1 or not isinstance(files[0], str): 

2609 return False 

2610 if not path_under_sink(files[0], sink): 

2611 return False 

2612 # **What happened to that path, not only which path it was.** A rename or a copy 

2613 # into the sink arrives as one entry naming a path inside it; `files` counts the 

2614 # path it came from too, and this refuses the status outright, so neither a moved 

2615 # reviewed file nor a deleted lesson reads as a landing. Absent is refused as well: 

2616 # a reader that cannot say what the commit did has not shown it only *added*. 

2617 statuses = commit.get("statuses") 

2618 if not isinstance(statuses, list) or len(statuses) != 1: 

2619 return False 

2620 if statuses[0] not in LANDING_FILE_STATUSES: 

2621 return False 

2622 sha = commit.get("sha") 

2623 if not isinstance(sha, str) or not sha: 

2624 return False 

2625 previous = sha 

2626 return previous == head 

2627 

2628 

2629def _land_path(artifact: str | None) -> str | None: 

2630 """``artifact`` as a repo-relative POSIX path, or ``None`` when it is not one. 

2631 

2632 Anchored on *any* platform for the same reason 

2633 :func:`learning_sink_in_worktree` is: a keel config and a keel ledger are the 

2634 same text wherever they are read, and a path this host calls relative is the 

2635 one the next host would push to its base branch. 

2636 """ 

2637 if artifact is None: 

2638 return None 

2639 raw = str(artifact).strip() 

2640 if not raw or raw.startswith("~"): 

2641 return None 

2642 normalized = posixpath.normpath(raw.replace("\\", "/")) 

2643 # **Both forms are tested, and the rewrite is why.** `is_root_anchored` asks each 

2644 # flavour of path whether it is absolute, and `PureWindowsPath("\\etc\\hostname")` 

2645 # says no — a rooted path with no drive letter is *drive-relative*, not absolute. 

2646 # The backslash rewrite then turned that same string into `/etc/hostname`, which 

2647 # `os.path.join(root, …)` resolves by discarding root entirely: `git hash-object -w` 

2648 # stored a file from outside the checkout in the object database, and the tree 

2649 # composition split it into an entry with an empty name. Asking the question after 

2650 # the rewrite as well as before is what closes it. 

2651 if workspace.is_root_anchored(raw) or workspace.is_root_anchored(normalized): 

2652 return None 

2653 if normalized in (".", "") or normalized.split("/")[0] == "..": 

2654 return None 

2655 return normalized 

2656 

2657 

2658#: git's own words for "the ref moved under you" — the only rejection worth retrying. 

2659#: Printed by the client when its remote-tracking ref is behind, and by the server in 

2660#: the reason it returns; both reach us through the same combined output. 

2661PUSH_CONTENTION_MARKERS = ("fetch first", "non-fast-forward", "non-fast forward") 

2662 

2663 

2664def push_rejection_is_contention(output: str | None) -> bool: 

2665 """Did this push fail because the ref **moved**, or because it is **refused**? 

2666 

2667 The retry exists for one case: another ship landed its lesson between this run's 

2668 read of ``<remote>/<base>`` and its push. Rebuilding on the branch as it now is 

2669 will then succeed, which is why that case retries. 

2670 

2671 Every other rejection will refuse again, and a protected base branch is the 

2672 ordinary one — ``! [remote rejected] … (protected branch hook declined)``, or a 

2673 ``pre-receive`` hook's own sentence. Treating it as contention burned three pushes 

2674 on something that cannot succeed and then reported that the branch *"moved under 

2675 every one of 3 attempt(s)"* — naming a cause that did not happen and hiding the 

2676 server's actual reason, which is the one thing the operator needs. 

2677 

2678 Judged from git's text because the exit code is 1 either way. An unrecognised 

2679 failure is **not** contention: this pushes to the shared base branch, so an 

2680 unexplained refusal stops after one attempt rather than being retried on a guess. 

2681 """ 

2682 text = (output or "").lower() 

2683 return any(marker in text for marker in PUSH_CONTENTION_MARKERS) 

2684 

2685 

2686#: One ``git ls-tree`` / ``git mktree`` line: ``<mode> SP <type> SP <sha> TAB <name>``. 

2687#: ``DOTALL`` because a name may contain a newline, which ``-z`` returns raw. 

2688_TREE_ENTRY_RE = re.compile(r"\A(\d{6}) (blob|tree|commit) ([0-9a-f]{40,64})\t(.+)\Z", re.DOTALL) 

2689 

2690#: git's own mode for a regular, non-executable file and for a subdirectory. 

2691TREE_MODE_BLOB = "100644" 

2692TREE_MODE_TREE = "040000" 

2693 

2694 

2695@dataclass(frozen=True) 

2696class TreeEntry: 

2697 """One entry of a git tree, as ``ls-tree`` prints it and ``mktree`` reads it.""" 

2698 

2699 mode: str 

2700 kind: str 

2701 sha: str 

2702 name: str 

2703 

2704 def render(self) -> str: 

2705 return f"{self.mode} {self.kind} {self.sha}\t{self.name}" 

2706 

2707 

2708def parse_tree_listing(listing: str | None) -> list[TreeEntry]: 

2709 """Parse ``git ls-tree -z`` output; unreadable records are dropped, not guessed at. 

2710 

2711 A missing directory is an empty listing rather than an error: landing a lesson 

2712 into a sink directory that does not exist on the base branch yet is the ordinary 

2713 first run, not a failure. 

2714 

2715 Records are NUL-separated (see :func:`keel.git.ls_tree`). A listing with no NUL in it 

2716 at all is read one record per line, so one that reached this from a LF-terminated 

2717 source still parses — the reader is permissive, the *writer* is the side that has to 

2718 be exact. 

2719 

2720 **Never both.** git allows a newline inside a name and ``-z`` returns it raw, so 

2721 splitting a ``-z`` listing on newlines as well cut that record in two; neither half 

2722 parsed, the entry was dropped, and the tree rebuilt from what was left no longer had 

2723 it — a landing commit that deleted a file it was never asked to touch. Without ``-z`` 

2724 git C-quotes such a name, so the line-per-record form has no raw newline to cut. 

2725 """ 

2726 text = listing or "" 

2727 entries: list[TreeEntry] = [] 

2728 for line in text.split("\0") if "\0" in text else text.split("\n"): 

2729 match = _TREE_ENTRY_RE.match(line) 

2730 if match is not None: 

2731 entries.append( 

2732 TreeEntry(match.group(1), match.group(2), match.group(3), match.group(4)) 

2733 ) 

2734 return entries 

2735 

2736 

2737def _tree_sort_key(entry: TreeEntry) -> str: 

2738 # git orders tree entries as if every directory name ended in "/", which is why 

2739 # "learning" and "learning.md" sort the way they do. `mktree` normalises the order 

2740 # itself; composing it correctly here keeps the pure result byte-stable so a test 

2741 # can assert the exact input the plumbing is handed. 

2742 return entry.name + "/" if entry.kind == "tree" else entry.name 

2743 

2744 

2745def upsert_tree_entry(listing: str | None, entry: TreeEntry) -> str: 

2746 """``listing`` with ``entry`` added or replacing the one of the same name. 

2747 

2748 Pure, and the whole reason the landing can be asserted offline: this is where a 

2749 lesson is grafted onto the base branch's tree, so "the commit differs from its 

2750 parent by exactly this one path" is a property of a string function rather than 

2751 of a live push nobody can re-run. 

2752 

2753 **NUL-terminated**, because the result is written to a subprocess's stdin and a 

2754 text-mode pipe rewrites ``\n`` as CRLF on Windows. ``git mktree`` accepts the 

2755 corrupted listing without complaint and writes a tree whose entry is named 

2756 ``<name>\r`` — a different SHA, exit 0, no error (measured). There is no newline 

2757 in this output to translate. 

2758 """ 

2759 kept = [existing for existing in parse_tree_listing(listing) if existing.name != entry.name] 

2760 kept.append(entry) 

2761 kept.sort(key=_tree_sort_key) 

2762 return "".join(f"{item.render()}\x00" for item in kept) 

2763 

2764 

2765def _content_fingerprint(content: str) -> str: 

2766 import hashlib 

2767 

2768 return hashlib.sha256(content.encode("utf-8")).hexdigest()