diff --git a/.claude-plugin/plugin.json b/.claude-plugin/plugin.json
index 1a1701b0..018f0384 100644
--- a/.claude-plugin/plugin.json
+++ b/.claude-plugin/plugin.json
@@ -1,7 +1,7 @@
{
"name": "commitlore",
"displayName": "CommitLore",
- "version": "1.2.2",
+ "version": "1.2.3",
"description": "Recorded decisions from git history, delivered to the agent before it edits. Constraints, alternatives already ruled out, and warnings left by whoever was here last.",
"author": {
"name": "MongLong0214",
diff --git a/.codex-plugin/plugin.json b/.codex-plugin/plugin.json
index 1a79f028..15cc39b1 100644
--- a/.codex-plugin/plugin.json
+++ b/.codex-plugin/plugin.json
@@ -1,6 +1,6 @@
{
"name": "commitlore",
- "version": "1.2.2",
+ "version": "1.2.3",
"description": "Decision memory from Git history, with verified capture for coding sessions.",
"author": {
"name": "MongLong0214",
diff --git a/CHANGELOG.md b/CHANGELOG.md
index 51d244d1..8150c2b6 100644
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -4,6 +4,59 @@ Release notes for 1.0.0, 1.0.1 and 1.0.2 are on the
[GitHub releases page](https://github.com/MongLong0214/commitlore/releases); they
were not written here.
+## 1.2.3
+
+The capture prompt was the session, so on a long session there was no prompt.
+
+**`capture` embedded the whole transcript in its prompt (#873).** Measured by the
+reporter: a 67,981,436-byte transcript produced a 67,468,122-byte prompt — 99.3%
+of the file, and larger than any model can consume, so the pipeline could not be
+completed and no record was ever written. The pipeline itself was fine, which is
+how they proved it: `tail -n 400` on the same transcript gave 537,250 bytes and
+the same command in the same repository worked.
+
+**The failure was silent, which is why it reads as the reason capture is not
+used.** Prompt-only mode reports `outcome: "empty"`, `staged: false`, exit 0. An
+operator who tried capture once on a real session got a prompt they could not
+use and no statement that anything was wrong, and did not try again. In the
+repository where this was measured, `stale` reported 1 record in 1000 commits
+while roughly twenty commits in one recent session carried real decision context
+and produced none — with `{"mode":"auto","unattended":true}` already set, so
+permission was never the obstacle.
+
+**The prompt now carries the end of the transcript within a byte budget**, 256
+KiB by default and `COMMITLORE_TRANSCRIPT_BUDGET_BYTES` to change it. The end
+rather than the beginning: a decision is taken near the end of the session that
+implements it, and the diff being captured is that end.
+
+**Three things the bound deliberately does not do.** It does not renumber — the
+window keeps the line numbers it has in the whole transcript, because
+verification reads the whole transcript and a window renumbered from 1 would
+have every `L-L` locator name a different line of the file it is
+checked against. It does not reach the hash — `source_hashes` and every quote
+check are still over the whole transcript, so a quote from outside the window
+still verifies and a caller passing the session it actually had is never told
+the transcript was substituted. And it does not stay quiet — the prompt says
+which lines it shows and how many were left out, and `transcript_window` says
+the same to `capture --json` and to `commitlore_prepare_capture`. A bounded
+prompt that did not say so would be the old silence in a smaller package.
+
+**One line can outrun the budget by itself.** A JSONL transcript line can hold an
+entire tool result. That line is shown from its end rather than dropped, and the
+window marks it partial, because a window of no lines is worse than a window of
+one partial line. The byte slice never leaves a split codepoint at the front: a
+replacement character inside a quotable line is a character nobody can copy back.
+
+**Not claimed.** Where the useful boundary is. 256 KiB is comfortably readable by
+current models and carries far more than the ~537 KB slice measured to work; the
+reporter said they had not measured where a good record stops needing context,
+and neither has this. Also unchanged, and worth stating because automation is
+built on it: `outcome` is `staged`, `empty` or `rejected`, and all three exit 0
+— anything driving capture must read `--json`, because an exit-code check reads
+a rejected record as a success.
+
+**Not reviewed.** No cross-provider review was run on this change.
+
## 1.2.2
One line of configuration, and the plugin's MCP server had never started for
diff --git a/README.ja.md b/README.ja.md
index b443698a..1ccf4a88 100644
--- a/README.ja.md
+++ b/README.ja.md
@@ -47,18 +47,18 @@
```bash
-curl -fsSL https://raw.githubusercontent.com/MongLong0214/commitlore/v1.2.2/install.sh | sh -s v1.2.2
+curl -fsSL https://raw.githubusercontent.com/MongLong0214/commitlore/v1.2.3/install.sh | sh -s v1.2.3
```
先にインストーラーを読みたいですか?
```bash
-curl -fsSLO https://raw.githubusercontent.com/MongLong0214/commitlore/v1.2.2/install.sh
-sh install.sh v1.2.2
+curl -fsSLO https://raw.githubusercontent.com/MongLong0214/commitlore/v1.2.3/install.sh
+sh install.sh v1.2.3
# あるいはスクリプトを使わずに。スクリプトが作るチェックアウトは自分でも作れます。
-git clone --depth 1 --branch v1.2.2 https://github.com/MongLong0214/commitlore
+git clone --depth 1 --branch v1.2.3 https://github.com/MongLong0214/commitlore
node commitlore/dist/commitlore.mjs --version
```
@@ -107,13 +107,13 @@ CommitLore はその判断をコードのそばに残します。
macOS と Linux:
```bash
-curl -fsSL https://raw.githubusercontent.com/MongLong0214/commitlore/v1.2.2/install.sh | sh -s v1.2.2
+curl -fsSL https://raw.githubusercontent.com/MongLong0214/commitlore/v1.2.3/install.sh | sh -s v1.2.3
```
Windows:
```powershell
-& ([scriptblock]::Create((irm https://raw.githubusercontent.com/MongLong0214/commitlore/v1.2.2/install.ps1))) v1.2.2
+& ([scriptblock]::Create((irm https://raw.githubusercontent.com/MongLong0214/commitlore/v1.2.3/install.ps1))) v1.2.3
```
Node.js 22.23.2+ と Git が必要です。スクリプトは何かを書き込む前に両方を確認します。
diff --git a/README.ko.md b/README.ko.md
index 396becd9..c5bb0b9f 100644
--- a/README.ko.md
+++ b/README.ko.md
@@ -47,18 +47,18 @@
```bash
-curl -fsSL https://raw.githubusercontent.com/MongLong0214/commitlore/v1.2.2/install.sh | sh -s v1.2.2
+curl -fsSL https://raw.githubusercontent.com/MongLong0214/commitlore/v1.2.3/install.sh | sh -s v1.2.3
```
먼저 설치기를 읽어 보고 싶나요?
```bash
-curl -fsSLO https://raw.githubusercontent.com/MongLong0214/commitlore/v1.2.2/install.sh
-sh install.sh v1.2.2
+curl -fsSLO https://raw.githubusercontent.com/MongLong0214/commitlore/v1.2.3/install.sh
+sh install.sh v1.2.3
# 또는 스크립트를 건너뜁니다. 스크립트가 만드는 체크아웃은 직접 만들 수 있습니다.
-git clone --depth 1 --branch v1.2.2 https://github.com/MongLong0214/commitlore
+git clone --depth 1 --branch v1.2.3 https://github.com/MongLong0214/commitlore
node commitlore/dist/commitlore.mjs --version
```
@@ -107,13 +107,13 @@ CommitLore는 그 판단을 코드 곁에 보관합니다.
macOS와 Linux:
```bash
-curl -fsSL https://raw.githubusercontent.com/MongLong0214/commitlore/v1.2.2/install.sh | sh -s v1.2.2
+curl -fsSL https://raw.githubusercontent.com/MongLong0214/commitlore/v1.2.3/install.sh | sh -s v1.2.3
```
Windows:
```powershell
-& ([scriptblock]::Create((irm https://raw.githubusercontent.com/MongLong0214/commitlore/v1.2.2/install.ps1))) v1.2.2
+& ([scriptblock]::Create((irm https://raw.githubusercontent.com/MongLong0214/commitlore/v1.2.3/install.ps1))) v1.2.3
```
Node.js 22.23.2+와 Git이 필요합니다. 스크립트는 무엇이든 쓰기 전에 둘을 확인합니다.
diff --git a/README.md b/README.md
index ce9331e3..9c0d0634 100644
--- a/README.md
+++ b/README.md
@@ -48,18 +48,18 @@
```bash
-curl -fsSL https://raw.githubusercontent.com/MongLong0214/commitlore/v1.2.2/install.sh | sh -s v1.2.2
+curl -fsSL https://raw.githubusercontent.com/MongLong0214/commitlore/v1.2.3/install.sh | sh -s v1.2.3
```
Prefer to read the installer first?
```bash
-curl -fsSLO https://raw.githubusercontent.com/MongLong0214/commitlore/v1.2.2/install.sh
-sh install.sh v1.2.2
+curl -fsSLO https://raw.githubusercontent.com/MongLong0214/commitlore/v1.2.3/install.sh
+sh install.sh v1.2.3
# Or skip the script: the checkout it makes is one you can make yourself.
-git clone --depth 1 --branch v1.2.2 https://github.com/MongLong0214/commitlore
+git clone --depth 1 --branch v1.2.3 https://github.com/MongLong0214/commitlore
node commitlore/dist/commitlore.mjs --version
```
@@ -109,13 +109,13 @@ preserve, not for narrating every change.
macOS and Linux:
```bash
-curl -fsSL https://raw.githubusercontent.com/MongLong0214/commitlore/v1.2.2/install.sh | sh -s v1.2.2
+curl -fsSL https://raw.githubusercontent.com/MongLong0214/commitlore/v1.2.3/install.sh | sh -s v1.2.3
```
Windows:
```powershell
-& ([scriptblock]::Create((irm https://raw.githubusercontent.com/MongLong0214/commitlore/v1.2.2/install.ps1))) v1.2.2
+& ([scriptblock]::Create((irm https://raw.githubusercontent.com/MongLong0214/commitlore/v1.2.3/install.ps1))) v1.2.3
```
Requires Node.js 22.23.2+ and Git. The script checks both before it writes anything.
diff --git a/README.zh-CN.md b/README.zh-CN.md
index 8448a86b..337ab275 100644
--- a/README.zh-CN.md
+++ b/README.zh-CN.md
@@ -47,18 +47,18 @@
```bash
-curl -fsSL https://raw.githubusercontent.com/MongLong0214/commitlore/v1.2.2/install.sh | sh -s v1.2.2
+curl -fsSL https://raw.githubusercontent.com/MongLong0214/commitlore/v1.2.3/install.sh | sh -s v1.2.3
```
想先阅读安装器吗?
```bash
-curl -fsSLO https://raw.githubusercontent.com/MongLong0214/commitlore/v1.2.2/install.sh
-sh install.sh v1.2.2
+curl -fsSLO https://raw.githubusercontent.com/MongLong0214/commitlore/v1.2.3/install.sh
+sh install.sh v1.2.3
# 或者跳过脚本:它创建的检出,你自己也能创建。
-git clone --depth 1 --branch v1.2.2 https://github.com/MongLong0214/commitlore
+git clone --depth 1 --branch v1.2.3 https://github.com/MongLong0214/commitlore
node commitlore/dist/commitlore.mjs --version
```
@@ -105,13 +105,13 @@ CommitLore 把那份判断留在代码旁边。
macOS 和 Linux:
```bash
-curl -fsSL https://raw.githubusercontent.com/MongLong0214/commitlore/v1.2.2/install.sh | sh -s v1.2.2
+curl -fsSL https://raw.githubusercontent.com/MongLong0214/commitlore/v1.2.3/install.sh | sh -s v1.2.3
```
Windows:
```powershell
-& ([scriptblock]::Create((irm https://raw.githubusercontent.com/MongLong0214/commitlore/v1.2.2/install.ps1))) v1.2.2
+& ([scriptblock]::Create((irm https://raw.githubusercontent.com/MongLong0214/commitlore/v1.2.3/install.ps1))) v1.2.3
```
需要 Node.js 22.23.2+ 和 Git。脚本会在写入任何内容前检查两者。
diff --git a/install.ps1 b/install.ps1
index 0e7bc0fd..055f8561 100644
--- a/install.ps1
+++ b/install.ps1
@@ -1,8 +1,8 @@
<#
Installs commitlore from source on Windows, for any agent that is not Claude Code.
- irm https://raw.githubusercontent.com/MongLong0214/commitlore/v1.2.2/install.ps1 | iex
- & ([scriptblock]::Create((irm https://raw.githubusercontent.com/MongLong0214/commitlore/v1.2.2/install.ps1))) v1.2.2
+ irm https://raw.githubusercontent.com/MongLong0214/commitlore/v1.2.3/install.ps1 | iex
+ & ([scriptblock]::Create((irm https://raw.githubusercontent.com/MongLong0214/commitlore/v1.2.3/install.ps1))) v1.2.3
Claude Code users do not need this script. The repository is itself a plugin
marketplace (ADR-0011), so two /plugin commands register the MCP server, the
diff --git a/install.sh b/install.sh
index c05ff382..f77ed7cb 100755
--- a/install.sh
+++ b/install.sh
@@ -1,8 +1,8 @@
#!/bin/sh
# Installs commitlore from source, for any agent that is not Claude Code.
#
-# curl -fsSL https://raw.githubusercontent.com/MongLong0214/commitlore/v1.2.2/install.sh | sh
-# curl -fsSL https://raw.githubusercontent.com/MongLong0214/commitlore/v1.2.2/install.sh | sh -s v1.2.2
+# curl -fsSL https://raw.githubusercontent.com/MongLong0214/commitlore/v1.2.3/install.sh | sh
+# curl -fsSL https://raw.githubusercontent.com/MongLong0214/commitlore/v1.2.3/install.sh | sh -s v1.2.3
#
# **Claude Code users do not need this script.** The repository is itself a
# plugin marketplace (ADR-0011), so two `/plugin` commands register the MCP
diff --git a/package-lock.json b/package-lock.json
index 24b1d386..282256ab 100644
--- a/package-lock.json
+++ b/package-lock.json
@@ -1,12 +1,12 @@
{
"name": "commitlore",
- "version": "1.2.2",
+ "version": "1.2.3",
"lockfileVersion": 3,
"requires": true,
"packages": {
"": {
"name": "commitlore",
- "version": "1.2.2",
+ "version": "1.2.3",
"license": "MIT",
"dependencies": {
"@modelcontextprotocol/sdk": "^1.30.0",
diff --git a/package.json b/package.json
index b2e96388..e3d0c2a1 100644
--- a/package.json
+++ b/package.json
@@ -1,6 +1,6 @@
{
"name": "commitlore",
- "version": "1.2.2",
+ "version": "1.2.3",
"description": "Git-native, lifecycle-aware decision memory for coding agents",
"license": "MIT",
"private": true,
diff --git a/server.json b/server.json
index f597fef4..503946f5 100644
--- a/server.json
+++ b/server.json
@@ -8,7 +8,7 @@
"source": "github"
},
"websiteUrl": "https://github.com/MongLong0214/commitlore#readme",
- "version": "1.2.2",
+ "version": "1.2.3",
"_meta": {
"io.modelcontextprotocol.registry/publisher-provided": {
"registryFit": "Distribution is a tagged git checkout plus a Claude Code plugin marketplace (ADR-0011 registry-free git distribution, ADR-0026 no compiled executables and no uploaded release asset), so no official package type applies and this record relies on websiteUrl plus publisher metadata.",
@@ -18,8 +18,8 @@
"/plugin marketplace add MongLong0214/commitlore",
"/plugin install commitlore@commitlore"
],
- "installer": "curl -fsSL https://raw.githubusercontent.com/MongLong0214/commitlore/v1.2.2/install.sh | sh -s v1.2.2",
- "release": "https://github.com/MongLong0214/commitlore/releases/tag/v1.2.2"
+ "installer": "curl -fsSL https://raw.githubusercontent.com/MongLong0214/commitlore/v1.2.3/install.sh | sh -s v1.2.3",
+ "release": "https://github.com/MongLong0214/commitlore/releases/tag/v1.2.3"
},
"runtime": {
"transport": "stdio",
diff --git a/src/commands/capture.ts b/src/commands/capture.ts
index f7e8de75..90e70de9 100644
--- a/src/commands/capture.ts
+++ b/src/commands/capture.ts
@@ -39,7 +39,7 @@ import {
configuredTrustedSignerFingerprints,
configuredTrustedAuthors,
} from '../core/trusted-authors.js';
-import { parseDraft } from '../core/harvest.js';
+import { parseDraft, type TranscriptWindow } from '../core/harvest.js';
import { gcPending } from '../core/pending-gc.js';
import type { GuardAdvisory } from '../core/pending.js';
@@ -80,6 +80,13 @@ export interface CaptureResult {
nonce: string | null;
staged: boolean;
prompt?: string;
+ /**
+ * What of the transcript `prompt` carries, present whenever `prompt` is
+ * (#873). The prompt is bounded; verification is not — it reads the whole
+ * transcript. Without this a caller could not tell a slice from the session,
+ * and the prompt was previously the session, at whatever size that was.
+ */
+ transcript_window?: TranscriptWindow;
guard_advisory?: GuardAdvisory | null;
/**
* Every reason a record was refused (#309). Both sources are included: the
@@ -238,6 +245,7 @@ const runCapturePipeline = (opts: {
nonce: null,
staged: false,
prompt: prepareResult.prompt,
+ transcript_window: prepareResult.transcript_window,
guard_advisory: prepareResult.guard_advisory,
};
}
@@ -383,7 +391,11 @@ export const register = (program: Command): void => {
// when a subcommand was invoked, so `capture gc` — which needs no transcript
// — would fail before its own action ran. The requirement is enforced in the
// action below instead, where it applies only to the capture flow itself.
- .option('--transcript ', 'path to the session transcript file')
+ .option(
+ '--transcript ',
+ 'path to the session transcript file (the prompt carries its last 256 KiB; ' +
+ 'COMMITLORE_TRANSCRIPT_BUDGET_BYTES changes that, and verification always reads all of it)',
+ )
.option('--diff ', 'path to the diff file (defaults to the staged diff)')
.option('--draft ', 'path to the draft JSON file (omit for prompt-only mode)')
.option('--out ', 'write the pending nonce to a file')
diff --git a/src/core/capture-prepare.ts b/src/core/capture-prepare.ts
index da1f1f82..6045de63 100644
--- a/src/core/capture-prepare.ts
+++ b/src/core/capture-prepare.ts
@@ -10,7 +10,7 @@ import { createHash, randomBytes } from 'node:crypto';
import { markCaptureError } from './capture-outcome.js';
import { execGitOrThrow } from './git.js';
import { guard, renderGuardMatch, type GuardResult } from './guard.js';
-import { buildHarvestPrompt } from './harvest.js';
+import { buildHarvestPromptWithWindow, type TranscriptWindow } from './harvest.js';
import { policySourceLabel, resolvePolicy } from './capture-policy.js';
import {
createPending,
@@ -136,6 +136,13 @@ export interface PrepareResult {
policy_identity_hash: string;
source_hashes: { transcript: string; diff: string };
prompt: string;
+ /**
+ * What of the transcript the prompt carries (#873). The prompt is bounded;
+ * `source_hashes.transcript` and every verification are not — they are over
+ * the whole transcript. A caller that assumed the prompt was the session had
+ * no way to tell, so this says it.
+ */
+ transcript_window: TranscriptWindow;
guard_advisory: GuardAdvisory | null;
/**
* A named reason when a policy file exists but could not be used (T-1110).
@@ -158,6 +165,7 @@ interface PreparedValues {
policy_identity_hash: string;
source_hashes: { transcript: string; diff: string };
prompt: string;
+ transcript_window: TranscriptWindow;
guard_advisory: GuardAdvisory | null;
policy_error: string | null;
}
@@ -248,13 +256,16 @@ const prepareValues = (opts: {
: { trustedSignerFingerprints: opts.trustedSignerFingerprints }),
});
+ const harvest = buildHarvestPromptWithWindow({ transcript, diff });
+
return {
base_head: baseHead,
staged_diff_hash: stagedDiffHash,
staged_tree_oid: stagedTreeOid,
policy_identity_hash: policy.identityHash,
source_hashes: sourceHashes,
- prompt: buildHarvestPrompt({ transcript, diff }),
+ prompt: harvest.prompt,
+ transcript_window: harvest.window,
guard_advisory: advisory,
policy_error: policy.error,
};
@@ -297,6 +308,7 @@ export const prepareCaptureContext = (opts: PrepareCaptureOptions): PrepareResul
policy_identity_hash: prepared.policy_identity_hash,
source_hashes: prepared.source_hashes,
prompt: prepared.prompt,
+ transcript_window: prepared.transcript_window,
policy_error: prepared.policy_error,
guard_advisory: prepared.guard_advisory,
};
@@ -335,6 +347,7 @@ export const prepareCaptureContextReadOnly = (opts: PrepareCaptureOptions & {
policy_identity_hash: prepared.policy_identity_hash,
source_hashes: prepared.source_hashes,
prompt: prepared.prompt,
+ transcript_window: prepared.transcript_window,
policy_error: prepared.policy_error,
guard_advisory: prepared.guard_advisory,
pending,
diff --git a/src/core/harvest.ts b/src/core/harvest.ts
index dbc51c1a..b07552d6 100644
--- a/src/core/harvest.ts
+++ b/src/core/harvest.ts
@@ -332,12 +332,156 @@ const vocabularyBlock = (entries: VocabularyEntry[]): string[] => {
return lines;
};
-/** Line-numbered so a citation can name a range the verifier can find again. */
-const numberLines = (text: string): string => {
+/**
+ * Line-numbered so a citation can name a range the verifier can find again.
+ *
+ * `firstLine` is the number the first line of `text` carries in the whole
+ * transcript, which is 1 unless the prompt is showing a window of it. The
+ * numbers have to be the transcript's own: verification reads the whole
+ * transcript, and a locator renumbered to a window would name a different line
+ * of the file it is checked against (#873).
+ */
+const numberLines = (text: string, firstLine = 1): string => {
const lines = text.split('\n');
if (lines.length > 1 && lines[lines.length - 1] === '') lines.pop();
- const width = String(lines.length).length;
- return lines.map((line, index) => `${String(index + 1).padStart(width)} | ${line}`).join('\n');
+ const width = String(firstLine + lines.length - 1).length;
+ return lines
+ .map((line, index) => `${String(firstLine + index).padStart(width)} | ${line}`)
+ .join('\n');
+};
+
+/**
+ * How much transcript the prompt may carry.
+ *
+ * #873: the prompt embedded the transcript whole, so a 67,981,436-byte session
+ * produced a 67,468,122-byte prompt — larger than any model can read, which
+ * makes the capture pipeline unusable exactly on the long sessions that have
+ * the most to record. The reporter measured the pipeline working again at
+ * roughly half a megabyte and said, correctly, that they had not measured where
+ * the useful boundary is. Neither has this: 256 KiB is a bound chosen to be
+ * comfortably readable by current models while still carrying far more than the
+ * ~537,250-byte slice that was shown to work, not a claim about how much
+ * context a good record needs.
+ *
+ * It is the last bytes rather than the first: a decision is taken near the end
+ * of the session that implements it, and the diff being captured is that end.
+ */
+const DEFAULT_TRANSCRIPT_BUDGET_BYTES = 256 * 1024;
+
+/** An override, for an operator whose model reads more, or less, than this. */
+const transcriptBudgetBytes = (): number => {
+ const raw = Number(process.env['COMMITLORE_TRANSCRIPT_BUDGET_BYTES']);
+ return Number.isFinite(raw) && raw > 0 ? Math.floor(raw) : DEFAULT_TRANSCRIPT_BUDGET_BYTES;
+};
+
+/**
+ * What the prompt actually showed, so a caller is never left to assume it was
+ * the session. Nothing here changes what verification reads — the whole
+ * transcript is still hashed and still searched for every quote.
+ */
+export interface TranscriptWindow {
+ /** 1-based number, in the whole transcript, of the window's first line. */
+ first_line: number;
+ last_line: number;
+ total_lines: number;
+ /** Bytes of the whole transcript, so the caller can see the size it did not get. */
+ total_bytes: number;
+ /** Bytes of the window as the prompt carries it. */
+ window_bytes: number;
+ /** False when the whole transcript fitted and the window is the session. */
+ truncated: boolean;
+ /**
+ * True when `first_line` is shown from its middle. One JSONL line can hold a
+ * whole tool result and outrun the budget by itself, and a window of no lines
+ * at all would be worse than a window of one partial one.
+ */
+ first_line_partial: boolean;
+}
+
+/** Byte length as the prompt will carry it, newline included. */
+const lineBytes = (line: string): number => Buffer.byteLength(line, 'utf8') + 1;
+
+/**
+ * The last `budget` bytes of one line, without a split codepoint at the front.
+ * Slicing a UTF-8 buffer mid-character yields U+FFFD, and a replacement
+ * character inside a quotable line is a character nobody can copy back.
+ */
+const tailBytes = (line: string, budget: number): string =>
+ Buffer.from(line, 'utf8').subarray(-budget).toString('utf8').replace(/^\uFFFD+/, '');
+
+/**
+ * The tail of `transcript` that fits in `budget` bytes, in whole lines where
+ * whole lines fit.
+ */
+export const windowTranscript = (
+ transcript: string,
+ budget: number = transcriptBudgetBytes(),
+): { text: string; window: TranscriptWindow } => {
+ const lines = transcript.split('\n');
+ if (lines.length > 1 && lines[lines.length - 1] === '') lines.pop();
+ const totalLines = lines.length;
+ const totalBytes = Buffer.byteLength(transcript, 'utf8');
+
+ let kept = 0;
+ let bytes = 0;
+ for (let index = totalLines - 1; index >= 0; index -= 1) {
+ const next = bytes + lineBytes(lines[index]!);
+ if (next > budget && kept > 0) break;
+ if (next > budget) break; // not even the last line fits whole
+ bytes = next;
+ kept += 1;
+ }
+
+ if (kept === 0) {
+ // One line longer than the whole budget. Showing its tail keeps the window
+ // anchored where the decision is, and the line number stays the file's.
+ const last = lines[totalLines - 1] ?? '';
+ const text = tailBytes(last, budget);
+ return {
+ text,
+ window: {
+ first_line: totalLines,
+ last_line: totalLines,
+ total_lines: totalLines,
+ total_bytes: totalBytes,
+ window_bytes: Buffer.byteLength(text, 'utf8'),
+ truncated: true,
+ first_line_partial: true,
+ },
+ };
+ }
+
+ const firstLine = totalLines - kept + 1;
+ const text = lines.slice(firstLine - 1).join('\n');
+ return {
+ text,
+ window: {
+ first_line: firstLine,
+ last_line: totalLines,
+ total_lines: totalLines,
+ total_bytes: totalBytes,
+ window_bytes: Buffer.byteLength(text, 'utf8'),
+ truncated: firstLine > 1,
+ first_line_partial: false,
+ },
+ };
+};
+
+/**
+ * The line the prompt carries when it is showing a window, so the reader is
+ * told rather than left to infer it from a transcript that starts mid-sentence.
+ */
+const windowNotice = (window: TranscriptWindow): string[] => {
+ if (!window.truncated) return [];
+ const omitted = window.first_line - 1;
+ return [
+ `(This is the end of the transcript: lines ${window.first_line}-${window.last_line} of ` +
+ `${window.total_lines}, ${omitted} earlier line(s) omitted to bound this prompt` +
+ `${window.first_line_partial ? `, and line ${window.first_line} is shown from its middle` : ''}. ` +
+ 'The numbers below are the transcript\'s own, so a locator you write still names ' +
+ 'the line in the whole file. Cite only what you can see here.)',
+ '',
+ ];
};
const outputBlock = (entries: VocabularyEntry[]): string[] => {
@@ -414,15 +558,19 @@ export const buildHarvestContract = (): string => {
};
/**
- * Builds the prompt contract handed to the user's agent session. Deterministic
- * by construction — no clock, no randomness, no model — so the same transcript
- * and diff always produce the same bytes.
+ * Builds the prompt contract handed to the user's agent session, and says what
+ * of the transcript it carries. Deterministic by construction — no clock, no
+ * randomness, no model — so the same transcript, diff and budget always produce
+ * the same bytes.
*/
-export const buildHarvestPrompt = (input: HarvestInput): string => {
+export const buildHarvestPromptWithWindow = (
+ input: HarvestInput,
+): { prompt: string; window: TranscriptWindow } => {
const entries = loadVocabulary().filter((entry) => entry.key !== 'Verified');
const diff = input.diff.trim() === '' ? '(no diff)' : input.diff.replace(/\n+$/, '');
+ const { text, window } = windowTranscript(input.transcript);
- return [
+ const prompt = [
'# CommitLore harvest',
'',
'You are recording the decision context for a change that is about to be',
@@ -443,15 +591,22 @@ export const buildHarvestPrompt = (input: HarvestInput): string => {
'',
'## TRANSCRIPT',
'',
- numberLines(input.transcript),
+ ...windowNotice(window),
+ numberLines(text, window.first_line),
'',
'## DIFF',
'',
diff,
'',
].join('\n');
+
+ return { prompt, window };
};
+/** The prompt alone, for the callers that only emit it. */
+export const buildHarvestPrompt = (input: HarvestInput): string =>
+ buildHarvestPromptWithWindow(input).prompt;
+
const RECORD_FIELDS = ['trailers', 'evidence'];
const EVIDENCE_FIELDS = ['key', 'source', 'quote', 'locator'];
const EVIDENCE_SOURCES: readonly EvidenceSource[] = ['transcript', 'diff'];
diff --git a/src/mcp/server.ts b/src/mcp/server.ts
index 5f2767f9..92292b34 100644
--- a/src/mcp/server.ts
+++ b/src/mcp/server.ts
@@ -379,7 +379,10 @@ const TOOLS: readonly Tool[] = [
description:
'Prepare a capture transaction: computes binding conditions (HEAD, staged diff, tree, ' +
'policy hash), generates the prompt contract for the agent to use, and persists a ' +
- 'phase:"prepared" pending transaction. Returns the nonce needed for verify and stage.',
+ 'phase:"prepared" pending transaction. Returns the nonce needed for verify and stage. ' +
+ 'The prompt carries the end of the transcript rather than all of it; transcript_window ' +
+ 'says which lines, numbered as the whole transcript numbers them. Verification still ' +
+ 'reads the whole transcript, so quote only what the prompt shows you.',
inputSchema: {
type: 'object',
properties: {
@@ -638,6 +641,11 @@ export const createServer = (opts: McpServerOptions = {}): Server => {
policy_identity_hash: result.policy_identity_hash,
source_hashes: result.source_hashes,
prompt: result.prompt,
+ // What of the transcript that prompt carries (#873). It travels here
+ // for the same reason the two fields below do: MCP is the first-class
+ // surface for every agent but the plugin, and an agent handed a slice
+ // of its own session with no way to tell would cite the whole of it.
+ transcript_window: result.transcript_window,
// MCP is the first-class surface for every agent other than the Claude
// Code plugin, so both of these must travel here and not only to the
// pending file and the CLI. `guard_advisory` is always present, never
diff --git a/test/capture-prompt-budget.test.ts b/test/capture-prompt-budget.test.ts
new file mode 100644
index 00000000..87c69d50
--- /dev/null
+++ b/test/capture-prompt-budget.test.ts
@@ -0,0 +1,230 @@
+/**
+ * #873: `capture` embedded the transcript in the prompt whole, so the prompt
+ * was the session. Measured by the reporter: a 67,981,436-byte transcript
+ * produced a 67,468,122-byte prompt — 99.3% of the file, and larger than any
+ * model can read. The pipeline itself was fine; the reporter confirmed that by
+ * slicing the transcript to `tail -n 400` (537,250 bytes) and watching the same
+ * command work.
+ *
+ * That made the failure worse than a size: it is silent. Prompt-only mode
+ * returns `outcome: "empty"`, `staged: false`, **exit 0**, so an operator who
+ * tried capture once on a real session got a prompt they could not use and no
+ * statement that anything was wrong. In the repository where it was measured,
+ * `stale` reported 1 record in 1000 commits with unattended capture already
+ * enabled — permission was never the obstacle.
+ *
+ * Three properties are load-bearing here, and each has a test below that fails
+ * when its half of the fix is reverted:
+ *
+ * 1. The prompt is bounded, whatever the transcript weighs.
+ * 2. The line numbers in that window are the *whole transcript's* numbers.
+ * Verification reads the whole transcript, so a window renumbered from 1
+ * would have every locator name a different line of the file it is checked
+ * against.
+ * 3. The caller is told it received a window. The old prompt was the session,
+ * so nothing downstream had any reason to ask; a bounded prompt that does
+ * not say so is the same silence in a smaller package.
+ */
+
+import { execSync } from 'node:child_process';
+import { mkdtempSync, rmSync, writeFileSync } from 'node:fs';
+import { tmpdir } from 'node:os';
+import { join } from 'node:path';
+
+import { afterAll, afterEach, describe, expect, it } from 'vitest';
+
+import { prepareCaptureContext } from '../src/core/capture-prepare.js';
+import { verifyCaptureRecords } from '../src/core/capture-verify.js';
+import { buildHarvestPromptWithWindow, windowTranscript, type DraftRecord } from '../src/core/harvest.js';
+
+const scratch: string[] = [];
+afterAll(() => {
+ for (const dir of scratch) rmSync(dir, { recursive: true, force: true });
+});
+
+const makeRepo = (): string => {
+ const dir = mkdtempSync(join(tmpdir(), 'capture-prompt-budget-'));
+ scratch.push(dir);
+ execSync('git init --quiet --initial-branch=main', { cwd: dir });
+ execSync('git config user.name "Test"', { cwd: dir });
+ execSync('git config user.email "test@test.com"', { cwd: dir });
+ execSync('git config commit.gpgsign false', { cwd: dir });
+ writeFileSync(join(dir, 'a.txt'), 'hello\n');
+ execSync('git add a.txt', { cwd: dir });
+ execSync('git commit -m "init" --quiet', { cwd: dir });
+ writeFileSync(join(dir, 'a.txt'), 'hello\nworld\n');
+ execSync('git add a.txt', { cwd: dir });
+ return dir;
+};
+
+/** A transcript of `lines` numbered lines, each recognisable by its own number. */
+const transcriptOf = (lines: number, padTo = 0): string =>
+ Array.from({ length: lines }, (_, index) => {
+ const body = `line ${index + 1} of the session`;
+ return padTo > body.length ? body + ' '.repeat(padTo - body.length) : body;
+ }).join('\n');
+
+const DIFF = 'diff --git a/x b/x\n--- a/x\n+++ b/x\n@@ -1 +1 @@\n-old\n+new\n';
+
+describe('#873 windowTranscript bounds what the prompt can carry', () => {
+ it('is the whole transcript, unmarked, when the whole transcript fits', () => {
+ const transcript = transcriptOf(20);
+ const { text, window } = windowTranscript(transcript, 64 * 1024);
+
+ expect(text).toBe(transcript);
+ expect(window.truncated).toBe(false);
+ expect(window.first_line).toBe(1);
+ expect(window.last_line).toBe(20);
+ expect(window.total_lines).toBe(20);
+ expect(window.first_line_partial).toBe(false);
+ });
+
+ it('keeps the end of a transcript that does not fit, within the budget', () => {
+ const transcript = transcriptOf(5000);
+ const { text, window } = windowTranscript(transcript, 4096);
+
+ expect(Buffer.byteLength(text, 'utf8')).toBeLessThanOrEqual(4096);
+ expect(window.truncated).toBe(true);
+ expect(window.last_line).toBe(5000);
+ expect(window.total_lines).toBe(5000);
+ expect(window.first_line).toBeGreaterThan(1);
+ // The end, not the beginning: the decision is taken where the session ends.
+ expect(text.endsWith('line 5000 of the session')).toBe(true);
+ expect(text.startsWith(`line ${window.first_line} of the session`)).toBe(true);
+ expect(window.total_bytes).toBe(Buffer.byteLength(transcript, 'utf8'));
+ expect(window.window_bytes).toBe(Buffer.byteLength(text, 'utf8'));
+ });
+
+ it('shows the tail of one line that outruns the budget by itself', () => {
+ // A single JSONL line can hold an entire tool result. A window of no lines
+ // at all would be worse than a window of one partial line.
+ const transcript = `first\n${'x'.repeat(20_000)}`;
+ const { text, window } = windowTranscript(transcript, 1024);
+
+ expect(Buffer.byteLength(text, 'utf8')).toBeLessThanOrEqual(1024);
+ expect(window.first_line).toBe(2);
+ expect(window.last_line).toBe(2);
+ expect(window.first_line_partial).toBe(true);
+ expect(window.truncated).toBe(true);
+ });
+
+ it('never leaves a split codepoint at the front of a partial line', () => {
+ // Slicing a UTF-8 buffer mid-character yields U+FFFD, and a replacement
+ // character inside a quotable line is a character nobody can copy back.
+ const transcript = '한'.repeat(4000);
+ const { text } = windowTranscript(transcript, 1000);
+
+ expect(text.startsWith('�')).toBe(false);
+ expect(Buffer.byteLength(text, 'utf8')).toBeLessThanOrEqual(1000);
+ });
+
+ it('honours COMMITLORE_TRANSCRIPT_BUDGET_BYTES', () => {
+ const previous = process.env['COMMITLORE_TRANSCRIPT_BUDGET_BYTES'];
+ process.env['COMMITLORE_TRANSCRIPT_BUDGET_BYTES'] = '2048';
+ try {
+ const { window } = windowTranscript(transcriptOf(5000));
+ expect(window.window_bytes).toBeLessThanOrEqual(2048);
+ expect(window.truncated).toBe(true);
+ } finally {
+ if (previous === undefined) delete process.env['COMMITLORE_TRANSCRIPT_BUDGET_BYTES'];
+ else process.env['COMMITLORE_TRANSCRIPT_BUDGET_BYTES'] = previous;
+ }
+ });
+});
+
+describe('#873 the prompt numbers the window as the transcript numbers it', () => {
+ const budgeted = (lines: number): ReturnType => {
+ const previous = process.env['COMMITLORE_TRANSCRIPT_BUDGET_BYTES'];
+ process.env['COMMITLORE_TRANSCRIPT_BUDGET_BYTES'] = '4096';
+ try {
+ return buildHarvestPromptWithWindow({ transcript: transcriptOf(lines), diff: DIFF });
+ } finally {
+ if (previous === undefined) delete process.env['COMMITLORE_TRANSCRIPT_BUDGET_BYTES'];
+ else process.env['COMMITLORE_TRANSCRIPT_BUDGET_BYTES'] = previous;
+ }
+ };
+
+ it('gives the window\'s first line its number in the whole transcript', () => {
+ const { prompt, window } = budgeted(5000);
+
+ // The exact property a renumbered window would break: the line the prompt
+ // labels N really is line N of the transcript verification will read.
+ expect(prompt).toContain(`${window.first_line} | line ${window.first_line} of the session`);
+ expect(prompt).toContain(`${window.last_line} | line ${window.last_line} of the session`);
+ expect(prompt).not.toContain('1 | line 1 of the session');
+ });
+
+ it('says it is a window, and how much was left out', () => {
+ const { prompt, window } = budgeted(5000);
+
+ expect(prompt).toContain(`lines ${window.first_line}-${window.last_line} of ${window.total_lines}`);
+ expect(prompt).toContain('earlier line(s) omitted to bound this prompt');
+ });
+
+ it('says nothing about a window when there is no window', () => {
+ const { prompt } = buildHarvestPromptWithWindow({ transcript: transcriptOf(5), diff: DIFF });
+
+ expect(prompt).not.toContain('omitted to bound this prompt');
+ expect(prompt).toContain('1 | line 1 of the session');
+ });
+});
+
+describe('#873 capture returns a prompt a model can read, and says what it is', () => {
+ const previous = process.env['COMMITLORE_TRANSCRIPT_BUDGET_BYTES'];
+ afterEach(() => {
+ if (previous === undefined) delete process.env['COMMITLORE_TRANSCRIPT_BUDGET_BYTES'];
+ else process.env['COMMITLORE_TRANSCRIPT_BUDGET_BYTES'] = previous;
+ });
+
+ it('does not return a prompt the size of the transcript', () => {
+ // The reporter's ratio, at a size a test can afford: the prompt used to be
+ // ~99.3% of the transcript, so it grew without bound with the session.
+ const cwd = makeRepo();
+ process.env['COMMITLORE_TRANSCRIPT_BUDGET_BYTES'] = '8192';
+ const transcript = transcriptOf(40_000, 200);
+
+ const result = prepareCaptureContext({ cwd, transcript });
+
+ expect(Buffer.byteLength(transcript, 'utf8')).toBeGreaterThan(4_000_000);
+ expect(Buffer.byteLength(result.prompt, 'utf8')).toBeLessThan(64 * 1024);
+ expect(result.transcript_window.truncated).toBe(true);
+ expect(result.transcript_window.total_lines).toBe(40_000);
+ expect(result.transcript_window.last_line).toBe(40_000);
+ });
+
+ it('hashes and verifies the whole transcript, not the window', () => {
+ // The bound is on the prompt alone. If it ever reached the hash, `verify`
+ // would be checking quotes against a slice — and a caller passing the
+ // session it actually had would be told the transcript was substituted.
+ const cwd = makeRepo();
+ process.env['COMMITLORE_TRANSCRIPT_BUDGET_BYTES'] = '4096';
+ const quote = 'we ruled out the queue worker because it loses ordering';
+ const transcript = `${quote}\n${transcriptOf(5000)}`;
+
+ const prepared = prepareCaptureContext({ cwd, transcript });
+ expect(prepared.transcript_window.first_line).toBeGreaterThan(1);
+
+ const draft: DraftRecord[] = [
+ {
+ trailers: [
+ { key: 'Ruled-out', value: 'queue worker | loses ordering' },
+ { key: 'Record-Id', value: 'r-window1' },
+ ],
+ evidence: [{ key: 'Ruled-out', source: 'transcript', quote, locator: 'L1-L1' }],
+ },
+ ];
+
+ const verified = verifyCaptureRecords({
+ nonce: prepared.nonce,
+ draft,
+ transcript,
+ diff: execSync('git diff --cached', { cwd, encoding: 'utf8' }),
+ cwd,
+ });
+
+ // The quote is on line 1, which the window does not show — and it verifies,
+ // because verification never looked at the window.
+ expect(verified.rejected).toHaveLength(0);
+ expect(verified.accepted).toHaveLength(1);
+ });
+});