Merge branch 'master' into nih-imp-gates

This commit is contained in:
Tianyi Cui
2026-07-27 16:30:44 +08:00
committed by GitHub
457 changed files with 22600 additions and 3395 deletions

View File

@@ -40,6 +40,8 @@ export const LINK_MAP: Record<string, string> = {
HookContext: 'core.md',
LlmCallConfig: 'core.md',
LlmModelContext: 'core.md',
LlmModelReasoningInfo: 'core.md',
LlmResolvedModelInfo: 'core.md',
LlmFailure: 'llm-streaming.md',
LlmModelInfo: 'core.md',
LlmProviderInfo: 'core.md',
@@ -92,6 +94,7 @@ export const LINK_MAP: Record<string, string> = {
CommandResult: 'commands.md',
CommandSurface: 'commands.md',
LlmAdapter: 'llm-streaming.md',
PreparedLlmCall: 'llm-streaming.md',
LlmService: 'llm-streaming.md',
StreamChunk: 'llm-streaming.md',
CreateSessionOptions: 'persistence.md',
@@ -207,6 +210,10 @@ const FOUNDATION_TYPE_NAMES = new Set([
/** Project types deliberately documented outside the core-data catalog. */
const TYPE_LINK_EXEMPTIONS: Readonly<Record<string, string>> = {
AgentFactory: 'agent creation seam is owned by packages/core/agent/README.md',
BeginCommandRequest: 'event-local request contract is owned by packages/client/ui-slash/src/types.ts',
InsertReferenceRequest: 'event-local request contract is owned by packages/client/ui-slash/src/types.ts',
ConsumeTokenRequest: 'event-local request contract is owned by packages/client/ui-slash/src/types.ts',
InsertTextRequest: 'event-local request contract is owned by packages/client/ui-slash/src/types.ts',
AgentHandle: 'agent ownership handle is owned by packages/core/agent/README.md',
BashEnvContributor: 'service-local extension type is owned by packages/bash/tool-bash/src/index.ts',
BashEnvVariableInfo: 'service-local metadata type is owned by packages/bash/tool-bash/src/index.ts',
@@ -389,7 +396,7 @@ export function collectEvents(scanRoot: string = root): EventEntry[] {
const where = `event '${name}' (${src})`
checkTypeLinks(where, member, sf, typeLinkViolations)
if (!mode) {
violations.push(`${where} is missing an @mode tag. Add '@mode emit|waterfall|parallel|serial' to its JSDoc (see AGENTS.md).`)
violations.push(`${where} is missing an @mode tag. Add '@mode emit|waterfall|parallel|serial|bail' to its JSDoc (see AGENTS.md).`)
}
// Conclusive structural check: a trailing `next: () => …` parameter is a
// waterfall. (emit vs parallel vs serial is not structurally
@@ -580,7 +587,7 @@ export function renderEvents(events: EventEntry[]): string {
'',
'The **harness tier** below (the `@deepseek-ai/dsh-*` packages) is the vocabulary this repo owns, grouped by scope. The **inherited tier** at the end is the cordis-core + loader/hmr/timer event surface a plugin also sees — pinned vendor source, summarized tersely. The event-dispatch methods themselves are generated in the [Cordis core Events API](core/events.md).',
'',
'Dispatch modes: **emit** (fire-and-forget), **waterfall** (each listener gets `next()` and may transform or veto — see [waterfall semantics](../cordis-primer.md#cordis-waterfall-semantics)), **parallel** (awaited fan-out; all listeners run), **serial** (awaited in registration order until one returns a bail value — anything other than `null`, `false`, or `undefined`).',
'Dispatch modes: **emit** (fire-and-forget), **waterfall** (each listener gets `next()` and may transform or veto — see [waterfall semantics](../cordis-primer.md#cordis-waterfall-semantics)), **parallel** (awaited fan-out; all listeners run), **serial** (awaited in registration order until one returns a bail value — anything other than `null`, `false`, or `undefined`), **bail** (synchronous in-order dispatch until one listener returns a bail value; the scoped input-mutation events use it for an applied/not-applied answer).',
'',
]
const scopes = [...new Set(events.map(e => e.scope))].sort()

View File

@@ -917,8 +917,13 @@ function renderEventRelations(pkgs: Pkg[]): string {
lines.push(`| \`${event.name}\` | \`${event.mode}\` | ${sourceLink(event.source)} | ${relationPackages(relation.dispatchers, pkgsByShort)} | ${listenerPackages(relation.listeners, pkgsByShort)} |`)
}
// Every declared event needs a dispatcher: zero means dead vocabulary or an
// unrecognized semantic dispatch shape. Listener-free extension points remain valid.
// unrecognized semantic dispatch shape. Listener-free extension points remain
// valid. Client-declared events are exempt: the relation scan seeds the HOST
// aggregate program only (host+client cannot share one program — the cordis
// Context merges collide), so client dispatch sites are structurally
// invisible here; their rows stay in the table for the declarations' sake.
const undispatched = [...events]
.filter(event => !event.source.startsWith('packages/client/'))
.filter(event => (relations.get(event.name)?.dispatchers.size ?? 0) === 0)
.map(event => event.name)
.sort()

View File

@@ -0,0 +1,302 @@
/**
* Print the minimal-update briefing for out-of-sync translation pairs:
* `pnpm run gen-translation-brief [--apply] [pair paths...]`. With no
* arguments it discovers every out-of-sync pair; with arguments (any file
* of a pair) it briefs exactly those pairs and fails loud on in-sync,
* incomplete, or out-of-scope requests. Each briefing maps the change at
* the narrowest safe granularity — code-fence-only splice, changed
* Markdown units, heading sections, whole document — and `--apply` writes
* the computed counterpart for pairs whose change is code-fence-only.
* The briefing contract lives in `scripts/translation-brief.ts`; the
* consuming workflow is `.agents/skills/dsh-translate-docs/SKILL.md`.
*/
import { spawnSync } from 'node:child_process'
import { existsSync, globSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs'
import { tmpdir } from 'node:os'
import { basename, join, resolve, sep } from 'node:path'
import {
isTranslationScopeFile,
pairAnchorOfArgument,
parseTranslationMarkdown,
parseTranslationPairingManifest,
TRANSLATION_SCOPE_GLOB_EXCLUDES,
translationStructureDiff,
translationStructureSignature,
} from './translation-pairing.ts'
import {
changedSpanIndices,
computeMechanicalUpdate,
firstOccurrenceContext,
markdownUnits,
relevantTerminologyRows,
renderTranslationBrief,
sectionSpans,
spansAligned,
type BriefBundle,
type BriefDirection,
type BriefScope,
type MarkdownSpan,
} from './translation-brief.ts'
const root = resolve(import.meta.dirname, '..')
const manifest = parseTranslationPairingManifest(readFileSync(join(root, 'scripts/translation-pairing.manifest.json'), 'utf8'))
const terminology = readFileSync(join(root, 'docs/i18n/terminology.md'), 'utf8')
function isExcluded(file: string): boolean {
return manifest.excluded.some(entry => (entry.endsWith('/') ? file.startsWith(entry) : file === entry))
}
/** Recorded hashes of one consistency record: basename → blob hash. */
function parseMeta(content: string): Map<string, string> | undefined {
const out = new Map<string, string>()
for (const line of content.split('\n')) {
if (line === '' || line.startsWith('#')) continue
const match = /^([^:#]+\.md): ([0-9a-f]{40})$/.exec(line)
if (!match?.[1] || !match[2]) return undefined
out.set(match[1], match[2])
}
return out
}
function git(args: string[], allowedExitCodes: number[] = [0]): string {
const result = spawnSync('git', ['-C', root, ...args], { encoding: 'utf8', maxBuffer: 1 << 26 })
if (result.error) throw result.error
if (!allowedExitCodes.includes(result.status ?? -1)) {
throw new Error(`git ${args.join(' ')} failed: ${result.stderr}`)
}
return result.stdout
}
function blobText(hash: string): string {
return git(['cat-file', '-p', hash])
}
/** Unified diff between two texts, headers stripped, via `git diff --no-index`. */
function diffTexts(before: string, after: string): string {
const dir = mkdtempSync(join(tmpdir(), 'translation-brief-'))
try {
writeFileSync(join(dir, 'last-confirmed.md'), before)
writeFileSync(join(dir, 'current.md'), after)
const raw = git(['diff', '--no-index', '--unified=2', join(dir, 'last-confirmed.md'), join(dir, 'current.md')], [0, 1])
return raw.split('\n')
.filter(line => !line.startsWith('diff --git') && !line.startsWith('index ') && !line.startsWith('--- ') && !line.startsWith('+++ '))
.join('\n')
.trim()
} finally {
rmSync(dir, { recursive: true, force: true })
}
}
interface PairState {
anchor: string
zh: string
meta: string
enDrifted: boolean
zhDrifted: boolean
enLast: string
zhLast: string
}
/** Load one pair's recorded and current state, or explain why it cannot be briefed. */
function loadPair(anchor: string): PairState | string {
const zh = anchor.replace(/\.md$/, '.zh.md')
const meta = anchor.replace(/\.md$/, '.i18n.yaml')
if (!isTranslationScopeFile(anchor) || isExcluded(anchor)) {
return `${anchor}: not an in-scope documentation pair (docs/i18n/README.md)`
}
const missing = [anchor, zh, meta].filter(file => !existsSync(join(root, file)))
if (missing.length > 0) {
return `${anchor}: incomplete pair (missing ${missing.join(', ')}) — a new counterpart is whole-document translation work, not a minimal update`
}
const record = parseMeta(readFileSync(join(root, meta), 'utf8'))
const enRecorded = record?.get(basename(anchor))
const zhRecorded = record?.get(basename(zh))
if (record === undefined || enRecorded === undefined || zhRecorded === undefined) {
return `${meta}: malformed consistency record`
}
const enCurrent = readFileSync(join(root, anchor), 'utf8')
const zhCurrent = readFileSync(join(root, zh), 'utf8')
const enLast = blobText(enRecorded)
const zhLast = blobText(zhRecorded)
return {
anchor,
zh,
meta,
enDrifted: enCurrent !== enLast,
zhDrifted: zhCurrent !== zhLast,
enLast,
zhLast,
}
}
/** Assemble bundles for the given changed + first-occurrence span indices. */
function bundlesFor(
indices: number[],
extraIndices: number[],
confirmed: MarkdownSpan[],
current: MarkdownSpan[],
counterpart: MarkdownSpan[],
): BriefBundle[] {
const extras = new Set(extraIndices)
return [...new Set([...indices, ...extraIndices])].sort((left, right) => left - right).map((index) => {
const confirmedSpan = confirmed[index]
const currentSpan = current[index]
const counterpartSpan = counterpart[index]
if (confirmedSpan === undefined || currentSpan === undefined || counterpartSpan === undefined) {
throw new Error(`gen-translation-brief: span ${index} is unmapped despite alignment`)
}
return {
index,
label: currentSpan.label,
reason: extras.has(index) && confirmedSpan.text === currentSpan.text ? 'first-occurrence' as const : undefined,
confirmedSourceText: confirmedSpan.text,
currentSourceText: currentSpan.text,
counterpartText: counterpartSpan.text,
counterpartStartLine: counterpartSpan.startLine,
}
})
}
interface PlannedBrief {
scope: BriefScope
/** Old + new text of the changed spans, for terminology matching. */
changedText: string
/** Computed counterpart for a mechanical scope, for `--apply`. */
mechanicalResult?: string | undefined
}
/** Choose the narrowest safely mapped granularity for one drifted side. */
function planScope(
sourceLast: string,
sourceCurrent: string,
counterpartCurrent: string,
direction: BriefDirection,
bothDrifted: boolean,
): PlannedBrief {
const wholeChangedText = `${sourceLast}\n${sourceCurrent}`
if (bothDrifted) {
return {
scope: { kind: 'document', reason: 'BOTH sides changed since the pair was last confirmed consistent, so no side is a trustworthy mapping anchor; decide which side owns each divergence.' },
changedText: wholeChangedText,
}
}
const mechanical = computeMechanicalUpdate(sourceLast, sourceCurrent, counterpartCurrent)
if (mechanical !== undefined) {
return { scope: { kind: 'mechanical' }, changedText: wholeChangedText, mechanicalResult: mechanical }
}
for (const [kind, spansOf] of [['units', markdownUnits], ['sections', sectionSpans]] as const) {
const confirmed = spansOf(sourceLast)
const current = spansOf(sourceCurrent)
const counterpart = spansOf(counterpartCurrent)
if (!spansAligned(confirmed, current) || !spansAligned(confirmed, counterpart)) continue
const changed = changedSpanIndices(confirmed, current)
if (changed.length === 0) continue
const changedText = changed.map(index => `${confirmed[index]?.text ?? ''}\n${current[index]?.text ?? ''}`).join('\n')
const rows = relevantTerminologyRows(terminology, direction, changedText)
const occurrence = direction === 'en-to-zh'
? firstOccurrenceContext(sourceLast, sourceCurrent, confirmed, current, rows, new Set(changed))
: { notes: [], extraSpanIndices: [] }
return {
scope: {
kind,
bundles: bundlesFor(changed, occurrence.extraSpanIndices, confirmed, current, counterpart),
firstOccurrenceNotes: occurrence.notes,
},
changedText,
}
}
return {
scope: { kind: 'document', reason: 'Neither fine-grained units nor heading sections align one to one across the last-confirmed source, current source, and current counterpart.' },
changedText: wholeChangedText,
}
}
/** Validate a computed mechanical counterpart and write it. */
function applyMechanical(counterpartPath: string, sourceCurrent: string, result: string): void {
const counterpartBase = basename(counterpartPath)
const sourceBase = counterpartBase.endsWith('.zh.md')
? counterpartBase.replace(/\.zh\.md$/, '.md')
: counterpartBase.replace(/\.md$/, '.zh.md')
const errors = translationStructureDiff(
translationStructureSignature(parseTranslationMarkdown(sourceCurrent), counterpartBase),
translationStructureSignature(parseTranslationMarkdown(result), sourceBase),
)
if (errors.length > 0) {
throw new Error(`gen-translation-brief: computed mechanical update for ${counterpartPath} violates the pair structure: ${errors.join('; ')}`)
}
writeFileSync(join(root, counterpartPath), result)
console.error(`gen-translation-brief: applied code-fence splice to ${counterpartPath}; review the diff, then record the pair.`)
}
/** Render (and under `--apply`, apply) the briefing for one drifted side. */
function briefDirection(pair: PairState, direction: BriefDirection, apply: boolean): string {
const sourceIsEnglish = direction === 'en-to-zh'
const sourcePath = sourceIsEnglish ? pair.anchor : pair.zh
const counterpartPath = sourceIsEnglish ? pair.zh : pair.anchor
const sourceLast = sourceIsEnglish ? pair.enLast : pair.zhLast
const sourceCurrent = readFileSync(join(root, sourcePath), 'utf8')
const counterpartCurrent = readFileSync(join(root, counterpartPath), 'utf8')
const diff = diffTexts(sourceLast, sourceCurrent)
const planned = planScope(sourceLast, sourceCurrent, counterpartCurrent, direction, pair.enDrifted && pair.zhDrifted)
if (apply && planned.mechanicalResult !== undefined) {
applyMechanical(counterpartPath, sourceCurrent, planned.mechanicalResult)
}
return renderTranslationBrief({
sourcePath,
counterpartPath,
direction,
diff,
scope: planned.scope,
terminology: relevantTerminologyRows(terminology, direction, planned.changedText),
})
}
const argv = process.argv.slice(2)
const flags = argv.filter(argument => argument.startsWith('--'))
const unknownFlags = flags.filter(flag => flag !== '--apply')
if (unknownFlags.length > 0) {
console.error(`gen-translation-brief: unknown flag(s): ${unknownFlags.join(', ')} (only --apply is supported)`)
process.exit(2)
}
const applyMode = flags.includes('--apply')
const requested = argv.filter(argument => !argument.startsWith('--')).map(pairAnchorOfArgument)
let anchors: string[]
if (requested.length > 0) {
anchors = [...new Set(requested)].sort()
} else {
const discovered = new Set<string>()
for (const match of globSync('**/*.i18n.yaml', { cwd: root, exclude: TRANSLATION_SCOPE_GLOB_EXCLUDES })) {
const normalized = match.split(sep).join('/')
if (isTranslationScopeFile(normalized)) discovered.add(normalized.replace(/\.i18n\.yaml$/, '.md'))
}
anchors = [...discovered].sort()
}
const briefs: string[] = []
const problems: string[] = []
const skipped: string[] = []
for (const anchor of anchors) {
const pair = loadPair(anchor)
if (typeof pair === 'string') {
if (requested.length > 0) problems.push(pair)
continue
}
if (!pair.enDrifted && !pair.zhDrifted) {
if (requested.length > 0) skipped.push(`${anchor}: pair is consistent with its record — nothing to brief`)
continue
}
if (pair.enDrifted) briefs.push(briefDirection(pair, 'en-to-zh', applyMode))
if (pair.zhDrifted) briefs.push(briefDirection(pair, 'zh-to-en', applyMode))
}
if (problems.length > 0 || skipped.length > 0) {
for (const message of [...problems, ...skipped]) console.error(`gen-translation-brief: ${message}`)
process.exit(2)
}
if (briefs.length === 0) {
console.log('gen-translation-brief: every recorded pair matches its consistency record; nothing to brief.')
process.exit(0)
}
console.log(briefs.join('\n\n---\n\n'))

View File

@@ -19,7 +19,7 @@ export function rawJsDoc(text: string, node: ts.Node): string {
}
/** A dispatch mode, rendered as the badge after an event name in the catalog. */
export type Mode = 'emit' | 'waterfall' | 'parallel' | 'serial'
export type Mode = 'emit' | 'waterfall' | 'parallel' | 'serial' | 'bail'
/**
* Parse a raw JSDoc block into description prose and an optional `@mode`. Prose
@@ -59,7 +59,7 @@ export function parseJsDoc(raw: string): { doc: string; mode: Mode | null; hasMo
}
for (const line of inner) {
const tagLine = line.trimStart()
const m = /^@mode\s+(emit|waterfall|parallel|serial)\s*$/.exec(tagLine)
const m = /^@mode\s+(emit|waterfall|parallel|serial|bail)\s*$/.exec(tagLine)
if (m) { mode = m[1] as Mode; hasMode = true; flushPara(); inTags = true; continue }
if (/^@mode\b/.test(tagLine)) { hasMode = true; flushPara(); inTags = true; continue }
if (tagLine.startsWith('@')) { flushPara(); inTags = true; continue }

File diff suppressed because one or more lines are too long

View File

@@ -0,0 +1,283 @@
/** Regression tests for the minimal-update briefing assembly. */
import { describe, expect, it } from 'vitest'
import {
changedSpanIndices,
computeMechanicalUpdate,
firstOccurrenceContext,
markdownUnits,
parseTerminologyRows,
relevantTerminologyRows,
renderTranslationBrief,
sectionSpans,
spansAligned,
termOffsets,
} from './translation-brief.ts'
const DOC = [
'Preamble line.',
'',
'# Title',
'',
'Intro paragraph.',
'',
'## First',
'',
'First body.',
'',
'```ts',
'const value = 1',
'```',
'',
'## Second',
'',
'| A | B |',
'|---|---|',
'| 1 | 2 |',
'',
'- item one',
'- item two',
].join('\n')
describe('markdown spans', () => {
it('lists units with container-scoped kinds in document order', () => {
const kinds = markdownUnits(DOC).map(span => span.kind)
expect(kinds).toEqual([
'root.0:paragraph',
'root.1:heading:1',
'root.2:paragraph',
'root.3:heading:2',
'root.4:paragraph',
'root.5:code',
'root.6:heading:2',
'root.7.0:tableRow',
'root.7.1:tableRow',
'root.8.0:listItem',
'root.8.1:listItem',
])
})
it('lists heading sections with a preamble span and heading labels', () => {
const sections = sectionSpans(DOC)
expect(sections.map(span => span.label)).toEqual([
'(preamble before the first heading)',
'Title',
'First',
'Second',
])
expect(sections[0]).toMatchObject({ startLine: 1, endLine: 2 })
expect(sections[2]).toMatchObject({ startLine: 7, endLine: 14 })
})
it('labels units by their node type', () => {
const units = markdownUnits(DOC)
expect(units[0]!.label).toBe('paragraph')
expect(units[1]!.label).toBe('heading')
expect(units[7]!.label).toBe('tableRow')
})
it('aligns sections by depth only, so translated heading text still maps', () => {
const zh = DOC.replace('## First', '## 第一节').replace('## Second', '## 第二节').replace('# Title', '# 标题')
expect(spansAligned(sectionSpans(DOC), sectionSpans(zh))).toBe(true)
})
it('aligns span lists only on equal non-empty kind sequences', () => {
const zh = DOC.replace('First body.', '第一段。').replace('item one', '第一项').replace('Intro paragraph.', '导语。')
expect(spansAligned(markdownUnits(DOC), markdownUnits(zh))).toBe(true)
const reshaped = DOC.replace('- item one\n- item two', 'merged paragraph')
expect(spansAligned(markdownUnits(DOC), markdownUnits(reshaped))).toBe(false)
expect(spansAligned([], [])).toBe(false)
})
it('reports the indices whose text changed', () => {
const edited = DOC.replace('First body.', 'First body, revised.').replace('| 1 | 2 |', '| 1 | 3 |')
expect(changedSpanIndices(markdownUnits(DOC), markdownUnits(edited))).toEqual([4, 8])
})
})
describe('mechanical code updates', () => {
const en = '# T\n\nProse.\n\n```sh\nrun one\n```\n'
const zh = '# T\n\n中文。\n\n```sh\nrun one\n```\n'
it('splices a fence-only edit into the counterpart', () => {
const edited = en.replace('run one', 'run two')
expect(computeMechanicalUpdate(en, edited, zh)).toBe(zh.replace('run one', 'run two'))
})
it('refuses when prose changed too', () => {
const edited = en.replace('Prose.', 'Prose!').replace('run one', 'run two')
expect(computeMechanicalUpdate(en, edited, zh)).toBeUndefined()
})
it('refuses when the counterpart fences already diverge from last-confirmed', () => {
const edited = en.replace('run one', 'run two')
expect(computeMechanicalUpdate(en, edited, zh.replace('run one', 'run stale'))).toBeUndefined()
})
it('refuses when fence counts differ or nothing changed', () => {
expect(computeMechanicalUpdate(en, `${en}\n\`\`\`sh\nextra\n\`\`\`\n`, zh)).toBeUndefined()
expect(computeMechanicalUpdate(en, en, zh)).toBeUndefined()
})
})
const TERMINOLOGY = [
'| English | 中文 | 首次出现 | 不要译作 | 备注 |',
'|---|---|---|---|---|',
'| agent | agent | agent智能体 | 智能体 | |',
'| session log | 会话日志 | | 会话记录 | |',
'| gate | 门禁 | | | |',
'| registry | 注册表 | | | |',
].join('\n')
describe('terminology', () => {
it('parses data rows and skips the header and separator', () => {
const rows = parseTerminologyRows(TERMINOLOGY)
expect(rows.map(row => row.english)).toEqual(['agent', 'session log', 'gate', 'registry'])
expect(rows[0]).toMatchObject({ chinese: 'agent', first: 'agent智能体' })
})
it('matches English terms on word boundaries with plural inflections', () => {
expect(termOffsets('two agents met', 'agent', true)).toEqual([4])
expect(termOffsets('two registries', 'registry', true)).toEqual([4])
expect(termOffsets('reagents', 'agent', true)).toEqual([])
expect(termOffsets('', 'agent', true)).toEqual([])
})
it('selects rows for the changed text per direction', () => {
expect(relevantTerminologyRows(TERMINOLOGY, 'en-to-zh', 'All agents write a session log.').map(row => row.english))
.toEqual(['agent', 'session log'])
expect(relevantTerminologyRows(TERMINOLOGY, 'zh-to-en', '门禁在提交时运行。').map(row => row.english))
.toEqual(['gate'])
expect(relevantTerminologyRows(TERMINOLOGY, 'en-to-zh', 'delegate the work')).toEqual([])
})
})
describe('first-occurrence tracking', () => {
const before = '# T\n\nAlpha paragraph.\n\nThe agent runs.\n'
const after = '# T\n\nAlpha paragraph with an agent.\n\nThe agent runs.\n'
const rows = parseTerminologyRows(TERMINOLOGY).filter(row => row.english === 'agent')
it('flags a moved first occurrence and pulls the vacated span in', () => {
const context = firstOccurrenceContext(
before, after, markdownUnits(before), markdownUnits(after), rows, new Set([1]),
)
expect(context.notes).toHaveLength(1)
expect(context.notes[0]).toContain('moved from #2 to #1')
expect(context.extraSpanIndices).toEqual([2])
})
it('stays silent when the first occurrence does not move', () => {
const unmoved = before.replace('Alpha paragraph.', 'Alpha paragraph, revised.')
const context = firstOccurrenceContext(
before, unmoved, markdownUnits(before), markdownUnits(unmoved), rows, new Set([1]),
)
expect(context.notes).toEqual([])
expect(context.extraSpanIndices).toEqual([])
})
it('ignores rows without a first-occurrence rendering', () => {
const bare = parseTerminologyRows(TERMINOLOGY).filter(row => row.english === 'gate')
const withGate = after.replace('The agent runs.', 'The gate runs.')
const context = firstOccurrenceContext(
before, withGate, markdownUnits(before), markdownUnits(withGate), bare, new Set([2]),
)
expect(context.notes).toEqual([])
})
})
describe('brief rendering', () => {
const base = {
sourcePath: 'docs/foo.md',
counterpartPath: 'docs/foo.zh.md',
direction: 'en-to-zh' as const,
diff: '@@ -5 +5 @@\n-old text about the agent\n+new text about the agent',
terminology: relevantTerminologyRows(TERMINOLOGY, 'en-to-zh', 'the agent'),
}
const bundle = {
index: 4,
label: 'paragraph',
confirmedSourceText: 'old text about the agent\n',
currentSourceText: 'new text about the agent\n',
counterpartText: '关于 agent 的旧文本\n',
counterpartStartLine: 9,
}
it('renders unit bundles with three-way context and line anchors', () => {
const brief = renderTranslationBrief({
...base,
scope: { kind: 'units', bundles: [bundle], firstOccurrenceNotes: ['agent: the document-wide first occurrence moved from #2 to #1; the agent智能体 form moves with it (later occurrences drop the annotation).'] },
})
expect(brief).toContain('# Translation update briefing: docs/foo.md')
expect(brief).toContain('## Changed units')
expect(brief).toContain('### #4 paragraph — counterpart at docs/foo.zh.md:9')
expect(brief).toContain('Last-confirmed English:')
expect(brief).toContain('Current Chinese (bring this along):')
expect(brief).toContain('## First-occurrence notes')
expect(brief).toContain('agent智能体')
expect(brief).toContain('首次出现 annotations attach to the document-wide first occurrence only')
expect(brief).toContain('verify-translation-pairing --write docs/foo.md')
})
it('marks first-occurrence bundles and omits their unchanged confirmed text', () => {
const brief = renderTranslationBrief({
...base,
scope: {
kind: 'units',
bundles: [{ ...bundle, reason: 'first-occurrence', confirmedSourceText: bundle.currentSourceText }],
firstOccurrenceNotes: [],
},
})
expect(brief).toContain('unchanged; included for a first-occurrence move')
expect(brief).not.toContain('Last-confirmed English:')
})
it('renders the mechanical scope with the --apply command', () => {
const brief = renderTranslationBrief({ ...base, scope: { kind: 'mechanical' } })
expect(brief).toContain('## Mechanical update — no translation judgment involved')
expect(brief).toContain('gen-translation-brief --apply docs/foo.md')
expect(brief).not.toContain('## Changed units')
})
it('renders the section fallback under its own heading', () => {
const brief = renderTranslationBrief({
...base,
scope: { kind: 'sections', bundles: [bundle], firstOccurrenceNotes: [] },
})
expect(brief).toContain('## Changed sections')
expect(brief).toContain('fine-grained units do not align')
})
it('renders the document fallback with its reason and no bundles', () => {
const brief = renderTranslationBrief({
...base,
scope: { kind: 'document', reason: 'BOTH sides changed since the pair was last confirmed consistent, so no side is a trustworthy mapping anchor; decide which side owns each divergence.' },
})
expect(brief).toContain('## Whole-document update required')
expect(brief).toContain('BOTH sides changed')
expect(brief).toContain('locate the affected regions yourself')
})
it('renders the English-target digest for zh-to-en updates', () => {
const brief = renderTranslationBrief({
...base,
direction: 'zh-to-en',
sourcePath: 'docs/foo.zh.md',
counterpartPath: 'docs/foo.md',
scope: { kind: 'units', bundles: [bundle], firstOccurrenceNotes: [] },
})
expect(brief).toContain('exactly what the new Chinese states')
expect(brief).toContain('verify-translation-pairing --write docs/foo.md')
})
it('grows bundle fences past tilde runs in the text', () => {
const brief = renderTranslationBrief({
...base,
scope: {
kind: 'units',
bundles: [{ ...bundle, counterpartText: '~~~~\ninner\n~~~~\n' }],
firstOccurrenceNotes: [],
},
})
expect(brief).toContain('~~~~~markdown')
})
})

View File

@@ -0,0 +1,513 @@
/**
* Pure assembly of the minimal-update briefing for one out-of-sync
* translation pair: the authored side's changes since the last confirmed
* state at the narrowest safely mapped granularity (code-fence-only splice,
* changed Markdown units, heading sections, whole document), the terminology
* rows those changes touch, first-occurrence movement notes, and a digest of
* the binding update rules. The unit mapping, mechanical code splice, and
* first-occurrence tracking adopt the planner mechanics validated in the
* incremental-pipeline work (PR #684). The CLI wrapper is
* `scripts/gen-translation-brief.ts`; the workflow that consumes the
* briefing is `.agents/skills/dsh-translate-docs/SKILL.md`.
*/
import type { Nodes } from 'mdast'
import { parseTranslationMarkdown } from './translation-pairing.ts'
/** One block-level span of a Markdown document, in document order. */
export interface MarkdownSpan {
/** Position in the span list; briefing ids derive from it. */
index: number
/**
* Structural kind compared for alignment, language-neutral: container path
* plus node type for units (`root.3:tableRow`), depth for sections (`section:2`).
*/
kind: string
/** Reader-facing label: heading text for sections, node type for units. */
label: string
/** 1-based first source line. */
startLine: number
/** 1-based last source line. */
endLine: number
/** The span's text, trailing newline normalized to exactly one. */
text: string
}
function linesOf(markdown: string): string[] {
const lines = markdown.replaceAll('\r\n', '\n').split('\n')
if (lines.at(-1) === '') lines.pop()
return lines
}
function sliceLines(lines: string[], startLine: number, endLine: number): string {
return `${lines.slice(startLine - 1, endLine).join('\n')}\n`
}
/**
* List a document's translation units: the outermost block nodes a minimal
* update can replace independently. Headings, paragraphs, code fences, table
* rows, list items, block quotes, HTML blocks, thematic breaks, and link
* definitions are units; the container path is part of the kind so kind
* sequences only align when container membership also aligns.
*
* @param markdown - Document text.
* @returns Units in document order.
*/
export function markdownUnits(markdown: string): MarkdownSpan[] {
const positions: Array<{ kind: string; label: string; startLine: number; endLine: number }> = []
const visit = (node: Nodes, path: string): void => {
let kind: string | undefined
switch (node.type) {
case 'heading':
kind = `${path}:heading:${node.depth}`
break
case 'paragraph':
case 'code':
case 'tableRow':
case 'listItem':
case 'blockquote':
case 'html':
case 'thematicBreak':
case 'definition':
kind = `${path}:${node.type}`
break
default:
break
}
if (kind !== undefined && node.position !== undefined) {
positions.push({ kind, label: node.type, startLine: node.position.start.line, endLine: node.position.end.line })
return
}
if ('children' in node) for (const [index, child] of node.children.entries()) visit(child, `${path}.${index}`)
}
visit(parseTranslationMarkdown(markdown), 'root')
positions.sort((left, right) => left.startLine - right.startLine)
const lines = linesOf(markdown)
return positions.map((position, index) => ({
index,
...position,
text: sliceLines(lines, position.startLine, position.endLine),
}))
}
/**
* List a document's heading-delimited sections, including a leading
* `preamble` span when content precedes the first heading.
*
* @param markdown - Document text.
* @returns Sections in document order.
*/
export function sectionSpans(markdown: string): MarkdownSpan[] {
const headings: Array<{ depth: number; line: number; label: string }> = []
const visit = (node: Nodes): void => {
if (node.type === 'heading' && node.position !== undefined) {
let label = ''
const collect = (child: Nodes): void => {
if ('value' in child && typeof child.value === 'string') label += child.value
if ('children' in child) for (const grandchild of child.children) collect(grandchild)
}
for (const child of node.children) collect(child)
headings.push({ depth: node.depth, line: node.position.start.line, label })
}
if ('children' in node) for (const child of node.children) visit(child)
}
visit(parseTranslationMarkdown(markdown))
headings.sort((left, right) => left.line - right.line)
const lines = linesOf(markdown)
const spans: MarkdownSpan[] = []
const firstHeadingLine = headings[0]?.line ?? lines.length + 1
if (firstHeadingLine > 1) {
spans.push({ index: 0, kind: 'preamble', label: '(preamble before the first heading)', startLine: 1, endLine: firstHeadingLine - 1, text: sliceLines(lines, 1, firstHeadingLine - 1) })
}
for (const [order, heading] of headings.entries()) {
const endLine = (headings[order + 1]?.line ?? lines.length + 1) - 1
spans.push({
index: spans.length,
// Depth only: heading TEXT is translated across a pair, so it cannot
// participate in cross-language alignment.
kind: `section:${heading.depth}`,
label: heading.label === '' ? '(untitled section)' : heading.label,
startLine: heading.line,
endLine,
text: sliceLines(lines, heading.line, endLine),
})
}
return spans
}
/**
* Whether two span lists map one to one: same non-zero length and the same
* kind at every position.
*
* @param left - One document's spans.
* @param right - The other document's spans.
* @returns True when index-wise mapping is sound.
*/
export function spansAligned(left: MarkdownSpan[], right: MarkdownSpan[]): boolean {
return left.length > 0
&& left.length === right.length
&& left.every((span, index) => span.kind === right[index]?.kind)
}
/**
* Indices whose text differs between two aligned span lists.
*
* @param before - Spans of the earlier state.
* @param after - Spans of the later state, aligned with `before`.
* @returns Ascending changed indices.
*/
export function changedSpanIndices(before: MarkdownSpan[], after: MarkdownSpan[]): number[] {
return before.filter((span, index) => span.text !== after[index]?.text).map(span => span.index)
}
function codeSpansOf(markdown: string): MarkdownSpan[] {
return markdownUnits(markdown).filter(span => span.kind.endsWith(':code'))
.map((span, index) => ({ ...span, index }))
}
function replaceSpanTexts(markdown: string, spans: MarkdownSpan[], replacements: Map<number, string>): string {
const lines = linesOf(markdown)
for (const [index, replacement] of [...replacements.entries()].sort((left, right) => right[0] - left[0])) {
const span = spans[index]
if (span === undefined) throw new Error(`translation brief: unknown replacement span ${index}`)
lines.splice(span.startLine - 1, span.endLine - span.startLine + 1, ...linesOf(replacement))
}
return `${lines.join('\n')}\n`
}
function maskCodeSpans(markdown: string, spans: MarkdownSpan[]): string {
return replaceSpanTexts(markdown, spans, new Map(spans.map(span => [span.index, `DSH_TRANSLATION_CODE_${span.index}\n`])))
}
/**
* Compute the counterpart update for a change confined to fenced code
* blocks. Fences are byte-identical across a pair, so when the source's
* prose is untouched and the counterpart's fences match the last-confirmed
* source, splicing the edited fences into the counterpart is the complete
* update — no translation judgment is involved.
*
* @param confirmedSource - The changed side's last-confirmed text.
* @param currentSource - The changed side's current text.
* @param counterpart - The other side's current text.
* @returns The updated counterpart, or undefined when the change is not code-only.
*/
export function computeMechanicalUpdate(confirmedSource: string, currentSource: string, counterpart: string): string | undefined {
const confirmed = codeSpansOf(confirmedSource)
const current = codeSpansOf(currentSource)
const target = codeSpansOf(counterpart)
if (confirmed.length === 0 || confirmed.length !== current.length || confirmed.length !== target.length) return undefined
if (maskCodeSpans(confirmedSource, confirmed) !== maskCodeSpans(currentSource, current)) return undefined
if (confirmed.some((span, index) => span.text !== target[index]?.text)) return undefined
const changed = current.filter((span, index) => span.text !== confirmed[index]?.text)
if (changed.length === 0) return undefined
return replaceSpanTexts(counterpart, target, new Map(changed.map(span => [span.index, span.text])))
}
/** One parsed terminology-table data row. */
export interface TerminologyRow {
english: string
chinese: string
/** The 首次出现 cell (first-occurrence rendering), possibly empty. */
first: string
/** The verbatim table row. */
line: string
}
/** Strip Markdown emphasis and code markers from a terminology cell. */
function plainTerm(cell: string): string {
return cell.replaceAll('`', '').replaceAll('**', '').trim()
}
/**
* Parse the data rows of the terminology table.
*
* @param terminology - Full `docs/i18n/terminology.md` contents.
* @returns Rows in table order.
*/
export function parseTerminologyRows(terminology: string): TerminologyRow[] {
const rows: TerminologyRow[] = []
for (const line of terminology.split('\n')) {
if (!line.startsWith('|')) continue
if (/^\|[\s:|-]+\|$/.test(line)) continue
const cells = line.split('|').map(cell => cell.trim())
const english = plainTerm(cells[1] ?? '')
if (english === '' || english === 'English') continue
rows.push({ english, chinese: plainTerm(cells[2] ?? ''), first: plainTerm(cells[3] ?? ''), line })
}
return rows
}
/**
* Character offsets of a term's occurrences. English word-like terms match
* on word boundaries and accept plural inflections (`agents`, `registries`);
* other terms match as case-insensitive substrings.
*
* @param text - Text to search.
* @param term - The term to find.
* @param englishInflections - Whether to accept English plural forms.
* @returns Ascending match offsets.
*/
export function termOffsets(text: string, term: string, englishInflections = false): number[] {
if (term === '') return []
const escape = (value: string): string => value.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')
const wordLike = /^[A-Za-z0-9][A-Za-z0-9 ._-]*[A-Za-z0-9]$/.test(term)
const inflected = englishInflections && wordLike
? /[^aeiou]y$/i.test(term)
? `${escape(term.slice(0, -1))}(?:y|ies)`
: `${escape(term)}(?:s|es)?`
: escape(term)
const expression = new RegExp(wordLike ? `(?<![A-Za-z0-9_])${inflected}(?![A-Za-z0-9_])` : inflected, 'gi')
return [...text.matchAll(expression)].map(match => match.index)
}
/** The two update directions a pair supports. */
export type BriefDirection = 'en-to-zh' | 'zh-to-en'
/** Whether a row's source-language term occurs in the given text. */
function rowOccurs(row: TerminologyRow, direction: BriefDirection, text: string): boolean {
const terms = direction === 'en-to-zh' ? [row.english] : [row.first, row.chinese].filter(term => /[一-鿿]/.test(term))
return terms.some(term => termOffsets(text, term, direction === 'en-to-zh').length > 0)
}
/**
* Select the terminology rows whose source-language term occurs in the
* changed text (old and new states combined).
*
* @param terminology - Full `docs/i18n/terminology.md` contents.
* @param direction - Update direction; decides which columns to match.
* @param changedText - Concatenated old and new text of the changed spans.
* @returns Matched rows in table order.
*/
export function relevantTerminologyRows(terminology: string, direction: BriefDirection, changedText: string): TerminologyRow[] {
return parseTerminologyRows(terminology).filter(row => rowOccurs(row, direction, changedText))
}
function lineAtOffset(text: string, offset: number): number {
return text.slice(0, offset).split('\n').length
}
function spanIndexAtOffset(text: string, spans: MarkdownSpan[], offset: number | undefined): number | undefined {
if (offset === undefined) return undefined
const line = lineAtOffset(text, offset)
return spans.find(span => line >= span.startLine && line <= span.endLine)?.index
}
/** First-occurrence guidance computed for a Chinese-target update. */
export interface FirstOccurrenceContext {
/** Human-readable notes for the briefing. */
notes: string[]
/** Unchanged span indices that must join the briefing because a first occurrence moved into or out of them. */
extraSpanIndices: number[]
}
/**
* Track document-wide first occurrences of the relevant English terms. The
* 首次出现 rendering attaches to a term's first occurrence, so when an edit
* moves that occurrence across spans, both the old and new spans need
* counterpart edits even when only one of them changed.
*
* @param confirmedSource - Last-confirmed English text.
* @param currentSource - Current English text.
* @param confirmedSpans - Spans of the last-confirmed English text.
* @param currentSpans - Spans of the current English text, aligned with `confirmedSpans`.
* @param rows - The relevant terminology rows.
* @param changed - Span indices already in the briefing.
* @returns Notes and extra span indices to include.
*/
export function firstOccurrenceContext(
confirmedSource: string,
currentSource: string,
confirmedSpans: MarkdownSpan[],
currentSpans: MarkdownSpan[],
rows: TerminologyRow[],
changed: Set<number>,
): FirstOccurrenceContext {
const notes: string[] = []
const extra = new Set<number>()
for (const row of rows) {
if (row.first === '') continue
const oldIndex = spanIndexAtOffset(confirmedSource, confirmedSpans, termOffsets(confirmedSource, row.english, true)[0])
const newIndex = spanIndexAtOffset(currentSource, currentSpans, termOffsets(currentSource, row.english, true)[0])
if (oldIndex === newIndex) continue
for (const index of [oldIndex, newIndex]) {
if (index !== undefined && !changed.has(index)) extra.add(index)
}
notes.push(`${row.english}: the document-wide first occurrence moved from ${oldIndex === undefined ? 'absent' : `#${oldIndex}`} to ${newIndex === undefined ? 'absent' : `#${newIndex}`}; the ${row.first} form moves with it (later occurrences drop the annotation).`)
}
return { notes, extraSpanIndices: [...extra].sort((left, right) => left - right) }
}
/** Smallest fence of `mark` characters that safely wraps `body`. */
function fenceFor(body: string, mark: '`' | '~'): string {
let longest = 2
for (const line of body.split('\n')) {
const run = new RegExp(`^\\s*(${mark === '`' ? '`' : '~'}{3,})`).exec(line)
if (run?.[1] !== undefined && run[1].length > longest) longest = run[1].length
}
return mark.repeat(longest + 1)
}
/** One changed (or first-occurrence) span with its three-way context. */
export interface BriefBundle {
/** Span index shared by the aligned documents. */
index: number
/** Human label: heading text or node type. */
label: string
/** Why the bundle is present when its source text did not change. */
reason?: 'first-occurrence' | undefined
confirmedSourceText: string
currentSourceText: string
counterpartText: string
/** 1-based line the counterpart span starts on. */
counterpartStartLine: number
}
/** The granularities a briefing can map the change at, narrowest first. */
export type BriefScope =
| { kind: 'mechanical' }
| { kind: 'units'; bundles: BriefBundle[]; firstOccurrenceNotes: string[] }
| { kind: 'sections'; bundles: BriefBundle[]; firstOccurrenceNotes: string[] }
| { kind: 'document'; reason: string }
/** Inputs for rendering one pair's briefing. */
export interface TranslationBriefInput {
/** Repo-relative path of the side that changed. */
sourcePath: string
/** Repo-relative path of the counterpart to update. */
counterpartPath: string
direction: BriefDirection
/** Unified diff of the changed side, last-confirmed to current. */
diff: string
scope: BriefScope
terminology: TerminologyRow[]
}
const ZH_TARGET_DIGEST = [
'- Edit ONLY what the change requires; preserve the reviewed phrasing of everything unchanged.',
'- Nothing added, nothing dropped: the Chinese must state exactly what the new English states.',
'- Write natural institutional technical Chinese, not word-by-word gloss; terse stays terse.',
'- Code fences byte-identical to the English side, comments included; inline code spans verbatim.',
'- Relative links keep the `.md` target; only the switcher line links `.zh.md`.',
'- Structure mirrors the counterpart: heading depths and order, list kinds and item counts, table rows and columns.',
'- 首次出现 annotations attach to the document-wide first occurrence only; later occurrences use the bare form, and an empty 首次出现 cell means never gloss.',
'- Typography: one half-width space between Chinese and Latin or digits; full-width punctuation in Chinese prose; 顿号 for enumerations; second person is 你.',
'- One physical line per paragraph; exactly one trailing newline.',
]
const EN_TARGET_DIGEST = [
'- Edit ONLY what the change requires; preserve the reviewed phrasing of everything unchanged.',
'- Nothing added, nothing dropped: the English must state exactly what the new Chinese states.',
'- Write concise professional developer prose, not word-by-word gloss; terse stays terse.',
'- Code fences byte-identical to the Chinese side, comments included; inline code spans verbatim.',
'- Relative links keep the `.md` target; only the switcher line links `.zh.md`.',
'- Structure mirrors the counterpart: heading depths and order, list kinds and item counts, table rows and columns.',
'- One physical line per paragraph; exactly one trailing newline.',
]
function renderBundles(out: string[], input: TranslationBriefInput, bundles: BriefBundle[], firstOccurrenceNotes: string[]): void {
const sourceLanguage = input.direction === 'en-to-zh' ? 'English' : 'Chinese'
const counterpartLanguage = input.direction === 'en-to-zh' ? 'Chinese' : 'English'
for (const bundle of bundles) {
out.push('')
out.push(`### #${bundle.index} ${bundle.label}${bundle.reason === 'first-occurrence' ? ' — unchanged; included for a first-occurrence move' : ''} — counterpart at ${input.counterpartPath}:${bundle.counterpartStartLine}`)
const fence = fenceFor([bundle.confirmedSourceText, bundle.currentSourceText, bundle.counterpartText].join('\n'), '~')
if (bundle.confirmedSourceText !== bundle.currentSourceText) {
out.push('')
out.push(`Last-confirmed ${sourceLanguage}:`)
out.push('')
out.push(`${fence}markdown`)
out.push(bundle.confirmedSourceText.trimEnd())
out.push(fence)
}
out.push('')
out.push(`Current ${sourceLanguage}:`)
out.push('')
out.push(`${fence}markdown`)
out.push(bundle.currentSourceText.trimEnd())
out.push(fence)
out.push('')
out.push(`Current ${counterpartLanguage} (bring this along):`)
out.push('')
out.push(`${fence}markdown`)
out.push(bundle.counterpartText.trimEnd())
out.push(fence)
}
if (firstOccurrenceNotes.length > 0) {
out.push('')
out.push('## First-occurrence notes')
out.push('')
for (const note of firstOccurrenceNotes) out.push(`- ${note}`)
}
}
/**
* Render the complete briefing for one out-of-sync pair.
*
* @param input - Diff, mapped scope, terminology, and pair identity.
* @returns Markdown briefing text.
*/
export function renderTranslationBrief(input: TranslationBriefInput): string {
const sourceLanguage = input.direction === 'en-to-zh' ? 'English' : 'Chinese'
const counterpartLanguage = input.direction === 'en-to-zh' ? 'Chinese' : 'English'
const out: string[] = []
out.push(`# Translation update briefing: ${input.sourcePath}`)
out.push('')
out.push(`The ${sourceLanguage} side changed; bring \`${input.counterpartPath}\` along with the smallest edit that covers the change.`)
if (input.scope.kind === 'mechanical') {
out.push('')
out.push('## Mechanical update — no translation judgment involved')
out.push('')
out.push(`Every change since the last confirmed state is inside fenced code blocks, which are byte-identical across the pair. Run \`pnpm run gen-translation-brief --apply ${input.sourcePath}\` to splice the updated fences into the counterpart (the result is structure-validated before writing), then record per the Finish steps.`)
}
out.push('')
out.push(`## ${sourceLanguage} diff (last-confirmed → current)`)
out.push('')
const diffFence = fenceFor(input.diff, '`')
out.push(`${diffFence}diff`)
out.push(input.diff.trimEnd())
out.push(diffFence)
switch (input.scope.kind) {
case 'mechanical':
break
case 'units':
out.push('')
out.push(`## Changed units (last-confirmed ${sourceLanguage} → current ${sourceLanguage}, with the current ${counterpartLanguage})`)
renderBundles(out, input, input.scope.bundles, input.scope.firstOccurrenceNotes)
break
case 'sections':
out.push('')
out.push('## Changed sections (fine-grained units do not align across the pair; whole heading sections shown)')
renderBundles(out, input, input.scope.bundles, input.scope.firstOccurrenceNotes)
break
case 'document':
out.push('')
out.push('## Whole-document update required')
out.push('')
out.push(`${input.scope.reason} Open \`${input.counterpartPath}\` directly, locate the affected regions yourself, and reconcile under docs/i18n/translation-rules.md.`)
break
default:
input.scope satisfies never
}
if (input.terminology.length > 0) {
out.push('')
out.push('## Binding terminology rows matching this change (docs/i18n/terminology.md)')
out.push('')
out.push('| English | 中文 | 首次出现 | 不要译作 | 备注 |')
out.push('|---|---|---|---|---|')
for (const row of input.terminology) out.push(row.line)
out.push('')
out.push('For any term you introduce that is not listed above, consult the full table before inventing a rendering.')
}
out.push('')
out.push('## Rules digest (full rules: docs/i18n/translation-rules.md)')
out.push('')
out.push(...(input.direction === 'en-to-zh' ? ZH_TARGET_DIGEST : EN_TARGET_DIGEST))
out.push('')
out.push('## Finish')
out.push('')
out.push('1. Apply the smallest counterpart edit that covers the change, then verify the changed spans clause by clause against the source.')
out.push(`2. \`pnpm run verify-translation-pairing --write ${input.sourcePath.replace(/\.zh\.md$/, '.md')}\``)
out.push(`3. \`pnpm run verify-translation-pairing ${input.sourcePath.replace(/\.zh\.md$/, '.md')}\``)
out.push('')
return out.join('\n')
}

View File

@@ -3,7 +3,9 @@
import { describe, expect, it } from 'vitest'
import {
isTranslationScopeFile,
pairAnchorOfArgument,
parseTranslationMarkdown,
parseTranslationPairingCliArgs,
parseTranslationPairingManifest,
translationStructureDiff,
translationStructureSignature,
@@ -102,3 +104,40 @@ describe('translation structural signature', () => {
])
})
})
describe('pair CLI arguments', () => {
it('normalizes any pair file or bare stem to the English anchor', () => {
expect(pairAnchorOfArgument('docs/foo.md')).toBe('docs/foo.md')
expect(pairAnchorOfArgument('docs/foo.zh.md')).toBe('docs/foo.md')
expect(pairAnchorOfArgument('docs/foo.i18n.yaml')).toBe('docs/foo.md')
expect(pairAnchorOfArgument('docs/foo')).toBe('docs/foo.md')
expect(pairAnchorOfArgument('.\\docs\\foo.zh.md')).toBe('docs/foo.md')
})
it('scopes a check to named pairs and dedupes the three spellings', () => {
expect(parseTranslationPairingCliArgs(['docs/foo.zh.md', 'docs/foo.i18n.yaml', 'docs/bar.md'])).toEqual({
mode: 'check',
scope: 'pairs',
anchors: ['docs/bar.md', 'docs/foo.md'],
})
expect(parseTranslationPairingCliArgs([])).toEqual({ mode: 'check', scope: 'corpus', anchors: [] })
})
it('requires --write to name confirmed pairs or opt into --all', () => {
expect(() => parseTranslationPairingCliArgs(['--write'])).toThrow('requires the pair(s) you confirmed')
expect(parseTranslationPairingCliArgs(['--write', 'docs/foo.md'])).toEqual({
mode: 'write',
scope: 'pairs',
anchors: ['docs/foo.md'],
})
expect(parseTranslationPairingCliArgs(['--write', '--all'])).toEqual({ mode: 'write', scope: 'corpus', anchors: [] })
expect(() => parseTranslationPairingCliArgs(['--write', '--all', 'docs/foo.md'])).toThrow('not both')
})
it('keeps --list corpus-only and rejects unknown flags', () => {
expect(parseTranslationPairingCliArgs(['--list'])).toEqual({ mode: 'list', scope: 'corpus', anchors: [] })
expect(() => parseTranslationPairingCliArgs(['--list', 'docs/foo.md'])).toThrow('takes no other flags or paths')
expect(() => parseTranslationPairingCliArgs(['--all'])).toThrow('--all only applies to --write')
expect(() => parseTranslationPairingCliArgs(['--frobnicate'])).toThrow('unknown flag(s): --frobnicate')
})
})

View File

@@ -102,6 +102,66 @@ export function parseTranslationPairingManifest(content: string): TranslationPai
return { excluded: excludedField(record) }
}
/**
* Normalize one CLI pair argument to its English anchor path: any of the
* pair's three files (`foo.md`, `foo.zh.md`, `foo.i18n.yaml`) or the bare
* `foo` stem names the same pair, and platform separators are accepted.
*
* @param argument - Repo-relative path as passed on a command line.
* @returns The pair's `foo.md` anchor path with `/` separators.
*/
export function pairAnchorOfArgument(argument: string): string {
const normalized = argument.split('\\').join('/').replace(/^\.\//, '')
if (normalized.endsWith('.zh.md')) return `${normalized.slice(0, -'.zh.md'.length)}.md`
if (normalized.endsWith('.i18n.yaml')) return `${normalized.slice(0, -'.i18n.yaml'.length)}.md`
if (normalized.endsWith('.md')) return normalized
return `${normalized}.md`
}
/** A parsed `verify-translation-pairing` invocation. */
export interface TranslationPairingCliRequest {
mode: 'check' | 'list' | 'write'
/** `corpus` runs discovery over the whole tree; `pairs` touches only the named anchors. */
scope: 'corpus' | 'pairs'
/** English anchor paths, empty for corpus scope. */
anchors: string[]
}
/**
* Parse and validate `verify-translation-pairing` CLI arguments.
*
* Check accepts optional pair paths; `--write` requires either pair paths or
* `--all` so a bulk re-record is always an explicit choice — a bare
* `--write` would silently bless every drifted pair in the tree, including
* ones the caller never confirmed. `--list` is corpus-only.
*
* @param argv - Arguments after the script name.
* @returns The validated request.
* @throws Error when flags or their combination are invalid.
*/
export function parseTranslationPairingCliArgs(argv: string[]): TranslationPairingCliRequest {
const flags = argv.filter(argument => argument.startsWith('--'))
const anchors = [...new Set(argv.filter(argument => !argument.startsWith('--')).map(pairAnchorOfArgument))].sort()
const unknown = flags.filter(flag => !['--list', '--write', '--all'].includes(flag))
if (unknown.length > 0) throw new Error(`unknown flag(s): ${unknown.join(', ')}`)
const listMode = flags.includes('--list')
const writeMode = flags.includes('--write')
const allMode = flags.includes('--all')
if (listMode && (writeMode || allMode || anchors.length > 0)) {
throw new Error('--list reports the whole corpus and takes no other flags or paths')
}
if (allMode && !writeMode) throw new Error('--all only applies to --write')
if (writeMode) {
if (anchors.length > 0 && allMode) throw new Error('--write takes either pair paths or --all, not both')
if (anchors.length === 0 && !allMode) {
throw new Error('--write requires the pair(s) you confirmed (any file of a pair), or --all to re-record every complete pair; recording pairs you did not review blesses unconfirmed content')
}
return { mode: 'write', scope: allMode ? 'corpus' : 'pairs', anchors }
}
if (listMode) return { mode: 'list', scope: 'corpus', anchors: [] }
return { mode: 'check', scope: anchors.length > 0 ? 'pairs' : 'corpus', anchors }
}
/** The structural surface compared between the two sides of a pair. */
export interface TranslationStructureSignature {
/** Heading depths in document order (h2 -> 2). */

View File

@@ -46,6 +46,26 @@
"symbol": "LlmModelContext",
"source": "packages/llm/llm/src/types.ts"
},
{
"doc": "docs/core-data-structures/core.md",
"symbol": "ReasoningEffortId",
"source": "packages/llm/llm/src/brand.ts"
},
{
"doc": "docs/core-data-structures/core.md",
"symbol": "LlmReasoningEffortInfo",
"source": "packages/llm/llm/src/types.ts"
},
{
"doc": "docs/core-data-structures/core.md",
"symbol": "LlmModelReasoningInfo",
"source": "packages/llm/llm/src/types.ts"
},
{
"doc": "docs/core-data-structures/core.md",
"symbol": "LlmResolvedModelInfo",
"source": "packages/llm/llm/src/types.ts"
},
{
"doc": "docs/core-data-structures/core.md",
"symbol": "GenerateOptions",
@@ -292,6 +312,11 @@
"source": "packages/llm/llm/src/assembler.ts",
"projection": "public-api"
},
{
"doc": "docs/core-data-structures/llm-streaming.md",
"symbol": "PreparedLlmCall",
"source": "packages/llm/llm/src/index.ts"
},
{
"doc": "docs/core-data-structures/llm-streaming.md",
"symbol": "LlmAdapter",

View File

@@ -55,6 +55,8 @@ const SENTENCE_MODEL_EXPERIENCE: Readonly<Record<string, SentenceContract>> = {
'packages/client/ui-layout': { kind: 'none', reason: 'Browser-side UI plugin layer; registers no model surface.' },
'packages/client/ui-sidebar': { kind: 'none', reason: 'Browser-side UI plugin layer; registers no model surface.' },
'packages/client/ui-conversation': { kind: 'none', reason: 'Browser-side UI plugin layer; registers no model surface.' },
'packages/client/ui-slash': { kind: 'none', reason: 'Browser-side UI plugin layer; registers no model surface.' },
'packages/client/ui-command': { kind: 'indirect', reason: 'The dispatch paths trigger the host command.execute RPC; each command handler\'s host package owns any model-visible effect.' },
'packages/client/ui-question': { kind: 'indirect', reason: 'The package mounts dsh-tool-ask-user; that tool owns the model-visible schema and answer rendering.' },
'packages/client/ui-trajectory': { kind: 'none', reason: 'Browser-side UI plugin layer; registers no model surface.' },
'packages/client/ui-workspace': { kind: 'none', reason: 'Browser-side UI plugin layer; registers no model surface.' },

View File

@@ -2,8 +2,10 @@
* Enforce complete English/Chinese pairs, matching structure, and recorded git
* blob hashes for every in-scope document. The manifest contains only explicit
* exclusions, which may have neither a counterpart nor a sidecar.
* `--list` reports state and `--write` records both sides after human review.
* Translation quality remains a review responsibility.
* `--list` reports state; `--write <pairs...>` records the named confirmed
* pairs (`--write --all` records every complete pair); a check or write named
* with pair paths touches only those pairs, so update iteration does not pay
* for a corpus scan. Translation quality remains a review responsibility.
* See `docs/i18n/README.md` for the owning contract.
*/
@@ -13,6 +15,7 @@ import { basename, join, resolve, sep } from 'node:path'
import {
linksTo,
parseTranslationMarkdown,
parseTranslationPairingCliArgs,
parseTranslationPairingManifest,
isTranslationScopeFile,
TRANSLATION_SCOPE_GLOB_EXCLUDES,
@@ -21,8 +24,15 @@ import {
} from './translation-pairing.ts'
const root = resolve(import.meta.dirname, '..')
const listMode = process.argv.includes('--list')
const writeMode = process.argv.includes('--write')
let request: ReturnType<typeof parseTranslationPairingCliArgs>
try {
request = parseTranslationPairingCliArgs(process.argv.slice(2))
} catch (error) {
console.error(`verify-translation-pairing: ${error instanceof Error ? error.message : String(error)}`)
process.exit(2)
}
const listMode = request.mode === 'list'
const writeMode = request.mode === 'write'
/** Discover source Markdown and pairing sidecars before applying the corpus predicate. */
const SCOPE_PATTERNS = [
@@ -77,32 +87,67 @@ function renderMeta(source: string, sourceHash: string, zh: string, zhHash: stri
'# Bilingual-pair consistency record (docs/i18n/README.md): the git blob hash of each',
'# side as of the last confirmed-consistent state. Both languages carry equal authority;',
'# after editing either side, bring the other along and re-record with:',
'# pnpm run verify-translation-pairing --write',
`# pnpm run verify-translation-pairing --write ${source}`,
`${basename(source)}: ${sourceHash}`,
`${basename(zh)}: ${zhHash}`,
'',
].join('\n')
}
// Enumerate the scope once.
// Enumerate the scope once: the whole corpus, or exactly the named pairs'
// three files (a named pair whose files are absent is caught by the same
// completeness rules that cover discovered remnants).
const files = new Set<string>()
for (const pattern of SCOPE_PATTERNS) {
for (const match of globSync(pattern, { cwd: root, exclude: TRANSLATION_SCOPE_GLOB_EXCLUDES })) {
const normalized = match.split(sep).join('/')
if (isTranslationScopeFile(normalized)) files.add(normalized)
if (request.scope === 'pairs') {
for (const anchor of request.anchors) {
for (const file of [anchor, ...Object.values(pairPaths(anchor))]) {
if (existsSync(join(root, file))) files.add(file)
}
// A named anchor with no files on disk still enters the source list so
// the check reports it instead of silently passing an empty scope.
if (!existsSync(join(root, anchor))) files.add(anchor)
}
} else {
for (const pattern of SCOPE_PATTERNS) {
for (const match of globSync(pattern, { cwd: root, exclude: TRANSLATION_SCOPE_GLOB_EXCLUDES })) {
const normalized = match.split(sep).join('/')
if (isTranslationScopeFile(normalized)) files.add(normalized)
}
}
}
const translations = [...files].filter(f => f.endsWith('.zh.md')).sort()
const metas = [...files].filter(f => f.endsWith('.i18n.yaml')).sort()
const sources = [...files].filter(f => f.endsWith('.md') && !f.endsWith('.zh.md')).sort()
// --write: (re)record both hashes for every complete pair, creating missing records.
if (request.scope === 'pairs') {
const rejected = request.anchors.filter(anchor => !isTranslationScopeFile(anchor) || isExcluded(anchor))
const absent = request.anchors.filter(anchor => ![anchor, ...Object.values(pairPaths(anchor))].some(file => existsSync(join(root, file))))
if (rejected.length > 0 || absent.length > 0) {
for (const anchor of rejected) {
console.error(`verify-translation-pairing: ${anchor} is not an in-scope pair (excluded or outside the documentation corpus; see docs/i18n/README.md)`)
}
for (const anchor of absent) {
console.error(`verify-translation-pairing: ${anchor} names no pair on disk (none of its three files exist)`)
}
process.exit(2)
}
}
// --write: (re)record both hashes for the requested complete pairs, creating
// missing records. A named pair that cannot be recorded (missing counterpart)
// fails loud; corpus scope (--all) skips pairless sources as before.
if (writeMode) {
let written = 0
for (const source of sources) {
if (isExcluded(source)) continue
const { zh, meta } = pairPaths(source)
if (!existsSync(join(root, zh))) continue
if (!existsSync(join(root, source)) || !existsSync(join(root, zh))) {
if (request.scope === 'pairs') {
console.error(`verify-translation-pairing: cannot record ${source}: missing ${existsSync(join(root, source)) ? zh : source}`)
process.exit(2)
}
continue
}
const record = renderMeta(source, blobHash(readFileSync(join(root, source))), zh, blobHash(readFileSync(join(root, zh))))
if (existsSync(join(root, meta)) && readFileSync(join(root, meta), 'utf8') === record) continue
writeFileSync(join(root, meta), record)
@@ -204,7 +249,9 @@ if (listMode) {
}
if (errors.length === 0) {
console.log(`verify-translation-pairing: ${pairAnchors.size} pair(s) checked across all in-scope documentation, all consistent.`)
console.log(request.scope === 'pairs'
? `verify-translation-pairing: ${pairAnchors.size} named pair(s) consistent; the corpus-wide check still runs in doc-sync.`
: `verify-translation-pairing: ${pairAnchors.size} pair(s) checked across all in-scope documentation, all consistent.`)
process.exit(0)
}