mirror of
https://github.com/deepseek-ai/deepseek-harness
synced 2026-08-15 21:04:50 +00:00
Merge branch 'master' into nih-imp-gates
This commit is contained in:
@@ -40,6 +40,8 @@ export const LINK_MAP: Record<string, string> = {
|
||||
HookContext: 'core.md',
|
||||
LlmCallConfig: 'core.md',
|
||||
LlmModelContext: 'core.md',
|
||||
LlmModelReasoningInfo: 'core.md',
|
||||
LlmResolvedModelInfo: 'core.md',
|
||||
LlmFailure: 'llm-streaming.md',
|
||||
LlmModelInfo: 'core.md',
|
||||
LlmProviderInfo: 'core.md',
|
||||
@@ -92,6 +94,7 @@ export const LINK_MAP: Record<string, string> = {
|
||||
CommandResult: 'commands.md',
|
||||
CommandSurface: 'commands.md',
|
||||
LlmAdapter: 'llm-streaming.md',
|
||||
PreparedLlmCall: 'llm-streaming.md',
|
||||
LlmService: 'llm-streaming.md',
|
||||
StreamChunk: 'llm-streaming.md',
|
||||
CreateSessionOptions: 'persistence.md',
|
||||
@@ -207,6 +210,10 @@ const FOUNDATION_TYPE_NAMES = new Set([
|
||||
/** Project types deliberately documented outside the core-data catalog. */
|
||||
const TYPE_LINK_EXEMPTIONS: Readonly<Record<string, string>> = {
|
||||
AgentFactory: 'agent creation seam is owned by packages/core/agent/README.md',
|
||||
BeginCommandRequest: 'event-local request contract is owned by packages/client/ui-slash/src/types.ts',
|
||||
InsertReferenceRequest: 'event-local request contract is owned by packages/client/ui-slash/src/types.ts',
|
||||
ConsumeTokenRequest: 'event-local request contract is owned by packages/client/ui-slash/src/types.ts',
|
||||
InsertTextRequest: 'event-local request contract is owned by packages/client/ui-slash/src/types.ts',
|
||||
AgentHandle: 'agent ownership handle is owned by packages/core/agent/README.md',
|
||||
BashEnvContributor: 'service-local extension type is owned by packages/bash/tool-bash/src/index.ts',
|
||||
BashEnvVariableInfo: 'service-local metadata type is owned by packages/bash/tool-bash/src/index.ts',
|
||||
@@ -389,7 +396,7 @@ export function collectEvents(scanRoot: string = root): EventEntry[] {
|
||||
const where = `event '${name}' (${src})`
|
||||
checkTypeLinks(where, member, sf, typeLinkViolations)
|
||||
if (!mode) {
|
||||
violations.push(`${where} is missing an @mode tag. Add '@mode emit|waterfall|parallel|serial' to its JSDoc (see AGENTS.md).`)
|
||||
violations.push(`${where} is missing an @mode tag. Add '@mode emit|waterfall|parallel|serial|bail' to its JSDoc (see AGENTS.md).`)
|
||||
}
|
||||
// Conclusive structural check: a trailing `next: () => …` parameter is a
|
||||
// waterfall. (emit vs parallel vs serial is not structurally
|
||||
@@ -580,7 +587,7 @@ export function renderEvents(events: EventEntry[]): string {
|
||||
'',
|
||||
'The **harness tier** below (the `@deepseek-ai/dsh-*` packages) is the vocabulary this repo owns, grouped by scope. The **inherited tier** at the end is the cordis-core + loader/hmr/timer event surface a plugin also sees — pinned vendor source, summarized tersely. The event-dispatch methods themselves are generated in the [Cordis core Events API](core/events.md).',
|
||||
'',
|
||||
'Dispatch modes: **emit** (fire-and-forget), **waterfall** (each listener gets `next()` and may transform or veto — see [waterfall semantics](../cordis-primer.md#cordis-waterfall-semantics)), **parallel** (awaited fan-out; all listeners run), **serial** (awaited in registration order until one returns a bail value — anything other than `null`, `false`, or `undefined`).',
|
||||
'Dispatch modes: **emit** (fire-and-forget), **waterfall** (each listener gets `next()` and may transform or veto — see [waterfall semantics](../cordis-primer.md#cordis-waterfall-semantics)), **parallel** (awaited fan-out; all listeners run), **serial** (awaited in registration order until one returns a bail value — anything other than `null`, `false`, or `undefined`), **bail** (synchronous in-order dispatch until one listener returns a bail value; the scoped input-mutation events use it for an applied/not-applied answer).',
|
||||
'',
|
||||
]
|
||||
const scopes = [...new Set(events.map(e => e.scope))].sort()
|
||||
|
||||
@@ -917,8 +917,13 @@ function renderEventRelations(pkgs: Pkg[]): string {
|
||||
lines.push(`| \`${event.name}\` | \`${event.mode}\` | ${sourceLink(event.source)} | ${relationPackages(relation.dispatchers, pkgsByShort)} | ${listenerPackages(relation.listeners, pkgsByShort)} |`)
|
||||
}
|
||||
// Every declared event needs a dispatcher: zero means dead vocabulary or an
|
||||
// unrecognized semantic dispatch shape. Listener-free extension points remain valid.
|
||||
// unrecognized semantic dispatch shape. Listener-free extension points remain
|
||||
// valid. Client-declared events are exempt: the relation scan seeds the HOST
|
||||
// aggregate program only (host+client cannot share one program — the cordis
|
||||
// Context merges collide), so client dispatch sites are structurally
|
||||
// invisible here; their rows stay in the table for the declarations' sake.
|
||||
const undispatched = [...events]
|
||||
.filter(event => !event.source.startsWith('packages/client/'))
|
||||
.filter(event => (relations.get(event.name)?.dispatchers.size ?? 0) === 0)
|
||||
.map(event => event.name)
|
||||
.sort()
|
||||
|
||||
302
scripts/gen-translation-brief.ts
Normal file
302
scripts/gen-translation-brief.ts
Normal file
@@ -0,0 +1,302 @@
|
||||
/**
|
||||
* Print the minimal-update briefing for out-of-sync translation pairs:
|
||||
* `pnpm run gen-translation-brief [--apply] [pair paths...]`. With no
|
||||
* arguments it discovers every out-of-sync pair; with arguments (any file
|
||||
* of a pair) it briefs exactly those pairs and fails loud on in-sync,
|
||||
* incomplete, or out-of-scope requests. Each briefing maps the change at
|
||||
* the narrowest safe granularity — code-fence-only splice, changed
|
||||
* Markdown units, heading sections, whole document — and `--apply` writes
|
||||
* the computed counterpart for pairs whose change is code-fence-only.
|
||||
* The briefing contract lives in `scripts/translation-brief.ts`; the
|
||||
* consuming workflow is `.agents/skills/dsh-translate-docs/SKILL.md`.
|
||||
*/
|
||||
|
||||
import { spawnSync } from 'node:child_process'
|
||||
import { existsSync, globSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { basename, join, resolve, sep } from 'node:path'
|
||||
import {
|
||||
isTranslationScopeFile,
|
||||
pairAnchorOfArgument,
|
||||
parseTranslationMarkdown,
|
||||
parseTranslationPairingManifest,
|
||||
TRANSLATION_SCOPE_GLOB_EXCLUDES,
|
||||
translationStructureDiff,
|
||||
translationStructureSignature,
|
||||
} from './translation-pairing.ts'
|
||||
import {
|
||||
changedSpanIndices,
|
||||
computeMechanicalUpdate,
|
||||
firstOccurrenceContext,
|
||||
markdownUnits,
|
||||
relevantTerminologyRows,
|
||||
renderTranslationBrief,
|
||||
sectionSpans,
|
||||
spansAligned,
|
||||
type BriefBundle,
|
||||
type BriefDirection,
|
||||
type BriefScope,
|
||||
type MarkdownSpan,
|
||||
} from './translation-brief.ts'
|
||||
|
||||
const root = resolve(import.meta.dirname, '..')
|
||||
const manifest = parseTranslationPairingManifest(readFileSync(join(root, 'scripts/translation-pairing.manifest.json'), 'utf8'))
|
||||
const terminology = readFileSync(join(root, 'docs/i18n/terminology.md'), 'utf8')
|
||||
|
||||
function isExcluded(file: string): boolean {
|
||||
return manifest.excluded.some(entry => (entry.endsWith('/') ? file.startsWith(entry) : file === entry))
|
||||
}
|
||||
|
||||
/** Recorded hashes of one consistency record: basename → blob hash. */
|
||||
function parseMeta(content: string): Map<string, string> | undefined {
|
||||
const out = new Map<string, string>()
|
||||
for (const line of content.split('\n')) {
|
||||
if (line === '' || line.startsWith('#')) continue
|
||||
const match = /^([^:#]+\.md): ([0-9a-f]{40})$/.exec(line)
|
||||
if (!match?.[1] || !match[2]) return undefined
|
||||
out.set(match[1], match[2])
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
function git(args: string[], allowedExitCodes: number[] = [0]): string {
|
||||
const result = spawnSync('git', ['-C', root, ...args], { encoding: 'utf8', maxBuffer: 1 << 26 })
|
||||
if (result.error) throw result.error
|
||||
if (!allowedExitCodes.includes(result.status ?? -1)) {
|
||||
throw new Error(`git ${args.join(' ')} failed: ${result.stderr}`)
|
||||
}
|
||||
return result.stdout
|
||||
}
|
||||
|
||||
function blobText(hash: string): string {
|
||||
return git(['cat-file', '-p', hash])
|
||||
}
|
||||
|
||||
/** Unified diff between two texts, headers stripped, via `git diff --no-index`. */
|
||||
function diffTexts(before: string, after: string): string {
|
||||
const dir = mkdtempSync(join(tmpdir(), 'translation-brief-'))
|
||||
try {
|
||||
writeFileSync(join(dir, 'last-confirmed.md'), before)
|
||||
writeFileSync(join(dir, 'current.md'), after)
|
||||
const raw = git(['diff', '--no-index', '--unified=2', join(dir, 'last-confirmed.md'), join(dir, 'current.md')], [0, 1])
|
||||
return raw.split('\n')
|
||||
.filter(line => !line.startsWith('diff --git') && !line.startsWith('index ') && !line.startsWith('--- ') && !line.startsWith('+++ '))
|
||||
.join('\n')
|
||||
.trim()
|
||||
} finally {
|
||||
rmSync(dir, { recursive: true, force: true })
|
||||
}
|
||||
}
|
||||
|
||||
interface PairState {
|
||||
anchor: string
|
||||
zh: string
|
||||
meta: string
|
||||
enDrifted: boolean
|
||||
zhDrifted: boolean
|
||||
enLast: string
|
||||
zhLast: string
|
||||
}
|
||||
|
||||
/** Load one pair's recorded and current state, or explain why it cannot be briefed. */
|
||||
function loadPair(anchor: string): PairState | string {
|
||||
const zh = anchor.replace(/\.md$/, '.zh.md')
|
||||
const meta = anchor.replace(/\.md$/, '.i18n.yaml')
|
||||
if (!isTranslationScopeFile(anchor) || isExcluded(anchor)) {
|
||||
return `${anchor}: not an in-scope documentation pair (docs/i18n/README.md)`
|
||||
}
|
||||
const missing = [anchor, zh, meta].filter(file => !existsSync(join(root, file)))
|
||||
if (missing.length > 0) {
|
||||
return `${anchor}: incomplete pair (missing ${missing.join(', ')}) — a new counterpart is whole-document translation work, not a minimal update`
|
||||
}
|
||||
const record = parseMeta(readFileSync(join(root, meta), 'utf8'))
|
||||
const enRecorded = record?.get(basename(anchor))
|
||||
const zhRecorded = record?.get(basename(zh))
|
||||
if (record === undefined || enRecorded === undefined || zhRecorded === undefined) {
|
||||
return `${meta}: malformed consistency record`
|
||||
}
|
||||
const enCurrent = readFileSync(join(root, anchor), 'utf8')
|
||||
const zhCurrent = readFileSync(join(root, zh), 'utf8')
|
||||
const enLast = blobText(enRecorded)
|
||||
const zhLast = blobText(zhRecorded)
|
||||
return {
|
||||
anchor,
|
||||
zh,
|
||||
meta,
|
||||
enDrifted: enCurrent !== enLast,
|
||||
zhDrifted: zhCurrent !== zhLast,
|
||||
enLast,
|
||||
zhLast,
|
||||
}
|
||||
}
|
||||
|
||||
/** Assemble bundles for the given changed + first-occurrence span indices. */
|
||||
function bundlesFor(
|
||||
indices: number[],
|
||||
extraIndices: number[],
|
||||
confirmed: MarkdownSpan[],
|
||||
current: MarkdownSpan[],
|
||||
counterpart: MarkdownSpan[],
|
||||
): BriefBundle[] {
|
||||
const extras = new Set(extraIndices)
|
||||
return [...new Set([...indices, ...extraIndices])].sort((left, right) => left - right).map((index) => {
|
||||
const confirmedSpan = confirmed[index]
|
||||
const currentSpan = current[index]
|
||||
const counterpartSpan = counterpart[index]
|
||||
if (confirmedSpan === undefined || currentSpan === undefined || counterpartSpan === undefined) {
|
||||
throw new Error(`gen-translation-brief: span ${index} is unmapped despite alignment`)
|
||||
}
|
||||
return {
|
||||
index,
|
||||
label: currentSpan.label,
|
||||
reason: extras.has(index) && confirmedSpan.text === currentSpan.text ? 'first-occurrence' as const : undefined,
|
||||
confirmedSourceText: confirmedSpan.text,
|
||||
currentSourceText: currentSpan.text,
|
||||
counterpartText: counterpartSpan.text,
|
||||
counterpartStartLine: counterpartSpan.startLine,
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
interface PlannedBrief {
|
||||
scope: BriefScope
|
||||
/** Old + new text of the changed spans, for terminology matching. */
|
||||
changedText: string
|
||||
/** Computed counterpart for a mechanical scope, for `--apply`. */
|
||||
mechanicalResult?: string | undefined
|
||||
}
|
||||
|
||||
/** Choose the narrowest safely mapped granularity for one drifted side. */
|
||||
function planScope(
|
||||
sourceLast: string,
|
||||
sourceCurrent: string,
|
||||
counterpartCurrent: string,
|
||||
direction: BriefDirection,
|
||||
bothDrifted: boolean,
|
||||
): PlannedBrief {
|
||||
const wholeChangedText = `${sourceLast}\n${sourceCurrent}`
|
||||
if (bothDrifted) {
|
||||
return {
|
||||
scope: { kind: 'document', reason: 'BOTH sides changed since the pair was last confirmed consistent, so no side is a trustworthy mapping anchor; decide which side owns each divergence.' },
|
||||
changedText: wholeChangedText,
|
||||
}
|
||||
}
|
||||
const mechanical = computeMechanicalUpdate(sourceLast, sourceCurrent, counterpartCurrent)
|
||||
if (mechanical !== undefined) {
|
||||
return { scope: { kind: 'mechanical' }, changedText: wholeChangedText, mechanicalResult: mechanical }
|
||||
}
|
||||
for (const [kind, spansOf] of [['units', markdownUnits], ['sections', sectionSpans]] as const) {
|
||||
const confirmed = spansOf(sourceLast)
|
||||
const current = spansOf(sourceCurrent)
|
||||
const counterpart = spansOf(counterpartCurrent)
|
||||
if (!spansAligned(confirmed, current) || !spansAligned(confirmed, counterpart)) continue
|
||||
const changed = changedSpanIndices(confirmed, current)
|
||||
if (changed.length === 0) continue
|
||||
const changedText = changed.map(index => `${confirmed[index]?.text ?? ''}\n${current[index]?.text ?? ''}`).join('\n')
|
||||
const rows = relevantTerminologyRows(terminology, direction, changedText)
|
||||
const occurrence = direction === 'en-to-zh'
|
||||
? firstOccurrenceContext(sourceLast, sourceCurrent, confirmed, current, rows, new Set(changed))
|
||||
: { notes: [], extraSpanIndices: [] }
|
||||
return {
|
||||
scope: {
|
||||
kind,
|
||||
bundles: bundlesFor(changed, occurrence.extraSpanIndices, confirmed, current, counterpart),
|
||||
firstOccurrenceNotes: occurrence.notes,
|
||||
},
|
||||
changedText,
|
||||
}
|
||||
}
|
||||
return {
|
||||
scope: { kind: 'document', reason: 'Neither fine-grained units nor heading sections align one to one across the last-confirmed source, current source, and current counterpart.' },
|
||||
changedText: wholeChangedText,
|
||||
}
|
||||
}
|
||||
|
||||
/** Validate a computed mechanical counterpart and write it. */
|
||||
function applyMechanical(counterpartPath: string, sourceCurrent: string, result: string): void {
|
||||
const counterpartBase = basename(counterpartPath)
|
||||
const sourceBase = counterpartBase.endsWith('.zh.md')
|
||||
? counterpartBase.replace(/\.zh\.md$/, '.md')
|
||||
: counterpartBase.replace(/\.md$/, '.zh.md')
|
||||
const errors = translationStructureDiff(
|
||||
translationStructureSignature(parseTranslationMarkdown(sourceCurrent), counterpartBase),
|
||||
translationStructureSignature(parseTranslationMarkdown(result), sourceBase),
|
||||
)
|
||||
if (errors.length > 0) {
|
||||
throw new Error(`gen-translation-brief: computed mechanical update for ${counterpartPath} violates the pair structure: ${errors.join('; ')}`)
|
||||
}
|
||||
writeFileSync(join(root, counterpartPath), result)
|
||||
console.error(`gen-translation-brief: applied code-fence splice to ${counterpartPath}; review the diff, then record the pair.`)
|
||||
}
|
||||
|
||||
/** Render (and under `--apply`, apply) the briefing for one drifted side. */
|
||||
function briefDirection(pair: PairState, direction: BriefDirection, apply: boolean): string {
|
||||
const sourceIsEnglish = direction === 'en-to-zh'
|
||||
const sourcePath = sourceIsEnglish ? pair.anchor : pair.zh
|
||||
const counterpartPath = sourceIsEnglish ? pair.zh : pair.anchor
|
||||
const sourceLast = sourceIsEnglish ? pair.enLast : pair.zhLast
|
||||
const sourceCurrent = readFileSync(join(root, sourcePath), 'utf8')
|
||||
const counterpartCurrent = readFileSync(join(root, counterpartPath), 'utf8')
|
||||
const diff = diffTexts(sourceLast, sourceCurrent)
|
||||
const planned = planScope(sourceLast, sourceCurrent, counterpartCurrent, direction, pair.enDrifted && pair.zhDrifted)
|
||||
if (apply && planned.mechanicalResult !== undefined) {
|
||||
applyMechanical(counterpartPath, sourceCurrent, planned.mechanicalResult)
|
||||
}
|
||||
return renderTranslationBrief({
|
||||
sourcePath,
|
||||
counterpartPath,
|
||||
direction,
|
||||
diff,
|
||||
scope: planned.scope,
|
||||
terminology: relevantTerminologyRows(terminology, direction, planned.changedText),
|
||||
})
|
||||
}
|
||||
|
||||
const argv = process.argv.slice(2)
|
||||
const flags = argv.filter(argument => argument.startsWith('--'))
|
||||
const unknownFlags = flags.filter(flag => flag !== '--apply')
|
||||
if (unknownFlags.length > 0) {
|
||||
console.error(`gen-translation-brief: unknown flag(s): ${unknownFlags.join(', ')} (only --apply is supported)`)
|
||||
process.exit(2)
|
||||
}
|
||||
const applyMode = flags.includes('--apply')
|
||||
const requested = argv.filter(argument => !argument.startsWith('--')).map(pairAnchorOfArgument)
|
||||
|
||||
let anchors: string[]
|
||||
if (requested.length > 0) {
|
||||
anchors = [...new Set(requested)].sort()
|
||||
} else {
|
||||
const discovered = new Set<string>()
|
||||
for (const match of globSync('**/*.i18n.yaml', { cwd: root, exclude: TRANSLATION_SCOPE_GLOB_EXCLUDES })) {
|
||||
const normalized = match.split(sep).join('/')
|
||||
if (isTranslationScopeFile(normalized)) discovered.add(normalized.replace(/\.i18n\.yaml$/, '.md'))
|
||||
}
|
||||
anchors = [...discovered].sort()
|
||||
}
|
||||
|
||||
const briefs: string[] = []
|
||||
const problems: string[] = []
|
||||
const skipped: string[] = []
|
||||
for (const anchor of anchors) {
|
||||
const pair = loadPair(anchor)
|
||||
if (typeof pair === 'string') {
|
||||
if (requested.length > 0) problems.push(pair)
|
||||
continue
|
||||
}
|
||||
if (!pair.enDrifted && !pair.zhDrifted) {
|
||||
if (requested.length > 0) skipped.push(`${anchor}: pair is consistent with its record — nothing to brief`)
|
||||
continue
|
||||
}
|
||||
if (pair.enDrifted) briefs.push(briefDirection(pair, 'en-to-zh', applyMode))
|
||||
if (pair.zhDrifted) briefs.push(briefDirection(pair, 'zh-to-en', applyMode))
|
||||
}
|
||||
|
||||
if (problems.length > 0 || skipped.length > 0) {
|
||||
for (const message of [...problems, ...skipped]) console.error(`gen-translation-brief: ${message}`)
|
||||
process.exit(2)
|
||||
}
|
||||
if (briefs.length === 0) {
|
||||
console.log('gen-translation-brief: every recorded pair matches its consistency record; nothing to brief.')
|
||||
process.exit(0)
|
||||
}
|
||||
console.log(briefs.join('\n\n---\n\n'))
|
||||
@@ -19,7 +19,7 @@ export function rawJsDoc(text: string, node: ts.Node): string {
|
||||
}
|
||||
|
||||
/** A dispatch mode, rendered as the badge after an event name in the catalog. */
|
||||
export type Mode = 'emit' | 'waterfall' | 'parallel' | 'serial'
|
||||
export type Mode = 'emit' | 'waterfall' | 'parallel' | 'serial' | 'bail'
|
||||
|
||||
/**
|
||||
* Parse a raw JSDoc block into description prose and an optional `@mode`. Prose
|
||||
@@ -59,7 +59,7 @@ export function parseJsDoc(raw: string): { doc: string; mode: Mode | null; hasMo
|
||||
}
|
||||
for (const line of inner) {
|
||||
const tagLine = line.trimStart()
|
||||
const m = /^@mode\s+(emit|waterfall|parallel|serial)\s*$/.exec(tagLine)
|
||||
const m = /^@mode\s+(emit|waterfall|parallel|serial|bail)\s*$/.exec(tagLine)
|
||||
if (m) { mode = m[1] as Mode; hasMode = true; flushPara(); inTags = true; continue }
|
||||
if (/^@mode\b/.test(tagLine)) { hasMode = true; flushPara(); inTags = true; continue }
|
||||
if (tagLine.startsWith('@')) { flushPara(); inTags = true; continue }
|
||||
|
||||
File diff suppressed because one or more lines are too long
283
scripts/translation-brief.spec.ts
Normal file
283
scripts/translation-brief.spec.ts
Normal file
@@ -0,0 +1,283 @@
|
||||
/** Regression tests for the minimal-update briefing assembly. */
|
||||
|
||||
import { describe, expect, it } from 'vitest'
|
||||
import {
|
||||
changedSpanIndices,
|
||||
computeMechanicalUpdate,
|
||||
firstOccurrenceContext,
|
||||
markdownUnits,
|
||||
parseTerminologyRows,
|
||||
relevantTerminologyRows,
|
||||
renderTranslationBrief,
|
||||
sectionSpans,
|
||||
spansAligned,
|
||||
termOffsets,
|
||||
} from './translation-brief.ts'
|
||||
|
||||
const DOC = [
|
||||
'Preamble line.',
|
||||
'',
|
||||
'# Title',
|
||||
'',
|
||||
'Intro paragraph.',
|
||||
'',
|
||||
'## First',
|
||||
'',
|
||||
'First body.',
|
||||
'',
|
||||
'```ts',
|
||||
'const value = 1',
|
||||
'```',
|
||||
'',
|
||||
'## Second',
|
||||
'',
|
||||
'| A | B |',
|
||||
'|---|---|',
|
||||
'| 1 | 2 |',
|
||||
'',
|
||||
'- item one',
|
||||
'- item two',
|
||||
].join('\n')
|
||||
|
||||
describe('markdown spans', () => {
|
||||
it('lists units with container-scoped kinds in document order', () => {
|
||||
const kinds = markdownUnits(DOC).map(span => span.kind)
|
||||
expect(kinds).toEqual([
|
||||
'root.0:paragraph',
|
||||
'root.1:heading:1',
|
||||
'root.2:paragraph',
|
||||
'root.3:heading:2',
|
||||
'root.4:paragraph',
|
||||
'root.5:code',
|
||||
'root.6:heading:2',
|
||||
'root.7.0:tableRow',
|
||||
'root.7.1:tableRow',
|
||||
'root.8.0:listItem',
|
||||
'root.8.1:listItem',
|
||||
])
|
||||
})
|
||||
|
||||
it('lists heading sections with a preamble span and heading labels', () => {
|
||||
const sections = sectionSpans(DOC)
|
||||
expect(sections.map(span => span.label)).toEqual([
|
||||
'(preamble before the first heading)',
|
||||
'Title',
|
||||
'First',
|
||||
'Second',
|
||||
])
|
||||
expect(sections[0]).toMatchObject({ startLine: 1, endLine: 2 })
|
||||
expect(sections[2]).toMatchObject({ startLine: 7, endLine: 14 })
|
||||
})
|
||||
|
||||
it('labels units by their node type', () => {
|
||||
const units = markdownUnits(DOC)
|
||||
expect(units[0]!.label).toBe('paragraph')
|
||||
expect(units[1]!.label).toBe('heading')
|
||||
expect(units[7]!.label).toBe('tableRow')
|
||||
})
|
||||
|
||||
it('aligns sections by depth only, so translated heading text still maps', () => {
|
||||
const zh = DOC.replace('## First', '## 第一节').replace('## Second', '## 第二节').replace('# Title', '# 标题')
|
||||
expect(spansAligned(sectionSpans(DOC), sectionSpans(zh))).toBe(true)
|
||||
})
|
||||
|
||||
it('aligns span lists only on equal non-empty kind sequences', () => {
|
||||
const zh = DOC.replace('First body.', '第一段。').replace('item one', '第一项').replace('Intro paragraph.', '导语。')
|
||||
expect(spansAligned(markdownUnits(DOC), markdownUnits(zh))).toBe(true)
|
||||
const reshaped = DOC.replace('- item one\n- item two', 'merged paragraph')
|
||||
expect(spansAligned(markdownUnits(DOC), markdownUnits(reshaped))).toBe(false)
|
||||
expect(spansAligned([], [])).toBe(false)
|
||||
})
|
||||
|
||||
it('reports the indices whose text changed', () => {
|
||||
const edited = DOC.replace('First body.', 'First body, revised.').replace('| 1 | 2 |', '| 1 | 3 |')
|
||||
expect(changedSpanIndices(markdownUnits(DOC), markdownUnits(edited))).toEqual([4, 8])
|
||||
})
|
||||
})
|
||||
|
||||
describe('mechanical code updates', () => {
|
||||
const en = '# T\n\nProse.\n\n```sh\nrun one\n```\n'
|
||||
const zh = '# T\n\n中文。\n\n```sh\nrun one\n```\n'
|
||||
|
||||
it('splices a fence-only edit into the counterpart', () => {
|
||||
const edited = en.replace('run one', 'run two')
|
||||
expect(computeMechanicalUpdate(en, edited, zh)).toBe(zh.replace('run one', 'run two'))
|
||||
})
|
||||
|
||||
it('refuses when prose changed too', () => {
|
||||
const edited = en.replace('Prose.', 'Prose!').replace('run one', 'run two')
|
||||
expect(computeMechanicalUpdate(en, edited, zh)).toBeUndefined()
|
||||
})
|
||||
|
||||
it('refuses when the counterpart fences already diverge from last-confirmed', () => {
|
||||
const edited = en.replace('run one', 'run two')
|
||||
expect(computeMechanicalUpdate(en, edited, zh.replace('run one', 'run stale'))).toBeUndefined()
|
||||
})
|
||||
|
||||
it('refuses when fence counts differ or nothing changed', () => {
|
||||
expect(computeMechanicalUpdate(en, `${en}\n\`\`\`sh\nextra\n\`\`\`\n`, zh)).toBeUndefined()
|
||||
expect(computeMechanicalUpdate(en, en, zh)).toBeUndefined()
|
||||
})
|
||||
})
|
||||
|
||||
const TERMINOLOGY = [
|
||||
'| English | 中文 | 首次出现 | 不要译作 | 备注 |',
|
||||
'|---|---|---|---|---|',
|
||||
'| agent | agent | agent(智能体) | 智能体 | |',
|
||||
'| session log | 会话日志 | | 会话记录 | |',
|
||||
'| gate | 门禁 | | | |',
|
||||
'| registry | 注册表 | | | |',
|
||||
].join('\n')
|
||||
|
||||
describe('terminology', () => {
|
||||
it('parses data rows and skips the header and separator', () => {
|
||||
const rows = parseTerminologyRows(TERMINOLOGY)
|
||||
expect(rows.map(row => row.english)).toEqual(['agent', 'session log', 'gate', 'registry'])
|
||||
expect(rows[0]).toMatchObject({ chinese: 'agent', first: 'agent(智能体)' })
|
||||
})
|
||||
|
||||
it('matches English terms on word boundaries with plural inflections', () => {
|
||||
expect(termOffsets('two agents met', 'agent', true)).toEqual([4])
|
||||
expect(termOffsets('two registries', 'registry', true)).toEqual([4])
|
||||
expect(termOffsets('reagents', 'agent', true)).toEqual([])
|
||||
expect(termOffsets('', 'agent', true)).toEqual([])
|
||||
})
|
||||
|
||||
it('selects rows for the changed text per direction', () => {
|
||||
expect(relevantTerminologyRows(TERMINOLOGY, 'en-to-zh', 'All agents write a session log.').map(row => row.english))
|
||||
.toEqual(['agent', 'session log'])
|
||||
expect(relevantTerminologyRows(TERMINOLOGY, 'zh-to-en', '门禁在提交时运行。').map(row => row.english))
|
||||
.toEqual(['gate'])
|
||||
expect(relevantTerminologyRows(TERMINOLOGY, 'en-to-zh', 'delegate the work')).toEqual([])
|
||||
})
|
||||
})
|
||||
|
||||
describe('first-occurrence tracking', () => {
|
||||
const before = '# T\n\nAlpha paragraph.\n\nThe agent runs.\n'
|
||||
const after = '# T\n\nAlpha paragraph with an agent.\n\nThe agent runs.\n'
|
||||
const rows = parseTerminologyRows(TERMINOLOGY).filter(row => row.english === 'agent')
|
||||
|
||||
it('flags a moved first occurrence and pulls the vacated span in', () => {
|
||||
const context = firstOccurrenceContext(
|
||||
before, after, markdownUnits(before), markdownUnits(after), rows, new Set([1]),
|
||||
)
|
||||
expect(context.notes).toHaveLength(1)
|
||||
expect(context.notes[0]).toContain('moved from #2 to #1')
|
||||
expect(context.extraSpanIndices).toEqual([2])
|
||||
})
|
||||
|
||||
it('stays silent when the first occurrence does not move', () => {
|
||||
const unmoved = before.replace('Alpha paragraph.', 'Alpha paragraph, revised.')
|
||||
const context = firstOccurrenceContext(
|
||||
before, unmoved, markdownUnits(before), markdownUnits(unmoved), rows, new Set([1]),
|
||||
)
|
||||
expect(context.notes).toEqual([])
|
||||
expect(context.extraSpanIndices).toEqual([])
|
||||
})
|
||||
|
||||
it('ignores rows without a first-occurrence rendering', () => {
|
||||
const bare = parseTerminologyRows(TERMINOLOGY).filter(row => row.english === 'gate')
|
||||
const withGate = after.replace('The agent runs.', 'The gate runs.')
|
||||
const context = firstOccurrenceContext(
|
||||
before, withGate, markdownUnits(before), markdownUnits(withGate), bare, new Set([2]),
|
||||
)
|
||||
expect(context.notes).toEqual([])
|
||||
})
|
||||
})
|
||||
|
||||
describe('brief rendering', () => {
|
||||
const base = {
|
||||
sourcePath: 'docs/foo.md',
|
||||
counterpartPath: 'docs/foo.zh.md',
|
||||
direction: 'en-to-zh' as const,
|
||||
diff: '@@ -5 +5 @@\n-old text about the agent\n+new text about the agent',
|
||||
terminology: relevantTerminologyRows(TERMINOLOGY, 'en-to-zh', 'the agent'),
|
||||
}
|
||||
const bundle = {
|
||||
index: 4,
|
||||
label: 'paragraph',
|
||||
confirmedSourceText: 'old text about the agent\n',
|
||||
currentSourceText: 'new text about the agent\n',
|
||||
counterpartText: '关于 agent 的旧文本\n',
|
||||
counterpartStartLine: 9,
|
||||
}
|
||||
|
||||
it('renders unit bundles with three-way context and line anchors', () => {
|
||||
const brief = renderTranslationBrief({
|
||||
...base,
|
||||
scope: { kind: 'units', bundles: [bundle], firstOccurrenceNotes: ['agent: the document-wide first occurrence moved from #2 to #1; the agent(智能体) form moves with it (later occurrences drop the annotation).'] },
|
||||
})
|
||||
expect(brief).toContain('# Translation update briefing: docs/foo.md')
|
||||
expect(brief).toContain('## Changed units')
|
||||
expect(brief).toContain('### #4 paragraph — counterpart at docs/foo.zh.md:9')
|
||||
expect(brief).toContain('Last-confirmed English:')
|
||||
expect(brief).toContain('Current Chinese (bring this along):')
|
||||
expect(brief).toContain('## First-occurrence notes')
|
||||
expect(brief).toContain('agent(智能体)')
|
||||
expect(brief).toContain('首次出现 annotations attach to the document-wide first occurrence only')
|
||||
expect(brief).toContain('verify-translation-pairing --write docs/foo.md')
|
||||
})
|
||||
|
||||
it('marks first-occurrence bundles and omits their unchanged confirmed text', () => {
|
||||
const brief = renderTranslationBrief({
|
||||
...base,
|
||||
scope: {
|
||||
kind: 'units',
|
||||
bundles: [{ ...bundle, reason: 'first-occurrence', confirmedSourceText: bundle.currentSourceText }],
|
||||
firstOccurrenceNotes: [],
|
||||
},
|
||||
})
|
||||
expect(brief).toContain('unchanged; included for a first-occurrence move')
|
||||
expect(brief).not.toContain('Last-confirmed English:')
|
||||
})
|
||||
|
||||
it('renders the mechanical scope with the --apply command', () => {
|
||||
const brief = renderTranslationBrief({ ...base, scope: { kind: 'mechanical' } })
|
||||
expect(brief).toContain('## Mechanical update — no translation judgment involved')
|
||||
expect(brief).toContain('gen-translation-brief --apply docs/foo.md')
|
||||
expect(brief).not.toContain('## Changed units')
|
||||
})
|
||||
|
||||
it('renders the section fallback under its own heading', () => {
|
||||
const brief = renderTranslationBrief({
|
||||
...base,
|
||||
scope: { kind: 'sections', bundles: [bundle], firstOccurrenceNotes: [] },
|
||||
})
|
||||
expect(brief).toContain('## Changed sections')
|
||||
expect(brief).toContain('fine-grained units do not align')
|
||||
})
|
||||
|
||||
it('renders the document fallback with its reason and no bundles', () => {
|
||||
const brief = renderTranslationBrief({
|
||||
...base,
|
||||
scope: { kind: 'document', reason: 'BOTH sides changed since the pair was last confirmed consistent, so no side is a trustworthy mapping anchor; decide which side owns each divergence.' },
|
||||
})
|
||||
expect(brief).toContain('## Whole-document update required')
|
||||
expect(brief).toContain('BOTH sides changed')
|
||||
expect(brief).toContain('locate the affected regions yourself')
|
||||
})
|
||||
|
||||
it('renders the English-target digest for zh-to-en updates', () => {
|
||||
const brief = renderTranslationBrief({
|
||||
...base,
|
||||
direction: 'zh-to-en',
|
||||
sourcePath: 'docs/foo.zh.md',
|
||||
counterpartPath: 'docs/foo.md',
|
||||
scope: { kind: 'units', bundles: [bundle], firstOccurrenceNotes: [] },
|
||||
})
|
||||
expect(brief).toContain('exactly what the new Chinese states')
|
||||
expect(brief).toContain('verify-translation-pairing --write docs/foo.md')
|
||||
})
|
||||
|
||||
it('grows bundle fences past tilde runs in the text', () => {
|
||||
const brief = renderTranslationBrief({
|
||||
...base,
|
||||
scope: {
|
||||
kind: 'units',
|
||||
bundles: [{ ...bundle, counterpartText: '~~~~\ninner\n~~~~\n' }],
|
||||
firstOccurrenceNotes: [],
|
||||
},
|
||||
})
|
||||
expect(brief).toContain('~~~~~markdown')
|
||||
})
|
||||
})
|
||||
513
scripts/translation-brief.ts
Normal file
513
scripts/translation-brief.ts
Normal file
@@ -0,0 +1,513 @@
|
||||
/**
|
||||
* Pure assembly of the minimal-update briefing for one out-of-sync
|
||||
* translation pair: the authored side's changes since the last confirmed
|
||||
* state at the narrowest safely mapped granularity (code-fence-only splice,
|
||||
* changed Markdown units, heading sections, whole document), the terminology
|
||||
* rows those changes touch, first-occurrence movement notes, and a digest of
|
||||
* the binding update rules. The unit mapping, mechanical code splice, and
|
||||
* first-occurrence tracking adopt the planner mechanics validated in the
|
||||
* incremental-pipeline work (PR #684). The CLI wrapper is
|
||||
* `scripts/gen-translation-brief.ts`; the workflow that consumes the
|
||||
* briefing is `.agents/skills/dsh-translate-docs/SKILL.md`.
|
||||
*/
|
||||
|
||||
import type { Nodes } from 'mdast'
|
||||
import { parseTranslationMarkdown } from './translation-pairing.ts'
|
||||
|
||||
/** One block-level span of a Markdown document, in document order. */
|
||||
export interface MarkdownSpan {
|
||||
/** Position in the span list; briefing ids derive from it. */
|
||||
index: number
|
||||
/**
|
||||
* Structural kind compared for alignment, language-neutral: container path
|
||||
* plus node type for units (`root.3:tableRow`), depth for sections (`section:2`).
|
||||
*/
|
||||
kind: string
|
||||
/** Reader-facing label: heading text for sections, node type for units. */
|
||||
label: string
|
||||
/** 1-based first source line. */
|
||||
startLine: number
|
||||
/** 1-based last source line. */
|
||||
endLine: number
|
||||
/** The span's text, trailing newline normalized to exactly one. */
|
||||
text: string
|
||||
}
|
||||
|
||||
function linesOf(markdown: string): string[] {
|
||||
const lines = markdown.replaceAll('\r\n', '\n').split('\n')
|
||||
if (lines.at(-1) === '') lines.pop()
|
||||
return lines
|
||||
}
|
||||
|
||||
function sliceLines(lines: string[], startLine: number, endLine: number): string {
|
||||
return `${lines.slice(startLine - 1, endLine).join('\n')}\n`
|
||||
}
|
||||
|
||||
/**
|
||||
* List a document's translation units: the outermost block nodes a minimal
|
||||
* update can replace independently. Headings, paragraphs, code fences, table
|
||||
* rows, list items, block quotes, HTML blocks, thematic breaks, and link
|
||||
* definitions are units; the container path is part of the kind so kind
|
||||
* sequences only align when container membership also aligns.
|
||||
*
|
||||
* @param markdown - Document text.
|
||||
* @returns Units in document order.
|
||||
*/
|
||||
export function markdownUnits(markdown: string): MarkdownSpan[] {
|
||||
const positions: Array<{ kind: string; label: string; startLine: number; endLine: number }> = []
|
||||
const visit = (node: Nodes, path: string): void => {
|
||||
let kind: string | undefined
|
||||
switch (node.type) {
|
||||
case 'heading':
|
||||
kind = `${path}:heading:${node.depth}`
|
||||
break
|
||||
case 'paragraph':
|
||||
case 'code':
|
||||
case 'tableRow':
|
||||
case 'listItem':
|
||||
case 'blockquote':
|
||||
case 'html':
|
||||
case 'thematicBreak':
|
||||
case 'definition':
|
||||
kind = `${path}:${node.type}`
|
||||
break
|
||||
default:
|
||||
break
|
||||
}
|
||||
if (kind !== undefined && node.position !== undefined) {
|
||||
positions.push({ kind, label: node.type, startLine: node.position.start.line, endLine: node.position.end.line })
|
||||
return
|
||||
}
|
||||
if ('children' in node) for (const [index, child] of node.children.entries()) visit(child, `${path}.${index}`)
|
||||
}
|
||||
visit(parseTranslationMarkdown(markdown), 'root')
|
||||
positions.sort((left, right) => left.startLine - right.startLine)
|
||||
const lines = linesOf(markdown)
|
||||
return positions.map((position, index) => ({
|
||||
index,
|
||||
...position,
|
||||
text: sliceLines(lines, position.startLine, position.endLine),
|
||||
}))
|
||||
}
|
||||
|
||||
/**
|
||||
* List a document's heading-delimited sections, including a leading
|
||||
* `preamble` span when content precedes the first heading.
|
||||
*
|
||||
* @param markdown - Document text.
|
||||
* @returns Sections in document order.
|
||||
*/
|
||||
export function sectionSpans(markdown: string): MarkdownSpan[] {
|
||||
const headings: Array<{ depth: number; line: number; label: string }> = []
|
||||
const visit = (node: Nodes): void => {
|
||||
if (node.type === 'heading' && node.position !== undefined) {
|
||||
let label = ''
|
||||
const collect = (child: Nodes): void => {
|
||||
if ('value' in child && typeof child.value === 'string') label += child.value
|
||||
if ('children' in child) for (const grandchild of child.children) collect(grandchild)
|
||||
}
|
||||
for (const child of node.children) collect(child)
|
||||
headings.push({ depth: node.depth, line: node.position.start.line, label })
|
||||
}
|
||||
if ('children' in node) for (const child of node.children) visit(child)
|
||||
}
|
||||
visit(parseTranslationMarkdown(markdown))
|
||||
headings.sort((left, right) => left.line - right.line)
|
||||
const lines = linesOf(markdown)
|
||||
const spans: MarkdownSpan[] = []
|
||||
const firstHeadingLine = headings[0]?.line ?? lines.length + 1
|
||||
if (firstHeadingLine > 1) {
|
||||
spans.push({ index: 0, kind: 'preamble', label: '(preamble before the first heading)', startLine: 1, endLine: firstHeadingLine - 1, text: sliceLines(lines, 1, firstHeadingLine - 1) })
|
||||
}
|
||||
for (const [order, heading] of headings.entries()) {
|
||||
const endLine = (headings[order + 1]?.line ?? lines.length + 1) - 1
|
||||
spans.push({
|
||||
index: spans.length,
|
||||
// Depth only: heading TEXT is translated across a pair, so it cannot
|
||||
// participate in cross-language alignment.
|
||||
kind: `section:${heading.depth}`,
|
||||
label: heading.label === '' ? '(untitled section)' : heading.label,
|
||||
startLine: heading.line,
|
||||
endLine,
|
||||
text: sliceLines(lines, heading.line, endLine),
|
||||
})
|
||||
}
|
||||
return spans
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether two span lists map one to one: same non-zero length and the same
|
||||
* kind at every position.
|
||||
*
|
||||
* @param left - One document's spans.
|
||||
* @param right - The other document's spans.
|
||||
* @returns True when index-wise mapping is sound.
|
||||
*/
|
||||
export function spansAligned(left: MarkdownSpan[], right: MarkdownSpan[]): boolean {
|
||||
return left.length > 0
|
||||
&& left.length === right.length
|
||||
&& left.every((span, index) => span.kind === right[index]?.kind)
|
||||
}
|
||||
|
||||
/**
|
||||
* Indices whose text differs between two aligned span lists.
|
||||
*
|
||||
* @param before - Spans of the earlier state.
|
||||
* @param after - Spans of the later state, aligned with `before`.
|
||||
* @returns Ascending changed indices.
|
||||
*/
|
||||
export function changedSpanIndices(before: MarkdownSpan[], after: MarkdownSpan[]): number[] {
|
||||
return before.filter((span, index) => span.text !== after[index]?.text).map(span => span.index)
|
||||
}
|
||||
|
||||
function codeSpansOf(markdown: string): MarkdownSpan[] {
|
||||
return markdownUnits(markdown).filter(span => span.kind.endsWith(':code'))
|
||||
.map((span, index) => ({ ...span, index }))
|
||||
}
|
||||
|
||||
function replaceSpanTexts(markdown: string, spans: MarkdownSpan[], replacements: Map<number, string>): string {
|
||||
const lines = linesOf(markdown)
|
||||
for (const [index, replacement] of [...replacements.entries()].sort((left, right) => right[0] - left[0])) {
|
||||
const span = spans[index]
|
||||
if (span === undefined) throw new Error(`translation brief: unknown replacement span ${index}`)
|
||||
lines.splice(span.startLine - 1, span.endLine - span.startLine + 1, ...linesOf(replacement))
|
||||
}
|
||||
return `${lines.join('\n')}\n`
|
||||
}
|
||||
|
||||
function maskCodeSpans(markdown: string, spans: MarkdownSpan[]): string {
|
||||
return replaceSpanTexts(markdown, spans, new Map(spans.map(span => [span.index, `DSH_TRANSLATION_CODE_${span.index}\n`])))
|
||||
}
|
||||
|
||||
/**
|
||||
* Compute the counterpart update for a change confined to fenced code
|
||||
* blocks. Fences are byte-identical across a pair, so when the source's
|
||||
* prose is untouched and the counterpart's fences match the last-confirmed
|
||||
* source, splicing the edited fences into the counterpart is the complete
|
||||
* update — no translation judgment is involved.
|
||||
*
|
||||
* @param confirmedSource - The changed side's last-confirmed text.
|
||||
* @param currentSource - The changed side's current text.
|
||||
* @param counterpart - The other side's current text.
|
||||
* @returns The updated counterpart, or undefined when the change is not code-only.
|
||||
*/
|
||||
export function computeMechanicalUpdate(confirmedSource: string, currentSource: string, counterpart: string): string | undefined {
|
||||
const confirmed = codeSpansOf(confirmedSource)
|
||||
const current = codeSpansOf(currentSource)
|
||||
const target = codeSpansOf(counterpart)
|
||||
if (confirmed.length === 0 || confirmed.length !== current.length || confirmed.length !== target.length) return undefined
|
||||
if (maskCodeSpans(confirmedSource, confirmed) !== maskCodeSpans(currentSource, current)) return undefined
|
||||
if (confirmed.some((span, index) => span.text !== target[index]?.text)) return undefined
|
||||
const changed = current.filter((span, index) => span.text !== confirmed[index]?.text)
|
||||
if (changed.length === 0) return undefined
|
||||
return replaceSpanTexts(counterpart, target, new Map(changed.map(span => [span.index, span.text])))
|
||||
}
|
||||
|
||||
/** One parsed terminology-table data row. */
|
||||
export interface TerminologyRow {
|
||||
english: string
|
||||
chinese: string
|
||||
/** The 首次出现 cell (first-occurrence rendering), possibly empty. */
|
||||
first: string
|
||||
/** The verbatim table row. */
|
||||
line: string
|
||||
}
|
||||
|
||||
/** Strip Markdown emphasis and code markers from a terminology cell. */
|
||||
function plainTerm(cell: string): string {
|
||||
return cell.replaceAll('`', '').replaceAll('**', '').trim()
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse the data rows of the terminology table.
|
||||
*
|
||||
* @param terminology - Full `docs/i18n/terminology.md` contents.
|
||||
* @returns Rows in table order.
|
||||
*/
|
||||
export function parseTerminologyRows(terminology: string): TerminologyRow[] {
|
||||
const rows: TerminologyRow[] = []
|
||||
for (const line of terminology.split('\n')) {
|
||||
if (!line.startsWith('|')) continue
|
||||
if (/^\|[\s:|-]+\|$/.test(line)) continue
|
||||
const cells = line.split('|').map(cell => cell.trim())
|
||||
const english = plainTerm(cells[1] ?? '')
|
||||
if (english === '' || english === 'English') continue
|
||||
rows.push({ english, chinese: plainTerm(cells[2] ?? ''), first: plainTerm(cells[3] ?? ''), line })
|
||||
}
|
||||
return rows
|
||||
}
|
||||
|
||||
/**
|
||||
* Character offsets of a term's occurrences. English word-like terms match
|
||||
* on word boundaries and accept plural inflections (`agents`, `registries`);
|
||||
* other terms match as case-insensitive substrings.
|
||||
*
|
||||
* @param text - Text to search.
|
||||
* @param term - The term to find.
|
||||
* @param englishInflections - Whether to accept English plural forms.
|
||||
* @returns Ascending match offsets.
|
||||
*/
|
||||
export function termOffsets(text: string, term: string, englishInflections = false): number[] {
|
||||
if (term === '') return []
|
||||
const escape = (value: string): string => value.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')
|
||||
const wordLike = /^[A-Za-z0-9][A-Za-z0-9 ._-]*[A-Za-z0-9]$/.test(term)
|
||||
const inflected = englishInflections && wordLike
|
||||
? /[^aeiou]y$/i.test(term)
|
||||
? `${escape(term.slice(0, -1))}(?:y|ies)`
|
||||
: `${escape(term)}(?:s|es)?`
|
||||
: escape(term)
|
||||
const expression = new RegExp(wordLike ? `(?<![A-Za-z0-9_])${inflected}(?![A-Za-z0-9_])` : inflected, 'gi')
|
||||
return [...text.matchAll(expression)].map(match => match.index)
|
||||
}
|
||||
|
||||
/** The two update directions a pair supports. */
|
||||
export type BriefDirection = 'en-to-zh' | 'zh-to-en'
|
||||
|
||||
/** Whether a row's source-language term occurs in the given text. */
|
||||
function rowOccurs(row: TerminologyRow, direction: BriefDirection, text: string): boolean {
|
||||
const terms = direction === 'en-to-zh' ? [row.english] : [row.first, row.chinese].filter(term => /[一-鿿]/.test(term))
|
||||
return terms.some(term => termOffsets(text, term, direction === 'en-to-zh').length > 0)
|
||||
}
|
||||
|
||||
/**
|
||||
* Select the terminology rows whose source-language term occurs in the
|
||||
* changed text (old and new states combined).
|
||||
*
|
||||
* @param terminology - Full `docs/i18n/terminology.md` contents.
|
||||
* @param direction - Update direction; decides which columns to match.
|
||||
* @param changedText - Concatenated old and new text of the changed spans.
|
||||
* @returns Matched rows in table order.
|
||||
*/
|
||||
export function relevantTerminologyRows(terminology: string, direction: BriefDirection, changedText: string): TerminologyRow[] {
|
||||
return parseTerminologyRows(terminology).filter(row => rowOccurs(row, direction, changedText))
|
||||
}
|
||||
|
||||
function lineAtOffset(text: string, offset: number): number {
|
||||
return text.slice(0, offset).split('\n').length
|
||||
}
|
||||
|
||||
function spanIndexAtOffset(text: string, spans: MarkdownSpan[], offset: number | undefined): number | undefined {
|
||||
if (offset === undefined) return undefined
|
||||
const line = lineAtOffset(text, offset)
|
||||
return spans.find(span => line >= span.startLine && line <= span.endLine)?.index
|
||||
}
|
||||
|
||||
/** First-occurrence guidance computed for a Chinese-target update. */
|
||||
export interface FirstOccurrenceContext {
|
||||
/** Human-readable notes for the briefing. */
|
||||
notes: string[]
|
||||
/** Unchanged span indices that must join the briefing because a first occurrence moved into or out of them. */
|
||||
extraSpanIndices: number[]
|
||||
}
|
||||
|
||||
/**
|
||||
* Track document-wide first occurrences of the relevant English terms. The
|
||||
* 首次出现 rendering attaches to a term's first occurrence, so when an edit
|
||||
* moves that occurrence across spans, both the old and new spans need
|
||||
* counterpart edits even when only one of them changed.
|
||||
*
|
||||
* @param confirmedSource - Last-confirmed English text.
|
||||
* @param currentSource - Current English text.
|
||||
* @param confirmedSpans - Spans of the last-confirmed English text.
|
||||
* @param currentSpans - Spans of the current English text, aligned with `confirmedSpans`.
|
||||
* @param rows - The relevant terminology rows.
|
||||
* @param changed - Span indices already in the briefing.
|
||||
* @returns Notes and extra span indices to include.
|
||||
*/
|
||||
export function firstOccurrenceContext(
|
||||
confirmedSource: string,
|
||||
currentSource: string,
|
||||
confirmedSpans: MarkdownSpan[],
|
||||
currentSpans: MarkdownSpan[],
|
||||
rows: TerminologyRow[],
|
||||
changed: Set<number>,
|
||||
): FirstOccurrenceContext {
|
||||
const notes: string[] = []
|
||||
const extra = new Set<number>()
|
||||
for (const row of rows) {
|
||||
if (row.first === '') continue
|
||||
const oldIndex = spanIndexAtOffset(confirmedSource, confirmedSpans, termOffsets(confirmedSource, row.english, true)[0])
|
||||
const newIndex = spanIndexAtOffset(currentSource, currentSpans, termOffsets(currentSource, row.english, true)[0])
|
||||
if (oldIndex === newIndex) continue
|
||||
for (const index of [oldIndex, newIndex]) {
|
||||
if (index !== undefined && !changed.has(index)) extra.add(index)
|
||||
}
|
||||
notes.push(`${row.english}: the document-wide first occurrence moved from ${oldIndex === undefined ? 'absent' : `#${oldIndex}`} to ${newIndex === undefined ? 'absent' : `#${newIndex}`}; the ${row.first} form moves with it (later occurrences drop the annotation).`)
|
||||
}
|
||||
return { notes, extraSpanIndices: [...extra].sort((left, right) => left - right) }
|
||||
}
|
||||
|
||||
/** Smallest fence of `mark` characters that safely wraps `body`. */
|
||||
function fenceFor(body: string, mark: '`' | '~'): string {
|
||||
let longest = 2
|
||||
for (const line of body.split('\n')) {
|
||||
const run = new RegExp(`^\\s*(${mark === '`' ? '`' : '~'}{3,})`).exec(line)
|
||||
if (run?.[1] !== undefined && run[1].length > longest) longest = run[1].length
|
||||
}
|
||||
return mark.repeat(longest + 1)
|
||||
}
|
||||
|
||||
/** One changed (or first-occurrence) span with its three-way context. */
|
||||
export interface BriefBundle {
|
||||
/** Span index shared by the aligned documents. */
|
||||
index: number
|
||||
/** Human label: heading text or node type. */
|
||||
label: string
|
||||
/** Why the bundle is present when its source text did not change. */
|
||||
reason?: 'first-occurrence' | undefined
|
||||
confirmedSourceText: string
|
||||
currentSourceText: string
|
||||
counterpartText: string
|
||||
/** 1-based line the counterpart span starts on. */
|
||||
counterpartStartLine: number
|
||||
}
|
||||
|
||||
/** The granularities a briefing can map the change at, narrowest first. */
|
||||
export type BriefScope =
|
||||
| { kind: 'mechanical' }
|
||||
| { kind: 'units'; bundles: BriefBundle[]; firstOccurrenceNotes: string[] }
|
||||
| { kind: 'sections'; bundles: BriefBundle[]; firstOccurrenceNotes: string[] }
|
||||
| { kind: 'document'; reason: string }
|
||||
|
||||
/** Inputs for rendering one pair's briefing. */
|
||||
export interface TranslationBriefInput {
|
||||
/** Repo-relative path of the side that changed. */
|
||||
sourcePath: string
|
||||
/** Repo-relative path of the counterpart to update. */
|
||||
counterpartPath: string
|
||||
direction: BriefDirection
|
||||
/** Unified diff of the changed side, last-confirmed to current. */
|
||||
diff: string
|
||||
scope: BriefScope
|
||||
terminology: TerminologyRow[]
|
||||
}
|
||||
|
||||
const ZH_TARGET_DIGEST = [
|
||||
'- Edit ONLY what the change requires; preserve the reviewed phrasing of everything unchanged.',
|
||||
'- Nothing added, nothing dropped: the Chinese must state exactly what the new English states.',
|
||||
'- Write natural institutional technical Chinese, not word-by-word gloss; terse stays terse.',
|
||||
'- Code fences byte-identical to the English side, comments included; inline code spans verbatim.',
|
||||
'- Relative links keep the `.md` target; only the switcher line links `.zh.md`.',
|
||||
'- Structure mirrors the counterpart: heading depths and order, list kinds and item counts, table rows and columns.',
|
||||
'- 首次出现 annotations attach to the document-wide first occurrence only; later occurrences use the bare form, and an empty 首次出现 cell means never gloss.',
|
||||
'- Typography: one half-width space between Chinese and Latin or digits; full-width punctuation in Chinese prose; 顿号 for enumerations; second person is 你.',
|
||||
'- One physical line per paragraph; exactly one trailing newline.',
|
||||
]
|
||||
|
||||
const EN_TARGET_DIGEST = [
|
||||
'- Edit ONLY what the change requires; preserve the reviewed phrasing of everything unchanged.',
|
||||
'- Nothing added, nothing dropped: the English must state exactly what the new Chinese states.',
|
||||
'- Write concise professional developer prose, not word-by-word gloss; terse stays terse.',
|
||||
'- Code fences byte-identical to the Chinese side, comments included; inline code spans verbatim.',
|
||||
'- Relative links keep the `.md` target; only the switcher line links `.zh.md`.',
|
||||
'- Structure mirrors the counterpart: heading depths and order, list kinds and item counts, table rows and columns.',
|
||||
'- One physical line per paragraph; exactly one trailing newline.',
|
||||
]
|
||||
|
||||
function renderBundles(out: string[], input: TranslationBriefInput, bundles: BriefBundle[], firstOccurrenceNotes: string[]): void {
|
||||
const sourceLanguage = input.direction === 'en-to-zh' ? 'English' : 'Chinese'
|
||||
const counterpartLanguage = input.direction === 'en-to-zh' ? 'Chinese' : 'English'
|
||||
for (const bundle of bundles) {
|
||||
out.push('')
|
||||
out.push(`### #${bundle.index} ${bundle.label}${bundle.reason === 'first-occurrence' ? ' — unchanged; included for a first-occurrence move' : ''} — counterpart at ${input.counterpartPath}:${bundle.counterpartStartLine}`)
|
||||
const fence = fenceFor([bundle.confirmedSourceText, bundle.currentSourceText, bundle.counterpartText].join('\n'), '~')
|
||||
if (bundle.confirmedSourceText !== bundle.currentSourceText) {
|
||||
out.push('')
|
||||
out.push(`Last-confirmed ${sourceLanguage}:`)
|
||||
out.push('')
|
||||
out.push(`${fence}markdown`)
|
||||
out.push(bundle.confirmedSourceText.trimEnd())
|
||||
out.push(fence)
|
||||
}
|
||||
out.push('')
|
||||
out.push(`Current ${sourceLanguage}:`)
|
||||
out.push('')
|
||||
out.push(`${fence}markdown`)
|
||||
out.push(bundle.currentSourceText.trimEnd())
|
||||
out.push(fence)
|
||||
out.push('')
|
||||
out.push(`Current ${counterpartLanguage} (bring this along):`)
|
||||
out.push('')
|
||||
out.push(`${fence}markdown`)
|
||||
out.push(bundle.counterpartText.trimEnd())
|
||||
out.push(fence)
|
||||
}
|
||||
if (firstOccurrenceNotes.length > 0) {
|
||||
out.push('')
|
||||
out.push('## First-occurrence notes')
|
||||
out.push('')
|
||||
for (const note of firstOccurrenceNotes) out.push(`- ${note}`)
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Render the complete briefing for one out-of-sync pair.
|
||||
*
|
||||
* @param input - Diff, mapped scope, terminology, and pair identity.
|
||||
* @returns Markdown briefing text.
|
||||
*/
|
||||
export function renderTranslationBrief(input: TranslationBriefInput): string {
|
||||
const sourceLanguage = input.direction === 'en-to-zh' ? 'English' : 'Chinese'
|
||||
const counterpartLanguage = input.direction === 'en-to-zh' ? 'Chinese' : 'English'
|
||||
const out: string[] = []
|
||||
out.push(`# Translation update briefing: ${input.sourcePath}`)
|
||||
out.push('')
|
||||
out.push(`The ${sourceLanguage} side changed; bring \`${input.counterpartPath}\` along with the smallest edit that covers the change.`)
|
||||
if (input.scope.kind === 'mechanical') {
|
||||
out.push('')
|
||||
out.push('## Mechanical update — no translation judgment involved')
|
||||
out.push('')
|
||||
out.push(`Every change since the last confirmed state is inside fenced code blocks, which are byte-identical across the pair. Run \`pnpm run gen-translation-brief --apply ${input.sourcePath}\` to splice the updated fences into the counterpart (the result is structure-validated before writing), then record per the Finish steps.`)
|
||||
}
|
||||
out.push('')
|
||||
out.push(`## ${sourceLanguage} diff (last-confirmed → current)`)
|
||||
out.push('')
|
||||
const diffFence = fenceFor(input.diff, '`')
|
||||
out.push(`${diffFence}diff`)
|
||||
out.push(input.diff.trimEnd())
|
||||
out.push(diffFence)
|
||||
switch (input.scope.kind) {
|
||||
case 'mechanical':
|
||||
break
|
||||
case 'units':
|
||||
out.push('')
|
||||
out.push(`## Changed units (last-confirmed ${sourceLanguage} → current ${sourceLanguage}, with the current ${counterpartLanguage})`)
|
||||
renderBundles(out, input, input.scope.bundles, input.scope.firstOccurrenceNotes)
|
||||
break
|
||||
case 'sections':
|
||||
out.push('')
|
||||
out.push('## Changed sections (fine-grained units do not align across the pair; whole heading sections shown)')
|
||||
renderBundles(out, input, input.scope.bundles, input.scope.firstOccurrenceNotes)
|
||||
break
|
||||
case 'document':
|
||||
out.push('')
|
||||
out.push('## Whole-document update required')
|
||||
out.push('')
|
||||
out.push(`${input.scope.reason} Open \`${input.counterpartPath}\` directly, locate the affected regions yourself, and reconcile under docs/i18n/translation-rules.md.`)
|
||||
break
|
||||
default:
|
||||
input.scope satisfies never
|
||||
}
|
||||
if (input.terminology.length > 0) {
|
||||
out.push('')
|
||||
out.push('## Binding terminology rows matching this change (docs/i18n/terminology.md)')
|
||||
out.push('')
|
||||
out.push('| English | 中文 | 首次出现 | 不要译作 | 备注 |')
|
||||
out.push('|---|---|---|---|---|')
|
||||
for (const row of input.terminology) out.push(row.line)
|
||||
out.push('')
|
||||
out.push('For any term you introduce that is not listed above, consult the full table before inventing a rendering.')
|
||||
}
|
||||
out.push('')
|
||||
out.push('## Rules digest (full rules: docs/i18n/translation-rules.md)')
|
||||
out.push('')
|
||||
out.push(...(input.direction === 'en-to-zh' ? ZH_TARGET_DIGEST : EN_TARGET_DIGEST))
|
||||
out.push('')
|
||||
out.push('## Finish')
|
||||
out.push('')
|
||||
out.push('1. Apply the smallest counterpart edit that covers the change, then verify the changed spans clause by clause against the source.')
|
||||
out.push(`2. \`pnpm run verify-translation-pairing --write ${input.sourcePath.replace(/\.zh\.md$/, '.md')}\``)
|
||||
out.push(`3. \`pnpm run verify-translation-pairing ${input.sourcePath.replace(/\.zh\.md$/, '.md')}\``)
|
||||
out.push('')
|
||||
return out.join('\n')
|
||||
}
|
||||
@@ -3,7 +3,9 @@
|
||||
import { describe, expect, it } from 'vitest'
|
||||
import {
|
||||
isTranslationScopeFile,
|
||||
pairAnchorOfArgument,
|
||||
parseTranslationMarkdown,
|
||||
parseTranslationPairingCliArgs,
|
||||
parseTranslationPairingManifest,
|
||||
translationStructureDiff,
|
||||
translationStructureSignature,
|
||||
@@ -102,3 +104,40 @@ describe('translation structural signature', () => {
|
||||
])
|
||||
})
|
||||
})
|
||||
|
||||
describe('pair CLI arguments', () => {
|
||||
it('normalizes any pair file or bare stem to the English anchor', () => {
|
||||
expect(pairAnchorOfArgument('docs/foo.md')).toBe('docs/foo.md')
|
||||
expect(pairAnchorOfArgument('docs/foo.zh.md')).toBe('docs/foo.md')
|
||||
expect(pairAnchorOfArgument('docs/foo.i18n.yaml')).toBe('docs/foo.md')
|
||||
expect(pairAnchorOfArgument('docs/foo')).toBe('docs/foo.md')
|
||||
expect(pairAnchorOfArgument('.\\docs\\foo.zh.md')).toBe('docs/foo.md')
|
||||
})
|
||||
|
||||
it('scopes a check to named pairs and dedupes the three spellings', () => {
|
||||
expect(parseTranslationPairingCliArgs(['docs/foo.zh.md', 'docs/foo.i18n.yaml', 'docs/bar.md'])).toEqual({
|
||||
mode: 'check',
|
||||
scope: 'pairs',
|
||||
anchors: ['docs/bar.md', 'docs/foo.md'],
|
||||
})
|
||||
expect(parseTranslationPairingCliArgs([])).toEqual({ mode: 'check', scope: 'corpus', anchors: [] })
|
||||
})
|
||||
|
||||
it('requires --write to name confirmed pairs or opt into --all', () => {
|
||||
expect(() => parseTranslationPairingCliArgs(['--write'])).toThrow('requires the pair(s) you confirmed')
|
||||
expect(parseTranslationPairingCliArgs(['--write', 'docs/foo.md'])).toEqual({
|
||||
mode: 'write',
|
||||
scope: 'pairs',
|
||||
anchors: ['docs/foo.md'],
|
||||
})
|
||||
expect(parseTranslationPairingCliArgs(['--write', '--all'])).toEqual({ mode: 'write', scope: 'corpus', anchors: [] })
|
||||
expect(() => parseTranslationPairingCliArgs(['--write', '--all', 'docs/foo.md'])).toThrow('not both')
|
||||
})
|
||||
|
||||
it('keeps --list corpus-only and rejects unknown flags', () => {
|
||||
expect(parseTranslationPairingCliArgs(['--list'])).toEqual({ mode: 'list', scope: 'corpus', anchors: [] })
|
||||
expect(() => parseTranslationPairingCliArgs(['--list', 'docs/foo.md'])).toThrow('takes no other flags or paths')
|
||||
expect(() => parseTranslationPairingCliArgs(['--all'])).toThrow('--all only applies to --write')
|
||||
expect(() => parseTranslationPairingCliArgs(['--frobnicate'])).toThrow('unknown flag(s): --frobnicate')
|
||||
})
|
||||
})
|
||||
|
||||
@@ -102,6 +102,66 @@ export function parseTranslationPairingManifest(content: string): TranslationPai
|
||||
return { excluded: excludedField(record) }
|
||||
}
|
||||
|
||||
/**
|
||||
* Normalize one CLI pair argument to its English anchor path: any of the
|
||||
* pair's three files (`foo.md`, `foo.zh.md`, `foo.i18n.yaml`) or the bare
|
||||
* `foo` stem names the same pair, and platform separators are accepted.
|
||||
*
|
||||
* @param argument - Repo-relative path as passed on a command line.
|
||||
* @returns The pair's `foo.md` anchor path with `/` separators.
|
||||
*/
|
||||
export function pairAnchorOfArgument(argument: string): string {
|
||||
const normalized = argument.split('\\').join('/').replace(/^\.\//, '')
|
||||
if (normalized.endsWith('.zh.md')) return `${normalized.slice(0, -'.zh.md'.length)}.md`
|
||||
if (normalized.endsWith('.i18n.yaml')) return `${normalized.slice(0, -'.i18n.yaml'.length)}.md`
|
||||
if (normalized.endsWith('.md')) return normalized
|
||||
return `${normalized}.md`
|
||||
}
|
||||
|
||||
/** A parsed `verify-translation-pairing` invocation. */
|
||||
export interface TranslationPairingCliRequest {
|
||||
mode: 'check' | 'list' | 'write'
|
||||
/** `corpus` runs discovery over the whole tree; `pairs` touches only the named anchors. */
|
||||
scope: 'corpus' | 'pairs'
|
||||
/** English anchor paths, empty for corpus scope. */
|
||||
anchors: string[]
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse and validate `verify-translation-pairing` CLI arguments.
|
||||
*
|
||||
* Check accepts optional pair paths; `--write` requires either pair paths or
|
||||
* `--all` so a bulk re-record is always an explicit choice — a bare
|
||||
* `--write` would silently bless every drifted pair in the tree, including
|
||||
* ones the caller never confirmed. `--list` is corpus-only.
|
||||
*
|
||||
* @param argv - Arguments after the script name.
|
||||
* @returns The validated request.
|
||||
* @throws Error when flags or their combination are invalid.
|
||||
*/
|
||||
export function parseTranslationPairingCliArgs(argv: string[]): TranslationPairingCliRequest {
|
||||
const flags = argv.filter(argument => argument.startsWith('--'))
|
||||
const anchors = [...new Set(argv.filter(argument => !argument.startsWith('--')).map(pairAnchorOfArgument))].sort()
|
||||
const unknown = flags.filter(flag => !['--list', '--write', '--all'].includes(flag))
|
||||
if (unknown.length > 0) throw new Error(`unknown flag(s): ${unknown.join(', ')}`)
|
||||
const listMode = flags.includes('--list')
|
||||
const writeMode = flags.includes('--write')
|
||||
const allMode = flags.includes('--all')
|
||||
if (listMode && (writeMode || allMode || anchors.length > 0)) {
|
||||
throw new Error('--list reports the whole corpus and takes no other flags or paths')
|
||||
}
|
||||
if (allMode && !writeMode) throw new Error('--all only applies to --write')
|
||||
if (writeMode) {
|
||||
if (anchors.length > 0 && allMode) throw new Error('--write takes either pair paths or --all, not both')
|
||||
if (anchors.length === 0 && !allMode) {
|
||||
throw new Error('--write requires the pair(s) you confirmed (any file of a pair), or --all to re-record every complete pair; recording pairs you did not review blesses unconfirmed content')
|
||||
}
|
||||
return { mode: 'write', scope: allMode ? 'corpus' : 'pairs', anchors }
|
||||
}
|
||||
if (listMode) return { mode: 'list', scope: 'corpus', anchors: [] }
|
||||
return { mode: 'check', scope: anchors.length > 0 ? 'pairs' : 'corpus', anchors }
|
||||
}
|
||||
|
||||
/** The structural surface compared between the two sides of a pair. */
|
||||
export interface TranslationStructureSignature {
|
||||
/** Heading depths in document order (h2 -> 2). */
|
||||
|
||||
@@ -46,6 +46,26 @@
|
||||
"symbol": "LlmModelContext",
|
||||
"source": "packages/llm/llm/src/types.ts"
|
||||
},
|
||||
{
|
||||
"doc": "docs/core-data-structures/core.md",
|
||||
"symbol": "ReasoningEffortId",
|
||||
"source": "packages/llm/llm/src/brand.ts"
|
||||
},
|
||||
{
|
||||
"doc": "docs/core-data-structures/core.md",
|
||||
"symbol": "LlmReasoningEffortInfo",
|
||||
"source": "packages/llm/llm/src/types.ts"
|
||||
},
|
||||
{
|
||||
"doc": "docs/core-data-structures/core.md",
|
||||
"symbol": "LlmModelReasoningInfo",
|
||||
"source": "packages/llm/llm/src/types.ts"
|
||||
},
|
||||
{
|
||||
"doc": "docs/core-data-structures/core.md",
|
||||
"symbol": "LlmResolvedModelInfo",
|
||||
"source": "packages/llm/llm/src/types.ts"
|
||||
},
|
||||
{
|
||||
"doc": "docs/core-data-structures/core.md",
|
||||
"symbol": "GenerateOptions",
|
||||
@@ -292,6 +312,11 @@
|
||||
"source": "packages/llm/llm/src/assembler.ts",
|
||||
"projection": "public-api"
|
||||
},
|
||||
{
|
||||
"doc": "docs/core-data-structures/llm-streaming.md",
|
||||
"symbol": "PreparedLlmCall",
|
||||
"source": "packages/llm/llm/src/index.ts"
|
||||
},
|
||||
{
|
||||
"doc": "docs/core-data-structures/llm-streaming.md",
|
||||
"symbol": "LlmAdapter",
|
||||
|
||||
@@ -55,6 +55,8 @@ const SENTENCE_MODEL_EXPERIENCE: Readonly<Record<string, SentenceContract>> = {
|
||||
'packages/client/ui-layout': { kind: 'none', reason: 'Browser-side UI plugin layer; registers no model surface.' },
|
||||
'packages/client/ui-sidebar': { kind: 'none', reason: 'Browser-side UI plugin layer; registers no model surface.' },
|
||||
'packages/client/ui-conversation': { kind: 'none', reason: 'Browser-side UI plugin layer; registers no model surface.' },
|
||||
'packages/client/ui-slash': { kind: 'none', reason: 'Browser-side UI plugin layer; registers no model surface.' },
|
||||
'packages/client/ui-command': { kind: 'indirect', reason: 'The dispatch paths trigger the host command.execute RPC; each command handler\'s host package owns any model-visible effect.' },
|
||||
'packages/client/ui-question': { kind: 'indirect', reason: 'The package mounts dsh-tool-ask-user; that tool owns the model-visible schema and answer rendering.' },
|
||||
'packages/client/ui-trajectory': { kind: 'none', reason: 'Browser-side UI plugin layer; registers no model surface.' },
|
||||
'packages/client/ui-workspace': { kind: 'none', reason: 'Browser-side UI plugin layer; registers no model surface.' },
|
||||
|
||||
@@ -2,8 +2,10 @@
|
||||
* Enforce complete English/Chinese pairs, matching structure, and recorded git
|
||||
* blob hashes for every in-scope document. The manifest contains only explicit
|
||||
* exclusions, which may have neither a counterpart nor a sidecar.
|
||||
* `--list` reports state and `--write` records both sides after human review.
|
||||
* Translation quality remains a review responsibility.
|
||||
* `--list` reports state; `--write <pairs...>` records the named confirmed
|
||||
* pairs (`--write --all` records every complete pair); a check or write named
|
||||
* with pair paths touches only those pairs, so update iteration does not pay
|
||||
* for a corpus scan. Translation quality remains a review responsibility.
|
||||
* See `docs/i18n/README.md` for the owning contract.
|
||||
*/
|
||||
|
||||
@@ -13,6 +15,7 @@ import { basename, join, resolve, sep } from 'node:path'
|
||||
import {
|
||||
linksTo,
|
||||
parseTranslationMarkdown,
|
||||
parseTranslationPairingCliArgs,
|
||||
parseTranslationPairingManifest,
|
||||
isTranslationScopeFile,
|
||||
TRANSLATION_SCOPE_GLOB_EXCLUDES,
|
||||
@@ -21,8 +24,15 @@ import {
|
||||
} from './translation-pairing.ts'
|
||||
|
||||
const root = resolve(import.meta.dirname, '..')
|
||||
const listMode = process.argv.includes('--list')
|
||||
const writeMode = process.argv.includes('--write')
|
||||
let request: ReturnType<typeof parseTranslationPairingCliArgs>
|
||||
try {
|
||||
request = parseTranslationPairingCliArgs(process.argv.slice(2))
|
||||
} catch (error) {
|
||||
console.error(`verify-translation-pairing: ${error instanceof Error ? error.message : String(error)}`)
|
||||
process.exit(2)
|
||||
}
|
||||
const listMode = request.mode === 'list'
|
||||
const writeMode = request.mode === 'write'
|
||||
|
||||
/** Discover source Markdown and pairing sidecars before applying the corpus predicate. */
|
||||
const SCOPE_PATTERNS = [
|
||||
@@ -77,32 +87,67 @@ function renderMeta(source: string, sourceHash: string, zh: string, zhHash: stri
|
||||
'# Bilingual-pair consistency record (docs/i18n/README.md): the git blob hash of each',
|
||||
'# side as of the last confirmed-consistent state. Both languages carry equal authority;',
|
||||
'# after editing either side, bring the other along and re-record with:',
|
||||
'# pnpm run verify-translation-pairing --write',
|
||||
`# pnpm run verify-translation-pairing --write ${source}`,
|
||||
`${basename(source)}: ${sourceHash}`,
|
||||
`${basename(zh)}: ${zhHash}`,
|
||||
'',
|
||||
].join('\n')
|
||||
}
|
||||
|
||||
// Enumerate the scope once.
|
||||
// Enumerate the scope once: the whole corpus, or exactly the named pairs'
|
||||
// three files (a named pair whose files are absent is caught by the same
|
||||
// completeness rules that cover discovered remnants).
|
||||
const files = new Set<string>()
|
||||
for (const pattern of SCOPE_PATTERNS) {
|
||||
for (const match of globSync(pattern, { cwd: root, exclude: TRANSLATION_SCOPE_GLOB_EXCLUDES })) {
|
||||
const normalized = match.split(sep).join('/')
|
||||
if (isTranslationScopeFile(normalized)) files.add(normalized)
|
||||
if (request.scope === 'pairs') {
|
||||
for (const anchor of request.anchors) {
|
||||
for (const file of [anchor, ...Object.values(pairPaths(anchor))]) {
|
||||
if (existsSync(join(root, file))) files.add(file)
|
||||
}
|
||||
// A named anchor with no files on disk still enters the source list so
|
||||
// the check reports it instead of silently passing an empty scope.
|
||||
if (!existsSync(join(root, anchor))) files.add(anchor)
|
||||
}
|
||||
} else {
|
||||
for (const pattern of SCOPE_PATTERNS) {
|
||||
for (const match of globSync(pattern, { cwd: root, exclude: TRANSLATION_SCOPE_GLOB_EXCLUDES })) {
|
||||
const normalized = match.split(sep).join('/')
|
||||
if (isTranslationScopeFile(normalized)) files.add(normalized)
|
||||
}
|
||||
}
|
||||
}
|
||||
const translations = [...files].filter(f => f.endsWith('.zh.md')).sort()
|
||||
const metas = [...files].filter(f => f.endsWith('.i18n.yaml')).sort()
|
||||
const sources = [...files].filter(f => f.endsWith('.md') && !f.endsWith('.zh.md')).sort()
|
||||
|
||||
// --write: (re)record both hashes for every complete pair, creating missing records.
|
||||
if (request.scope === 'pairs') {
|
||||
const rejected = request.anchors.filter(anchor => !isTranslationScopeFile(anchor) || isExcluded(anchor))
|
||||
const absent = request.anchors.filter(anchor => ![anchor, ...Object.values(pairPaths(anchor))].some(file => existsSync(join(root, file))))
|
||||
if (rejected.length > 0 || absent.length > 0) {
|
||||
for (const anchor of rejected) {
|
||||
console.error(`verify-translation-pairing: ${anchor} is not an in-scope pair (excluded or outside the documentation corpus; see docs/i18n/README.md)`)
|
||||
}
|
||||
for (const anchor of absent) {
|
||||
console.error(`verify-translation-pairing: ${anchor} names no pair on disk (none of its three files exist)`)
|
||||
}
|
||||
process.exit(2)
|
||||
}
|
||||
}
|
||||
|
||||
// --write: (re)record both hashes for the requested complete pairs, creating
|
||||
// missing records. A named pair that cannot be recorded (missing counterpart)
|
||||
// fails loud; corpus scope (--all) skips pairless sources as before.
|
||||
if (writeMode) {
|
||||
let written = 0
|
||||
for (const source of sources) {
|
||||
if (isExcluded(source)) continue
|
||||
const { zh, meta } = pairPaths(source)
|
||||
if (!existsSync(join(root, zh))) continue
|
||||
if (!existsSync(join(root, source)) || !existsSync(join(root, zh))) {
|
||||
if (request.scope === 'pairs') {
|
||||
console.error(`verify-translation-pairing: cannot record ${source}: missing ${existsSync(join(root, source)) ? zh : source}`)
|
||||
process.exit(2)
|
||||
}
|
||||
continue
|
||||
}
|
||||
const record = renderMeta(source, blobHash(readFileSync(join(root, source))), zh, blobHash(readFileSync(join(root, zh))))
|
||||
if (existsSync(join(root, meta)) && readFileSync(join(root, meta), 'utf8') === record) continue
|
||||
writeFileSync(join(root, meta), record)
|
||||
@@ -204,7 +249,9 @@ if (listMode) {
|
||||
}
|
||||
|
||||
if (errors.length === 0) {
|
||||
console.log(`verify-translation-pairing: ${pairAnchors.size} pair(s) checked across all in-scope documentation, all consistent.`)
|
||||
console.log(request.scope === 'pairs'
|
||||
? `verify-translation-pairing: ${pairAnchors.size} named pair(s) consistent; the corpus-wide check still runs in doc-sync.`
|
||||
: `verify-translation-pairing: ${pairAnchors.size} pair(s) checked across all in-scope documentation, all consistent.`)
|
||||
process.exit(0)
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user