Merge branch 'master' into sdk/ts-client-and-subagent

This commit is contained in:
Tianyi Cui
2026-07-27 17:01:20 +08:00
committed by GitHub
312 changed files with 7972 additions and 1668 deletions

View File

@@ -17,8 +17,8 @@
* `watch` through API-level inline config (tsdown workspace mode fills inline
* keys under each package's file config, and no package config defines it).
*/
import { readdirSync, readFileSync } from 'node:fs'
import { join } from 'node:path'
import { globSync, readFileSync } from 'node:fs'
import { dirname, join, sep } from 'node:path'
import { fileURLToPath } from 'node:url'
import { build } from 'tsdown'
@@ -33,20 +33,9 @@ const repoRoot = fileURLToPath(new URL('..', import.meta.url))
*/
function discoverPluginDirs(): string[] {
const dirs: string[] = []
for (const group of readdirSync(join(repoRoot, 'packages'), { withFileTypes: true })) {
if (!group.isDirectory()) continue
for (const pkg of readdirSync(join(repoRoot, 'packages', group.name), { withFileTypes: true })) {
if (!pkg.isDirectory()) continue
let manifest: { dshClient?: { platform?: unknown } }
try {
manifest = JSON.parse(
readFileSync(join(repoRoot, 'packages', group.name, pkg.name, 'package.json'), 'utf8'),
) as { dshClient?: { platform?: unknown } }
} catch {
continue // no package.json (support dirs, scratch): not a workspace package
}
if (manifest.dshClient?.platform === 'web') dirs.push(`packages/${group.name}/${pkg.name}`)
}
for (const manifestPath of globSync('packages/*/*/package.json', { cwd: repoRoot }).sort()) {
const manifest = JSON.parse(readFileSync(join(repoRoot, manifestPath), 'utf8')) as { dshClient?: { platform?: unknown } }
if (manifest.dshClient?.platform === 'web') dirs.push(dirname(manifestPath).split(sep).join('/'))
}
return dirs
}

View File

@@ -10,7 +10,7 @@ import { globSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node
import { join, relative, resolve } from 'node:path'
import ts from 'typescript'
import { builtDeclarationPath } from './doc-typecheck-paths.ts'
import { extractFences } from './md-fences.ts'
import { markdownFences } from './markdown.ts'
import { partitionPairedMarkdownDerivatives } from './paired-markdown-derivatives.ts'
import { isArchivedAgentNotePath } from './repo-files.ts'
@@ -46,8 +46,10 @@ const KIND_BY_INFO: Record<string, BlockKind> = {
/** Extract every recognized TypeScript fence from one Markdown file. */
function extractBlocks(absPath: string): Block[] {
const file = relative(root, absPath)
return extractFences(absPath, info => KIND_BY_INFO[info] ?? null)
.map(f => ({ file, line: f.line, kind: f.kind, code: f.code }))
return markdownFences(readFileSync(absPath, 'utf8')).flatMap((fence) => {
const kind = KIND_BY_INFO[fence.info]
return kind === undefined ? [] : [{ file, line: fence.line, kind, code: fence.code }]
})
}
const configHost: ts.ParseConfigFileHost = {

View File

@@ -40,6 +40,8 @@ export const LINK_MAP: Record<string, string> = {
HookContext: 'core.md',
LlmCallConfig: 'core.md',
LlmModelContext: 'core.md',
LlmModelReasoningInfo: 'core.md',
LlmResolvedModelInfo: 'core.md',
LlmFailure: 'llm-streaming.md',
LlmModelInfo: 'core.md',
LlmProviderInfo: 'core.md',
@@ -92,6 +94,7 @@ export const LINK_MAP: Record<string, string> = {
CommandResult: 'commands.md',
CommandSurface: 'commands.md',
LlmAdapter: 'llm-streaming.md',
PreparedLlmCall: 'llm-streaming.md',
LlmService: 'llm-streaming.md',
StreamChunk: 'llm-streaming.md',
CreateSessionOptions: 'persistence.md',

View File

@@ -0,0 +1,302 @@
/**
* Print the minimal-update briefing for out-of-sync translation pairs:
* `pnpm run gen-translation-brief [--apply] [pair paths...]`. With no
* arguments it discovers every out-of-sync pair; with arguments (any file
* of a pair) it briefs exactly those pairs and fails loud on in-sync,
* incomplete, or out-of-scope requests. Each briefing maps the change at
* the narrowest safe granularity — code-fence-only splice, changed
* Markdown units, heading sections, whole document — and `--apply` writes
* the computed counterpart for pairs whose change is code-fence-only.
* The briefing contract lives in `scripts/translation-brief.ts`; the
* consuming workflow is `.agents/skills/dsh-translate-docs/SKILL.md`.
*/
import { spawnSync } from 'node:child_process'
import { existsSync, globSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs'
import { tmpdir } from 'node:os'
import { basename, join, resolve, sep } from 'node:path'
import {
isTranslationScopeFile,
pairAnchorOfArgument,
parseTranslationMarkdown,
parseTranslationPairingManifest,
TRANSLATION_SCOPE_GLOB_EXCLUDES,
translationStructureDiff,
translationStructureSignature,
} from './translation-pairing.ts'
import {
changedSpanIndices,
computeMechanicalUpdate,
firstOccurrenceContext,
markdownUnits,
relevantTerminologyRows,
renderTranslationBrief,
sectionSpans,
spansAligned,
type BriefBundle,
type BriefDirection,
type BriefScope,
type MarkdownSpan,
} from './translation-brief.ts'
const root = resolve(import.meta.dirname, '..')
const manifest = parseTranslationPairingManifest(readFileSync(join(root, 'scripts/translation-pairing.manifest.json'), 'utf8'))
const terminology = readFileSync(join(root, 'docs/i18n/terminology.md'), 'utf8')
function isExcluded(file: string): boolean {
return manifest.excluded.some(entry => (entry.endsWith('/') ? file.startsWith(entry) : file === entry))
}
/** Recorded hashes of one consistency record: basename → blob hash. */
function parseMeta(content: string): Map<string, string> | undefined {
const out = new Map<string, string>()
for (const line of content.split('\n')) {
if (line === '' || line.startsWith('#')) continue
const match = /^([^:#]+\.md): ([0-9a-f]{40})$/.exec(line)
if (!match?.[1] || !match[2]) return undefined
out.set(match[1], match[2])
}
return out
}
function git(args: string[], allowedExitCodes: number[] = [0]): string {
const result = spawnSync('git', ['-C', root, ...args], { encoding: 'utf8', maxBuffer: 1 << 26 })
if (result.error) throw result.error
if (!allowedExitCodes.includes(result.status ?? -1)) {
throw new Error(`git ${args.join(' ')} failed: ${result.stderr}`)
}
return result.stdout
}
function blobText(hash: string): string {
return git(['cat-file', '-p', hash])
}
/** Unified diff between two texts, headers stripped, via `git diff --no-index`. */
function diffTexts(before: string, after: string): string {
const dir = mkdtempSync(join(tmpdir(), 'translation-brief-'))
try {
writeFileSync(join(dir, 'last-confirmed.md'), before)
writeFileSync(join(dir, 'current.md'), after)
const raw = git(['diff', '--no-index', '--unified=2', join(dir, 'last-confirmed.md'), join(dir, 'current.md')], [0, 1])
return raw.split('\n')
.filter(line => !line.startsWith('diff --git') && !line.startsWith('index ') && !line.startsWith('--- ') && !line.startsWith('+++ '))
.join('\n')
.trim()
} finally {
rmSync(dir, { recursive: true, force: true })
}
}
interface PairState {
anchor: string
zh: string
meta: string
enDrifted: boolean
zhDrifted: boolean
enLast: string
zhLast: string
}
/** Load one pair's recorded and current state, or explain why it cannot be briefed. */
function loadPair(anchor: string): PairState | string {
const zh = anchor.replace(/\.md$/, '.zh.md')
const meta = anchor.replace(/\.md$/, '.i18n.yaml')
if (!isTranslationScopeFile(anchor) || isExcluded(anchor)) {
return `${anchor}: not an in-scope documentation pair (docs/i18n/README.md)`
}
const missing = [anchor, zh, meta].filter(file => !existsSync(join(root, file)))
if (missing.length > 0) {
return `${anchor}: incomplete pair (missing ${missing.join(', ')}) — a new counterpart is whole-document translation work, not a minimal update`
}
const record = parseMeta(readFileSync(join(root, meta), 'utf8'))
const enRecorded = record?.get(basename(anchor))
const zhRecorded = record?.get(basename(zh))
if (record === undefined || enRecorded === undefined || zhRecorded === undefined) {
return `${meta}: malformed consistency record`
}
const enCurrent = readFileSync(join(root, anchor), 'utf8')
const zhCurrent = readFileSync(join(root, zh), 'utf8')
const enLast = blobText(enRecorded)
const zhLast = blobText(zhRecorded)
return {
anchor,
zh,
meta,
enDrifted: enCurrent !== enLast,
zhDrifted: zhCurrent !== zhLast,
enLast,
zhLast,
}
}
/** Assemble bundles for the given changed + first-occurrence span indices. */
function bundlesFor(
indices: number[],
extraIndices: number[],
confirmed: MarkdownSpan[],
current: MarkdownSpan[],
counterpart: MarkdownSpan[],
): BriefBundle[] {
const extras = new Set(extraIndices)
return [...new Set([...indices, ...extraIndices])].sort((left, right) => left - right).map((index) => {
const confirmedSpan = confirmed[index]
const currentSpan = current[index]
const counterpartSpan = counterpart[index]
if (confirmedSpan === undefined || currentSpan === undefined || counterpartSpan === undefined) {
throw new Error(`gen-translation-brief: span ${index} is unmapped despite alignment`)
}
return {
index,
label: currentSpan.label,
reason: extras.has(index) && confirmedSpan.text === currentSpan.text ? 'first-occurrence' as const : undefined,
confirmedSourceText: confirmedSpan.text,
currentSourceText: currentSpan.text,
counterpartText: counterpartSpan.text,
counterpartStartLine: counterpartSpan.startLine,
}
})
}
interface PlannedBrief {
scope: BriefScope
/** Old + new text of the changed spans, for terminology matching. */
changedText: string
/** Computed counterpart for a mechanical scope, for `--apply`. */
mechanicalResult?: string | undefined
}
/** Choose the narrowest safely mapped granularity for one drifted side. */
function planScope(
sourceLast: string,
sourceCurrent: string,
counterpartCurrent: string,
direction: BriefDirection,
bothDrifted: boolean,
): PlannedBrief {
const wholeChangedText = `${sourceLast}\n${sourceCurrent}`
if (bothDrifted) {
return {
scope: { kind: 'document', reason: 'BOTH sides changed since the pair was last confirmed consistent, so no side is a trustworthy mapping anchor; decide which side owns each divergence.' },
changedText: wholeChangedText,
}
}
const mechanical = computeMechanicalUpdate(sourceLast, sourceCurrent, counterpartCurrent)
if (mechanical !== undefined) {
return { scope: { kind: 'mechanical' }, changedText: wholeChangedText, mechanicalResult: mechanical }
}
for (const [kind, spansOf] of [['units', markdownUnits], ['sections', sectionSpans]] as const) {
const confirmed = spansOf(sourceLast)
const current = spansOf(sourceCurrent)
const counterpart = spansOf(counterpartCurrent)
if (!spansAligned(confirmed, current) || !spansAligned(confirmed, counterpart)) continue
const changed = changedSpanIndices(confirmed, current)
if (changed.length === 0) continue
const changedText = changed.map(index => `${confirmed[index]?.text ?? ''}\n${current[index]?.text ?? ''}`).join('\n')
const rows = relevantTerminologyRows(terminology, direction, changedText)
const occurrence = direction === 'en-to-zh'
? firstOccurrenceContext(sourceLast, sourceCurrent, confirmed, current, rows, new Set(changed))
: { notes: [], extraSpanIndices: [] }
return {
scope: {
kind,
bundles: bundlesFor(changed, occurrence.extraSpanIndices, confirmed, current, counterpart),
firstOccurrenceNotes: occurrence.notes,
},
changedText,
}
}
return {
scope: { kind: 'document', reason: 'Neither fine-grained units nor heading sections align one to one across the last-confirmed source, current source, and current counterpart.' },
changedText: wholeChangedText,
}
}
/** Validate a computed mechanical counterpart and write it. */
function applyMechanical(counterpartPath: string, sourceCurrent: string, result: string): void {
const counterpartBase = basename(counterpartPath)
const sourceBase = counterpartBase.endsWith('.zh.md')
? counterpartBase.replace(/\.zh\.md$/, '.md')
: counterpartBase.replace(/\.md$/, '.zh.md')
const errors = translationStructureDiff(
translationStructureSignature(parseTranslationMarkdown(sourceCurrent), counterpartBase),
translationStructureSignature(parseTranslationMarkdown(result), sourceBase),
)
if (errors.length > 0) {
throw new Error(`gen-translation-brief: computed mechanical update for ${counterpartPath} violates the pair structure: ${errors.join('; ')}`)
}
writeFileSync(join(root, counterpartPath), result)
console.error(`gen-translation-brief: applied code-fence splice to ${counterpartPath}; review the diff, then record the pair.`)
}
/** Render (and under `--apply`, apply) the briefing for one drifted side. */
function briefDirection(pair: PairState, direction: BriefDirection, apply: boolean): string {
const sourceIsEnglish = direction === 'en-to-zh'
const sourcePath = sourceIsEnglish ? pair.anchor : pair.zh
const counterpartPath = sourceIsEnglish ? pair.zh : pair.anchor
const sourceLast = sourceIsEnglish ? pair.enLast : pair.zhLast
const sourceCurrent = readFileSync(join(root, sourcePath), 'utf8')
const counterpartCurrent = readFileSync(join(root, counterpartPath), 'utf8')
const diff = diffTexts(sourceLast, sourceCurrent)
const planned = planScope(sourceLast, sourceCurrent, counterpartCurrent, direction, pair.enDrifted && pair.zhDrifted)
if (apply && planned.mechanicalResult !== undefined) {
applyMechanical(counterpartPath, sourceCurrent, planned.mechanicalResult)
}
return renderTranslationBrief({
sourcePath,
counterpartPath,
direction,
diff,
scope: planned.scope,
terminology: relevantTerminologyRows(terminology, direction, planned.changedText),
})
}
const argv = process.argv.slice(2)
const flags = argv.filter(argument => argument.startsWith('--'))
const unknownFlags = flags.filter(flag => flag !== '--apply')
if (unknownFlags.length > 0) {
console.error(`gen-translation-brief: unknown flag(s): ${unknownFlags.join(', ')} (only --apply is supported)`)
process.exit(2)
}
const applyMode = flags.includes('--apply')
const requested = argv.filter(argument => !argument.startsWith('--')).map(pairAnchorOfArgument)
let anchors: string[]
if (requested.length > 0) {
anchors = [...new Set(requested)].sort()
} else {
const discovered = new Set<string>()
for (const match of globSync('**/*.i18n.yaml', { cwd: root, exclude: TRANSLATION_SCOPE_GLOB_EXCLUDES })) {
const normalized = match.split(sep).join('/')
if (isTranslationScopeFile(normalized)) discovered.add(normalized.replace(/\.i18n\.yaml$/, '.md'))
}
anchors = [...discovered].sort()
}
const briefs: string[] = []
const problems: string[] = []
const skipped: string[] = []
for (const anchor of anchors) {
const pair = loadPair(anchor)
if (typeof pair === 'string') {
if (requested.length > 0) problems.push(pair)
continue
}
if (!pair.enDrifted && !pair.zhDrifted) {
if (requested.length > 0) skipped.push(`${anchor}: pair is consistent with its record — nothing to brief`)
continue
}
if (pair.enDrifted) briefs.push(briefDirection(pair, 'en-to-zh', applyMode))
if (pair.zhDrifted) briefs.push(briefDirection(pair, 'zh-to-en', applyMode))
}
if (problems.length > 0 || skipped.length > 0) {
for (const message of [...problems, ...skipped]) console.error(`gen-translation-brief: ${message}`)
process.exit(2)
}
if (briefs.length === 0) {
console.log('gen-translation-brief: every recorded pair matches its consistency record; nothing to brief.')
process.exit(0)
}
console.log(briefs.join('\n\n---\n\n'))

View File

@@ -21,6 +21,24 @@ export interface MarkdownHeadingLine extends MarkdownProseLine {
text: string
}
/** One code block from a parsed Markdown source. */
export interface MarkdownFence {
/** 1-based source line of the opening fence. */
line: number
/** Info-string language (its first word), null on a bare or indented block. */
lang: string | null
/** Full info string (e.g. `ts ignore-check`), '' on a bare or indented block. */
info: string
/** Block body without the fence delimiters. */
code: string
/**
* Whether a closing fence delimiter terminates the block — mdast silently
* closes an unterminated fence at end of file. False on indented
* (non-fenced) blocks, whose end line is code.
*/
closed: boolean
}
/** Parse GitHub-flavored Markdown with the repository's standard extensions. */
export function parseMarkdown(source: string): Nodes {
return fromMarkdown(source, { extensions: [gfm()], mdastExtensions: [gfmFromMarkdown()] })
@@ -38,6 +56,26 @@ export function visitMarkdown(node: Nodes, visitor: (node: Nodes) => boolean | v
}
}
/**
* Extract every parsed code block with its info string, in document order.
* @param source - Markdown source to scan.
* @returns each block's opening line, language, info string, and body.
*/
export function markdownFences(source: string): MarkdownFence[] {
const lines = source.split('\n')
const fences: MarkdownFence[] = []
visitMarkdown(parseMarkdown(source), (node) => {
if (node.type !== 'code' || node.position === undefined) return
const lang = node.lang ?? null
const meta = node.meta ?? ''
const info = lang === null ? '' : meta === '' ? lang : `${lang} ${meta}`
const endLine = lines[node.position.end.line - 1] ?? ''
const closed = /^ {0,3}(`{3,}|~{3,})\s*$/.test(endLine)
fences.push({ line: node.position.start.line, lang, info, code: node.value, closed })
})
return fences
}
/** Text a reader sees from one Markdown node; raw HTML itself contributes none. */
function renderedText(node: Nodes): string {
if (node.type === 'text' || node.type === 'inlineCode') return node.value
@@ -115,27 +153,22 @@ function hasRenderedTextOutsideComments(raw: string, ranges: readonly ColumnRang
}
/**
* Return source lines outside backtick or tilde fences and HTML comments.
* Return source lines outside code blocks and HTML comments.
* @param source - Markdown source whose prose should be retained verbatim.
* @returns unfenced lines with their original 1-based locations.
*/
export function markdownProseLines(source: string): MarkdownProseLine[] {
let fence: { marker: '`' | '~'; length: number } | undefined
const kept: MarkdownProseLine[] = []
const rawLines = source.split('\n')
const comments = htmlCommentRanges(source, rawLines)
const fenced = new Set<number>()
visitMarkdown(parseMarkdown(source), (node) => {
if (node.type !== 'code' || node.position === undefined) return
for (let line = node.position.start.line; line <= node.position.end.line; line += 1) fenced.add(line)
})
const kept: MarkdownProseLine[] = []
rawLines.forEach((raw, i) => {
const token = /^ {0,3}(`{3,}|~{3,})/.exec(raw)?.[1]
if (token !== undefined) {
const marker = token[0] as '`' | '~'
if (fence === undefined) {
fence = { marker, length: token.length }
} else if (marker === fence.marker && token.length >= fence.length) {
fence = undefined
}
return
}
if (fence === undefined && hasRenderedTextOutsideComments(raw, comments.get(i + 1))) {
if (fenced.has(i + 1)) return
if (hasRenderedTextOutsideComments(raw, comments.get(i + 1))) {
kept.push({ index: i + 1, raw })
}
})

View File

@@ -1,55 +0,0 @@
/**
* Shared fenced-code-block extractor for the Markdown doc gates
* (currently `doc-typecheck.ts`; future Markdown gates can share it). One scanner, per-gate
* classification: each gate maps a fence info string (` ```ts `,
* ` ```yaml ignore-check `, …) to its own kind tag and receives every
* classified block with its 1-based opening-fence line.
*/
import { readFileSync } from 'node:fs'
/** One extracted fenced block, classified by the caller's `classify`. */
export interface Fence<K> {
/** 1-based line of the opening fence. */
line: number
kind: K
code: string
}
/**
* Extract every fenced block of `absPath` whose info string `classify` maps
* to a kind. Blocks classified `null` are skipped (their bodies are still
* consumed, so an unrelated fence can never leak into a tracked one).
*
* @param absPath — absolute path of the Markdown file.
* @param classify — info string (trimmed, e.g. `ts ignore-check`) → kind, or
* null for fences this gate does not track.
* @returns the classified blocks in document order.
*/
export function extractFences<K>(absPath: string, classify: (info: string) => K | null): Fence<K>[] {
const lines = readFileSync(absPath, 'utf8').split('\n')
const blocks: Fence<K>[] = []
let open: { line: number; kind: K; body: string[] } | null = null
let skipping = false
lines.forEach((raw, i) => {
const fence = /^```(\s*)(\S.*)?$/.exec(raw)
if (!fence) {
if (open) open.body.push(raw)
return
}
if (open) {
blocks.push({ line: open.line, kind: open.kind, code: open.body.join('\n') })
open = null
return
}
if (skipping) {
skipping = false
return
}
const kind = classify((fence[2] ?? '').trim())
if (kind !== null) open = { line: i + 1, kind, body: [] }
else skipping = true
})
return blocks
}

View File

@@ -2,19 +2,23 @@
import {
globSync,
readFileSync,
readdirSync,
readFileSync,
statSync,
} from 'node:fs'
import { availableParallelism } from 'node:os'
import { dirname, relative, resolve, sep } from 'node:path'
import { parseArgs } from 'node:util'
import { publint, type Message, type PackFile } from 'publint'
import { formatMessage } from 'publint/utils'
const CONCURRENCY_ENV = 'DSH_PUBLINT_CONCURRENCY'
const repositoryRoot = resolve(import.meta.dirname, '..')
const options = parseOptions(process.argv.slice(2))
const packagesRoot = resolve(options.get('--packages-root') ?? repositoryRoot)
const { values: options } = parseArgs({
args: process.argv.slice(2),
options: { 'packages-root': { type: 'string' } },
})
const packagesRoot = resolve(options['packages-root'] ?? repositoryRoot)
interface PackageTarget {
path: string
@@ -88,7 +92,12 @@ function publicationFiles(target: PackageTarget): PackFile[] {
function addPath(path: string, paths: Set<string>): void {
const stat = statSync(path)
if (stat.isDirectory()) {
for (const entry of readdirSync(path)) addPath(resolve(path, entry), paths)
// readdirSync, not globSync: `**/*` skips dot-prefixed segments, but npm
// pack publishes dotfiles inside included directories, and this view must
// match what npm publishes.
for (const entry of readdirSync(path, { recursive: true, withFileTypes: true })) {
if (entry.isFile()) paths.add(resolve(entry.parentPath, entry.name))
}
} else if (stat.isFile()) {
paths.add(path)
}
@@ -144,20 +153,6 @@ function printResult(result: PublintResult): void {
if (result.status === 'passed' && result.messages.length === 0) console.log('All good!')
}
function parseOptions(args: string[]): Map<string, string> {
const parsed = new Map<string, string>()
for (let index = 0; index < args.length; index += 2) {
const name = args[index]
const value = args[index + 1]
if (name !== '--packages-root' || value === undefined || value.startsWith('--')) {
throw new Error(`publint-all: expected [--packages-root PATH], got ${JSON.stringify(args)}.`)
}
if (parsed.has(name)) throw new Error(`publint-all: duplicate option ${name}.`)
parsed.set(name, value)
}
return parsed
}
const packages = workspacePackages()
const concurrency = publintConcurrency(packages.length)
console.log(`publint-all: linting ${packages.length} package(s) with ${concurrency} worker(s).`)

File diff suppressed because one or more lines are too long

View File

@@ -0,0 +1,283 @@
/** Regression tests for the minimal-update briefing assembly. */
import { describe, expect, it } from 'vitest'
import {
changedSpanIndices,
computeMechanicalUpdate,
firstOccurrenceContext,
markdownUnits,
parseTerminologyRows,
relevantTerminologyRows,
renderTranslationBrief,
sectionSpans,
spansAligned,
termOffsets,
} from './translation-brief.ts'
const DOC = [
'Preamble line.',
'',
'# Title',
'',
'Intro paragraph.',
'',
'## First',
'',
'First body.',
'',
'```ts',
'const value = 1',
'```',
'',
'## Second',
'',
'| A | B |',
'|---|---|',
'| 1 | 2 |',
'',
'- item one',
'- item two',
].join('\n')
describe('markdown spans', () => {
it('lists units with container-scoped kinds in document order', () => {
const kinds = markdownUnits(DOC).map(span => span.kind)
expect(kinds).toEqual([
'root.0:paragraph',
'root.1:heading:1',
'root.2:paragraph',
'root.3:heading:2',
'root.4:paragraph',
'root.5:code',
'root.6:heading:2',
'root.7.0:tableRow',
'root.7.1:tableRow',
'root.8.0:listItem',
'root.8.1:listItem',
])
})
it('lists heading sections with a preamble span and heading labels', () => {
const sections = sectionSpans(DOC)
expect(sections.map(span => span.label)).toEqual([
'(preamble before the first heading)',
'Title',
'First',
'Second',
])
expect(sections[0]).toMatchObject({ startLine: 1, endLine: 2 })
expect(sections[2]).toMatchObject({ startLine: 7, endLine: 14 })
})
it('labels units by their node type', () => {
const units = markdownUnits(DOC)
expect(units[0]!.label).toBe('paragraph')
expect(units[1]!.label).toBe('heading')
expect(units[7]!.label).toBe('tableRow')
})
it('aligns sections by depth only, so translated heading text still maps', () => {
const zh = DOC.replace('## First', '## 第一节').replace('## Second', '## 第二节').replace('# Title', '# 标题')
expect(spansAligned(sectionSpans(DOC), sectionSpans(zh))).toBe(true)
})
it('aligns span lists only on equal non-empty kind sequences', () => {
const zh = DOC.replace('First body.', '第一段。').replace('item one', '第一项').replace('Intro paragraph.', '导语。')
expect(spansAligned(markdownUnits(DOC), markdownUnits(zh))).toBe(true)
const reshaped = DOC.replace('- item one\n- item two', 'merged paragraph')
expect(spansAligned(markdownUnits(DOC), markdownUnits(reshaped))).toBe(false)
expect(spansAligned([], [])).toBe(false)
})
it('reports the indices whose text changed', () => {
const edited = DOC.replace('First body.', 'First body, revised.').replace('| 1 | 2 |', '| 1 | 3 |')
expect(changedSpanIndices(markdownUnits(DOC), markdownUnits(edited))).toEqual([4, 8])
})
})
describe('mechanical code updates', () => {
const en = '# T\n\nProse.\n\n```sh\nrun one\n```\n'
const zh = '# T\n\n中文。\n\n```sh\nrun one\n```\n'
it('splices a fence-only edit into the counterpart', () => {
const edited = en.replace('run one', 'run two')
expect(computeMechanicalUpdate(en, edited, zh)).toBe(zh.replace('run one', 'run two'))
})
it('refuses when prose changed too', () => {
const edited = en.replace('Prose.', 'Prose!').replace('run one', 'run two')
expect(computeMechanicalUpdate(en, edited, zh)).toBeUndefined()
})
it('refuses when the counterpart fences already diverge from last-confirmed', () => {
const edited = en.replace('run one', 'run two')
expect(computeMechanicalUpdate(en, edited, zh.replace('run one', 'run stale'))).toBeUndefined()
})
it('refuses when fence counts differ or nothing changed', () => {
expect(computeMechanicalUpdate(en, `${en}\n\`\`\`sh\nextra\n\`\`\`\n`, zh)).toBeUndefined()
expect(computeMechanicalUpdate(en, en, zh)).toBeUndefined()
})
})
const TERMINOLOGY = [
'| English | 中文 | 首次出现 | 不要译作 | 备注 |',
'|---|---|---|---|---|',
'| agent | agent | agent智能体 | 智能体 | |',
'| session log | 会话日志 | | 会话记录 | |',
'| gate | 门禁 | | | |',
'| registry | 注册表 | | | |',
].join('\n')
describe('terminology', () => {
it('parses data rows and skips the header and separator', () => {
const rows = parseTerminologyRows(TERMINOLOGY)
expect(rows.map(row => row.english)).toEqual(['agent', 'session log', 'gate', 'registry'])
expect(rows[0]).toMatchObject({ chinese: 'agent', first: 'agent智能体' })
})
it('matches English terms on word boundaries with plural inflections', () => {
expect(termOffsets('two agents met', 'agent', true)).toEqual([4])
expect(termOffsets('two registries', 'registry', true)).toEqual([4])
expect(termOffsets('reagents', 'agent', true)).toEqual([])
expect(termOffsets('', 'agent', true)).toEqual([])
})
it('selects rows for the changed text per direction', () => {
expect(relevantTerminologyRows(TERMINOLOGY, 'en-to-zh', 'All agents write a session log.').map(row => row.english))
.toEqual(['agent', 'session log'])
expect(relevantTerminologyRows(TERMINOLOGY, 'zh-to-en', '门禁在提交时运行。').map(row => row.english))
.toEqual(['gate'])
expect(relevantTerminologyRows(TERMINOLOGY, 'en-to-zh', 'delegate the work')).toEqual([])
})
})
describe('first-occurrence tracking', () => {
const before = '# T\n\nAlpha paragraph.\n\nThe agent runs.\n'
const after = '# T\n\nAlpha paragraph with an agent.\n\nThe agent runs.\n'
const rows = parseTerminologyRows(TERMINOLOGY).filter(row => row.english === 'agent')
it('flags a moved first occurrence and pulls the vacated span in', () => {
const context = firstOccurrenceContext(
before, after, markdownUnits(before), markdownUnits(after), rows, new Set([1]),
)
expect(context.notes).toHaveLength(1)
expect(context.notes[0]).toContain('moved from #2 to #1')
expect(context.extraSpanIndices).toEqual([2])
})
it('stays silent when the first occurrence does not move', () => {
const unmoved = before.replace('Alpha paragraph.', 'Alpha paragraph, revised.')
const context = firstOccurrenceContext(
before, unmoved, markdownUnits(before), markdownUnits(unmoved), rows, new Set([1]),
)
expect(context.notes).toEqual([])
expect(context.extraSpanIndices).toEqual([])
})
it('ignores rows without a first-occurrence rendering', () => {
const bare = parseTerminologyRows(TERMINOLOGY).filter(row => row.english === 'gate')
const withGate = after.replace('The agent runs.', 'The gate runs.')
const context = firstOccurrenceContext(
before, withGate, markdownUnits(before), markdownUnits(withGate), bare, new Set([2]),
)
expect(context.notes).toEqual([])
})
})
describe('brief rendering', () => {
const base = {
sourcePath: 'docs/foo.md',
counterpartPath: 'docs/foo.zh.md',
direction: 'en-to-zh' as const,
diff: '@@ -5 +5 @@\n-old text about the agent\n+new text about the agent',
terminology: relevantTerminologyRows(TERMINOLOGY, 'en-to-zh', 'the agent'),
}
const bundle = {
index: 4,
label: 'paragraph',
confirmedSourceText: 'old text about the agent\n',
currentSourceText: 'new text about the agent\n',
counterpartText: '关于 agent 的旧文本\n',
counterpartStartLine: 9,
}
it('renders unit bundles with three-way context and line anchors', () => {
const brief = renderTranslationBrief({
...base,
scope: { kind: 'units', bundles: [bundle], firstOccurrenceNotes: ['agent: the document-wide first occurrence moved from #2 to #1; the agent智能体 form moves with it (later occurrences drop the annotation).'] },
})
expect(brief).toContain('# Translation update briefing: docs/foo.md')
expect(brief).toContain('## Changed units')
expect(brief).toContain('### #4 paragraph — counterpart at docs/foo.zh.md:9')
expect(brief).toContain('Last-confirmed English:')
expect(brief).toContain('Current Chinese (bring this along):')
expect(brief).toContain('## First-occurrence notes')
expect(brief).toContain('agent智能体')
expect(brief).toContain('首次出现 annotations attach to the document-wide first occurrence only')
expect(brief).toContain('verify-translation-pairing --write docs/foo.md')
})
it('marks first-occurrence bundles and omits their unchanged confirmed text', () => {
const brief = renderTranslationBrief({
...base,
scope: {
kind: 'units',
bundles: [{ ...bundle, reason: 'first-occurrence', confirmedSourceText: bundle.currentSourceText }],
firstOccurrenceNotes: [],
},
})
expect(brief).toContain('unchanged; included for a first-occurrence move')
expect(brief).not.toContain('Last-confirmed English:')
})
it('renders the mechanical scope with the --apply command', () => {
const brief = renderTranslationBrief({ ...base, scope: { kind: 'mechanical' } })
expect(brief).toContain('## Mechanical update — no translation judgment involved')
expect(brief).toContain('gen-translation-brief --apply docs/foo.md')
expect(brief).not.toContain('## Changed units')
})
it('renders the section fallback under its own heading', () => {
const brief = renderTranslationBrief({
...base,
scope: { kind: 'sections', bundles: [bundle], firstOccurrenceNotes: [] },
})
expect(brief).toContain('## Changed sections')
expect(brief).toContain('fine-grained units do not align')
})
it('renders the document fallback with its reason and no bundles', () => {
const brief = renderTranslationBrief({
...base,
scope: { kind: 'document', reason: 'BOTH sides changed since the pair was last confirmed consistent, so no side is a trustworthy mapping anchor; decide which side owns each divergence.' },
})
expect(brief).toContain('## Whole-document update required')
expect(brief).toContain('BOTH sides changed')
expect(brief).toContain('locate the affected regions yourself')
})
it('renders the English-target digest for zh-to-en updates', () => {
const brief = renderTranslationBrief({
...base,
direction: 'zh-to-en',
sourcePath: 'docs/foo.zh.md',
counterpartPath: 'docs/foo.md',
scope: { kind: 'units', bundles: [bundle], firstOccurrenceNotes: [] },
})
expect(brief).toContain('exactly what the new Chinese states')
expect(brief).toContain('verify-translation-pairing --write docs/foo.md')
})
it('grows bundle fences past tilde runs in the text', () => {
const brief = renderTranslationBrief({
...base,
scope: {
kind: 'units',
bundles: [{ ...bundle, counterpartText: '~~~~\ninner\n~~~~\n' }],
firstOccurrenceNotes: [],
},
})
expect(brief).toContain('~~~~~markdown')
})
})

View File

@@ -0,0 +1,513 @@
/**
* Pure assembly of the minimal-update briefing for one out-of-sync
* translation pair: the authored side's changes since the last confirmed
* state at the narrowest safely mapped granularity (code-fence-only splice,
* changed Markdown units, heading sections, whole document), the terminology
* rows those changes touch, first-occurrence movement notes, and a digest of
* the binding update rules. The unit mapping, mechanical code splice, and
* first-occurrence tracking adopt the planner mechanics validated in the
* incremental-pipeline work (PR #684). The CLI wrapper is
* `scripts/gen-translation-brief.ts`; the workflow that consumes the
* briefing is `.agents/skills/dsh-translate-docs/SKILL.md`.
*/
import type { Nodes } from 'mdast'
import { parseTranslationMarkdown } from './translation-pairing.ts'
/** One block-level span of a Markdown document, in document order. */
export interface MarkdownSpan {
/** Position in the span list; briefing ids derive from it. */
index: number
/**
* Structural kind compared for alignment, language-neutral: container path
* plus node type for units (`root.3:tableRow`), depth for sections (`section:2`).
*/
kind: string
/** Reader-facing label: heading text for sections, node type for units. */
label: string
/** 1-based first source line. */
startLine: number
/** 1-based last source line. */
endLine: number
/** The span's text, trailing newline normalized to exactly one. */
text: string
}
function linesOf(markdown: string): string[] {
const lines = markdown.replaceAll('\r\n', '\n').split('\n')
if (lines.at(-1) === '') lines.pop()
return lines
}
function sliceLines(lines: string[], startLine: number, endLine: number): string {
return `${lines.slice(startLine - 1, endLine).join('\n')}\n`
}
/**
* List a document's translation units: the outermost block nodes a minimal
* update can replace independently. Headings, paragraphs, code fences, table
* rows, list items, block quotes, HTML blocks, thematic breaks, and link
* definitions are units; the container path is part of the kind so kind
* sequences only align when container membership also aligns.
*
* @param markdown - Document text.
* @returns Units in document order.
*/
export function markdownUnits(markdown: string): MarkdownSpan[] {
const positions: Array<{ kind: string; label: string; startLine: number; endLine: number }> = []
const visit = (node: Nodes, path: string): void => {
let kind: string | undefined
switch (node.type) {
case 'heading':
kind = `${path}:heading:${node.depth}`
break
case 'paragraph':
case 'code':
case 'tableRow':
case 'listItem':
case 'blockquote':
case 'html':
case 'thematicBreak':
case 'definition':
kind = `${path}:${node.type}`
break
default:
break
}
if (kind !== undefined && node.position !== undefined) {
positions.push({ kind, label: node.type, startLine: node.position.start.line, endLine: node.position.end.line })
return
}
if ('children' in node) for (const [index, child] of node.children.entries()) visit(child, `${path}.${index}`)
}
visit(parseTranslationMarkdown(markdown), 'root')
positions.sort((left, right) => left.startLine - right.startLine)
const lines = linesOf(markdown)
return positions.map((position, index) => ({
index,
...position,
text: sliceLines(lines, position.startLine, position.endLine),
}))
}
/**
* List a document's heading-delimited sections, including a leading
* `preamble` span when content precedes the first heading.
*
* @param markdown - Document text.
* @returns Sections in document order.
*/
export function sectionSpans(markdown: string): MarkdownSpan[] {
const headings: Array<{ depth: number; line: number; label: string }> = []
const visit = (node: Nodes): void => {
if (node.type === 'heading' && node.position !== undefined) {
let label = ''
const collect = (child: Nodes): void => {
if ('value' in child && typeof child.value === 'string') label += child.value
if ('children' in child) for (const grandchild of child.children) collect(grandchild)
}
for (const child of node.children) collect(child)
headings.push({ depth: node.depth, line: node.position.start.line, label })
}
if ('children' in node) for (const child of node.children) visit(child)
}
visit(parseTranslationMarkdown(markdown))
headings.sort((left, right) => left.line - right.line)
const lines = linesOf(markdown)
const spans: MarkdownSpan[] = []
const firstHeadingLine = headings[0]?.line ?? lines.length + 1
if (firstHeadingLine > 1) {
spans.push({ index: 0, kind: 'preamble', label: '(preamble before the first heading)', startLine: 1, endLine: firstHeadingLine - 1, text: sliceLines(lines, 1, firstHeadingLine - 1) })
}
for (const [order, heading] of headings.entries()) {
const endLine = (headings[order + 1]?.line ?? lines.length + 1) - 1
spans.push({
index: spans.length,
// Depth only: heading TEXT is translated across a pair, so it cannot
// participate in cross-language alignment.
kind: `section:${heading.depth}`,
label: heading.label === '' ? '(untitled section)' : heading.label,
startLine: heading.line,
endLine,
text: sliceLines(lines, heading.line, endLine),
})
}
return spans
}
/**
* Whether two span lists map one to one: same non-zero length and the same
* kind at every position.
*
* @param left - One document's spans.
* @param right - The other document's spans.
* @returns True when index-wise mapping is sound.
*/
export function spansAligned(left: MarkdownSpan[], right: MarkdownSpan[]): boolean {
return left.length > 0
&& left.length === right.length
&& left.every((span, index) => span.kind === right[index]?.kind)
}
/**
* Indices whose text differs between two aligned span lists.
*
* @param before - Spans of the earlier state.
* @param after - Spans of the later state, aligned with `before`.
* @returns Ascending changed indices.
*/
export function changedSpanIndices(before: MarkdownSpan[], after: MarkdownSpan[]): number[] {
return before.filter((span, index) => span.text !== after[index]?.text).map(span => span.index)
}
function codeSpansOf(markdown: string): MarkdownSpan[] {
return markdownUnits(markdown).filter(span => span.kind.endsWith(':code'))
.map((span, index) => ({ ...span, index }))
}
function replaceSpanTexts(markdown: string, spans: MarkdownSpan[], replacements: Map<number, string>): string {
const lines = linesOf(markdown)
for (const [index, replacement] of [...replacements.entries()].sort((left, right) => right[0] - left[0])) {
const span = spans[index]
if (span === undefined) throw new Error(`translation brief: unknown replacement span ${index}`)
lines.splice(span.startLine - 1, span.endLine - span.startLine + 1, ...linesOf(replacement))
}
return `${lines.join('\n')}\n`
}
function maskCodeSpans(markdown: string, spans: MarkdownSpan[]): string {
return replaceSpanTexts(markdown, spans, new Map(spans.map(span => [span.index, `DSH_TRANSLATION_CODE_${span.index}\n`])))
}
/**
* Compute the counterpart update for a change confined to fenced code
* blocks. Fences are byte-identical across a pair, so when the source's
* prose is untouched and the counterpart's fences match the last-confirmed
* source, splicing the edited fences into the counterpart is the complete
* update — no translation judgment is involved.
*
* @param confirmedSource - The changed side's last-confirmed text.
* @param currentSource - The changed side's current text.
* @param counterpart - The other side's current text.
* @returns The updated counterpart, or undefined when the change is not code-only.
*/
export function computeMechanicalUpdate(confirmedSource: string, currentSource: string, counterpart: string): string | undefined {
const confirmed = codeSpansOf(confirmedSource)
const current = codeSpansOf(currentSource)
const target = codeSpansOf(counterpart)
if (confirmed.length === 0 || confirmed.length !== current.length || confirmed.length !== target.length) return undefined
if (maskCodeSpans(confirmedSource, confirmed) !== maskCodeSpans(currentSource, current)) return undefined
if (confirmed.some((span, index) => span.text !== target[index]?.text)) return undefined
const changed = current.filter((span, index) => span.text !== confirmed[index]?.text)
if (changed.length === 0) return undefined
return replaceSpanTexts(counterpart, target, new Map(changed.map(span => [span.index, span.text])))
}
/** One parsed terminology-table data row. */
export interface TerminologyRow {
english: string
chinese: string
/** The 首次出现 cell (first-occurrence rendering), possibly empty. */
first: string
/** The verbatim table row. */
line: string
}
/** Strip Markdown emphasis and code markers from a terminology cell. */
function plainTerm(cell: string): string {
return cell.replaceAll('`', '').replaceAll('**', '').trim()
}
/**
* Parse the data rows of the terminology table.
*
* @param terminology - Full `docs/i18n/terminology.md` contents.
* @returns Rows in table order.
*/
export function parseTerminologyRows(terminology: string): TerminologyRow[] {
const rows: TerminologyRow[] = []
for (const line of terminology.split('\n')) {
if (!line.startsWith('|')) continue
if (/^\|[\s:|-]+\|$/.test(line)) continue
const cells = line.split('|').map(cell => cell.trim())
const english = plainTerm(cells[1] ?? '')
if (english === '' || english === 'English') continue
rows.push({ english, chinese: plainTerm(cells[2] ?? ''), first: plainTerm(cells[3] ?? ''), line })
}
return rows
}
/**
* Character offsets of a term's occurrences. English word-like terms match
* on word boundaries and accept plural inflections (`agents`, `registries`);
* other terms match as case-insensitive substrings.
*
* @param text - Text to search.
* @param term - The term to find.
* @param englishInflections - Whether to accept English plural forms.
* @returns Ascending match offsets.
*/
export function termOffsets(text: string, term: string, englishInflections = false): number[] {
if (term === '') return []
const escape = (value: string): string => value.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')
const wordLike = /^[A-Za-z0-9][A-Za-z0-9 ._-]*[A-Za-z0-9]$/.test(term)
const inflected = englishInflections && wordLike
? /[^aeiou]y$/i.test(term)
? `${escape(term.slice(0, -1))}(?:y|ies)`
: `${escape(term)}(?:s|es)?`
: escape(term)
const expression = new RegExp(wordLike ? `(?<![A-Za-z0-9_])${inflected}(?![A-Za-z0-9_])` : inflected, 'gi')
return [...text.matchAll(expression)].map(match => match.index)
}
/** The two update directions a pair supports. */
export type BriefDirection = 'en-to-zh' | 'zh-to-en'
/** Whether a row's source-language term occurs in the given text. */
function rowOccurs(row: TerminologyRow, direction: BriefDirection, text: string): boolean {
const terms = direction === 'en-to-zh' ? [row.english] : [row.first, row.chinese].filter(term => /[一-鿿]/.test(term))
return terms.some(term => termOffsets(text, term, direction === 'en-to-zh').length > 0)
}
/**
* Select the terminology rows whose source-language term occurs in the
* changed text (old and new states combined).
*
* @param terminology - Full `docs/i18n/terminology.md` contents.
* @param direction - Update direction; decides which columns to match.
* @param changedText - Concatenated old and new text of the changed spans.
* @returns Matched rows in table order.
*/
export function relevantTerminologyRows(terminology: string, direction: BriefDirection, changedText: string): TerminologyRow[] {
return parseTerminologyRows(terminology).filter(row => rowOccurs(row, direction, changedText))
}
function lineAtOffset(text: string, offset: number): number {
return text.slice(0, offset).split('\n').length
}
function spanIndexAtOffset(text: string, spans: MarkdownSpan[], offset: number | undefined): number | undefined {
if (offset === undefined) return undefined
const line = lineAtOffset(text, offset)
return spans.find(span => line >= span.startLine && line <= span.endLine)?.index
}
/** First-occurrence guidance computed for a Chinese-target update. */
export interface FirstOccurrenceContext {
/** Human-readable notes for the briefing. */
notes: string[]
/** Unchanged span indices that must join the briefing because a first occurrence moved into or out of them. */
extraSpanIndices: number[]
}
/**
* Track document-wide first occurrences of the relevant English terms. The
* 首次出现 rendering attaches to a term's first occurrence, so when an edit
* moves that occurrence across spans, both the old and new spans need
* counterpart edits even when only one of them changed.
*
* @param confirmedSource - Last-confirmed English text.
* @param currentSource - Current English text.
* @param confirmedSpans - Spans of the last-confirmed English text.
* @param currentSpans - Spans of the current English text, aligned with `confirmedSpans`.
* @param rows - The relevant terminology rows.
* @param changed - Span indices already in the briefing.
* @returns Notes and extra span indices to include.
*/
export function firstOccurrenceContext(
confirmedSource: string,
currentSource: string,
confirmedSpans: MarkdownSpan[],
currentSpans: MarkdownSpan[],
rows: TerminologyRow[],
changed: Set<number>,
): FirstOccurrenceContext {
const notes: string[] = []
const extra = new Set<number>()
for (const row of rows) {
if (row.first === '') continue
const oldIndex = spanIndexAtOffset(confirmedSource, confirmedSpans, termOffsets(confirmedSource, row.english, true)[0])
const newIndex = spanIndexAtOffset(currentSource, currentSpans, termOffsets(currentSource, row.english, true)[0])
if (oldIndex === newIndex) continue
for (const index of [oldIndex, newIndex]) {
if (index !== undefined && !changed.has(index)) extra.add(index)
}
notes.push(`${row.english}: the document-wide first occurrence moved from ${oldIndex === undefined ? 'absent' : `#${oldIndex}`} to ${newIndex === undefined ? 'absent' : `#${newIndex}`}; the ${row.first} form moves with it (later occurrences drop the annotation).`)
}
return { notes, extraSpanIndices: [...extra].sort((left, right) => left - right) }
}
/** Smallest fence of `mark` characters that safely wraps `body`. */
function fenceFor(body: string, mark: '`' | '~'): string {
let longest = 2
for (const line of body.split('\n')) {
const run = new RegExp(`^\\s*(${mark === '`' ? '`' : '~'}{3,})`).exec(line)
if (run?.[1] !== undefined && run[1].length > longest) longest = run[1].length
}
return mark.repeat(longest + 1)
}
/** One changed (or first-occurrence) span with its three-way context. */
export interface BriefBundle {
/** Span index shared by the aligned documents. */
index: number
/** Human label: heading text or node type. */
label: string
/** Why the bundle is present when its source text did not change. */
reason?: 'first-occurrence' | undefined
confirmedSourceText: string
currentSourceText: string
counterpartText: string
/** 1-based line the counterpart span starts on. */
counterpartStartLine: number
}
/** The granularities a briefing can map the change at, narrowest first. */
export type BriefScope =
| { kind: 'mechanical' }
| { kind: 'units'; bundles: BriefBundle[]; firstOccurrenceNotes: string[] }
| { kind: 'sections'; bundles: BriefBundle[]; firstOccurrenceNotes: string[] }
| { kind: 'document'; reason: string }
/** Inputs for rendering one pair's briefing. */
export interface TranslationBriefInput {
/** Repo-relative path of the side that changed. */
sourcePath: string
/** Repo-relative path of the counterpart to update. */
counterpartPath: string
direction: BriefDirection
/** Unified diff of the changed side, last-confirmed to current. */
diff: string
scope: BriefScope
terminology: TerminologyRow[]
}
const ZH_TARGET_DIGEST = [
'- Edit ONLY what the change requires; preserve the reviewed phrasing of everything unchanged.',
'- Nothing added, nothing dropped: the Chinese must state exactly what the new English states.',
'- Write natural institutional technical Chinese, not word-by-word gloss; terse stays terse.',
'- Code fences byte-identical to the English side, comments included; inline code spans verbatim.',
'- Relative links keep the `.md` target; only the switcher line links `.zh.md`.',
'- Structure mirrors the counterpart: heading depths and order, list kinds and item counts, table rows and columns.',
'- 首次出现 annotations attach to the document-wide first occurrence only; later occurrences use the bare form, and an empty 首次出现 cell means never gloss.',
'- Typography: one half-width space between Chinese and Latin or digits; full-width punctuation in Chinese prose; 顿号 for enumerations; second person is 你.',
'- One physical line per paragraph; exactly one trailing newline.',
]
const EN_TARGET_DIGEST = [
'- Edit ONLY what the change requires; preserve the reviewed phrasing of everything unchanged.',
'- Nothing added, nothing dropped: the English must state exactly what the new Chinese states.',
'- Write concise professional developer prose, not word-by-word gloss; terse stays terse.',
'- Code fences byte-identical to the Chinese side, comments included; inline code spans verbatim.',
'- Relative links keep the `.md` target; only the switcher line links `.zh.md`.',
'- Structure mirrors the counterpart: heading depths and order, list kinds and item counts, table rows and columns.',
'- One physical line per paragraph; exactly one trailing newline.',
]
function renderBundles(out: string[], input: TranslationBriefInput, bundles: BriefBundle[], firstOccurrenceNotes: string[]): void {
const sourceLanguage = input.direction === 'en-to-zh' ? 'English' : 'Chinese'
const counterpartLanguage = input.direction === 'en-to-zh' ? 'Chinese' : 'English'
for (const bundle of bundles) {
out.push('')
out.push(`### #${bundle.index} ${bundle.label}${bundle.reason === 'first-occurrence' ? ' — unchanged; included for a first-occurrence move' : ''} — counterpart at ${input.counterpartPath}:${bundle.counterpartStartLine}`)
const fence = fenceFor([bundle.confirmedSourceText, bundle.currentSourceText, bundle.counterpartText].join('\n'), '~')
if (bundle.confirmedSourceText !== bundle.currentSourceText) {
out.push('')
out.push(`Last-confirmed ${sourceLanguage}:`)
out.push('')
out.push(`${fence}markdown`)
out.push(bundle.confirmedSourceText.trimEnd())
out.push(fence)
}
out.push('')
out.push(`Current ${sourceLanguage}:`)
out.push('')
out.push(`${fence}markdown`)
out.push(bundle.currentSourceText.trimEnd())
out.push(fence)
out.push('')
out.push(`Current ${counterpartLanguage} (bring this along):`)
out.push('')
out.push(`${fence}markdown`)
out.push(bundle.counterpartText.trimEnd())
out.push(fence)
}
if (firstOccurrenceNotes.length > 0) {
out.push('')
out.push('## First-occurrence notes')
out.push('')
for (const note of firstOccurrenceNotes) out.push(`- ${note}`)
}
}
/**
* Render the complete briefing for one out-of-sync pair.
*
* @param input - Diff, mapped scope, terminology, and pair identity.
* @returns Markdown briefing text.
*/
export function renderTranslationBrief(input: TranslationBriefInput): string {
const sourceLanguage = input.direction === 'en-to-zh' ? 'English' : 'Chinese'
const counterpartLanguage = input.direction === 'en-to-zh' ? 'Chinese' : 'English'
const out: string[] = []
out.push(`# Translation update briefing: ${input.sourcePath}`)
out.push('')
out.push(`The ${sourceLanguage} side changed; bring \`${input.counterpartPath}\` along with the smallest edit that covers the change.`)
if (input.scope.kind === 'mechanical') {
out.push('')
out.push('## Mechanical update — no translation judgment involved')
out.push('')
out.push(`Every change since the last confirmed state is inside fenced code blocks, which are byte-identical across the pair. Run \`pnpm run gen-translation-brief --apply ${input.sourcePath}\` to splice the updated fences into the counterpart (the result is structure-validated before writing), then record per the Finish steps.`)
}
out.push('')
out.push(`## ${sourceLanguage} diff (last-confirmed → current)`)
out.push('')
const diffFence = fenceFor(input.diff, '`')
out.push(`${diffFence}diff`)
out.push(input.diff.trimEnd())
out.push(diffFence)
switch (input.scope.kind) {
case 'mechanical':
break
case 'units':
out.push('')
out.push(`## Changed units (last-confirmed ${sourceLanguage} → current ${sourceLanguage}, with the current ${counterpartLanguage})`)
renderBundles(out, input, input.scope.bundles, input.scope.firstOccurrenceNotes)
break
case 'sections':
out.push('')
out.push('## Changed sections (fine-grained units do not align across the pair; whole heading sections shown)')
renderBundles(out, input, input.scope.bundles, input.scope.firstOccurrenceNotes)
break
case 'document':
out.push('')
out.push('## Whole-document update required')
out.push('')
out.push(`${input.scope.reason} Open \`${input.counterpartPath}\` directly, locate the affected regions yourself, and reconcile under docs/i18n/translation-rules.md.`)
break
default:
input.scope satisfies never
}
if (input.terminology.length > 0) {
out.push('')
out.push('## Binding terminology rows matching this change (docs/i18n/terminology.md)')
out.push('')
out.push('| English | 中文 | 首次出现 | 不要译作 | 备注 |')
out.push('|---|---|---|---|---|')
for (const row of input.terminology) out.push(row.line)
out.push('')
out.push('For any term you introduce that is not listed above, consult the full table before inventing a rendering.')
}
out.push('')
out.push('## Rules digest (full rules: docs/i18n/translation-rules.md)')
out.push('')
out.push(...(input.direction === 'en-to-zh' ? ZH_TARGET_DIGEST : EN_TARGET_DIGEST))
out.push('')
out.push('## Finish')
out.push('')
out.push('1. Apply the smallest counterpart edit that covers the change, then verify the changed spans clause by clause against the source.')
out.push(`2. \`pnpm run verify-translation-pairing --write ${input.sourcePath.replace(/\.zh\.md$/, '.md')}\``)
out.push(`3. \`pnpm run verify-translation-pairing ${input.sourcePath.replace(/\.zh\.md$/, '.md')}\``)
out.push('')
return out.join('\n')
}

View File

@@ -3,7 +3,9 @@
import { describe, expect, it } from 'vitest'
import {
isTranslationScopeFile,
pairAnchorOfArgument,
parseTranslationMarkdown,
parseTranslationPairingCliArgs,
parseTranslationPairingManifest,
translationStructureDiff,
translationStructureSignature,
@@ -102,3 +104,40 @@ describe('translation structural signature', () => {
])
})
})
describe('pair CLI arguments', () => {
it('normalizes any pair file or bare stem to the English anchor', () => {
expect(pairAnchorOfArgument('docs/foo.md')).toBe('docs/foo.md')
expect(pairAnchorOfArgument('docs/foo.zh.md')).toBe('docs/foo.md')
expect(pairAnchorOfArgument('docs/foo.i18n.yaml')).toBe('docs/foo.md')
expect(pairAnchorOfArgument('docs/foo')).toBe('docs/foo.md')
expect(pairAnchorOfArgument('.\\docs\\foo.zh.md')).toBe('docs/foo.md')
})
it('scopes a check to named pairs and dedupes the three spellings', () => {
expect(parseTranslationPairingCliArgs(['docs/foo.zh.md', 'docs/foo.i18n.yaml', 'docs/bar.md'])).toEqual({
mode: 'check',
scope: 'pairs',
anchors: ['docs/bar.md', 'docs/foo.md'],
})
expect(parseTranslationPairingCliArgs([])).toEqual({ mode: 'check', scope: 'corpus', anchors: [] })
})
it('requires --write to name confirmed pairs or opt into --all', () => {
expect(() => parseTranslationPairingCliArgs(['--write'])).toThrow('requires the pair(s) you confirmed')
expect(parseTranslationPairingCliArgs(['--write', 'docs/foo.md'])).toEqual({
mode: 'write',
scope: 'pairs',
anchors: ['docs/foo.md'],
})
expect(parseTranslationPairingCliArgs(['--write', '--all'])).toEqual({ mode: 'write', scope: 'corpus', anchors: [] })
expect(() => parseTranslationPairingCliArgs(['--write', '--all', 'docs/foo.md'])).toThrow('not both')
})
it('keeps --list corpus-only and rejects unknown flags', () => {
expect(parseTranslationPairingCliArgs(['--list'])).toEqual({ mode: 'list', scope: 'corpus', anchors: [] })
expect(() => parseTranslationPairingCliArgs(['--list', 'docs/foo.md'])).toThrow('takes no other flags or paths')
expect(() => parseTranslationPairingCliArgs(['--all'])).toThrow('--all only applies to --write')
expect(() => parseTranslationPairingCliArgs(['--frobnicate'])).toThrow('unknown flag(s): --frobnicate')
})
})

View File

@@ -102,6 +102,66 @@ export function parseTranslationPairingManifest(content: string): TranslationPai
return { excluded: excludedField(record) }
}
/**
* Normalize one CLI pair argument to its English anchor path: any of the
* pair's three files (`foo.md`, `foo.zh.md`, `foo.i18n.yaml`) or the bare
* `foo` stem names the same pair, and platform separators are accepted.
*
* @param argument - Repo-relative path as passed on a command line.
* @returns The pair's `foo.md` anchor path with `/` separators.
*/
export function pairAnchorOfArgument(argument: string): string {
const normalized = argument.split('\\').join('/').replace(/^\.\//, '')
if (normalized.endsWith('.zh.md')) return `${normalized.slice(0, -'.zh.md'.length)}.md`
if (normalized.endsWith('.i18n.yaml')) return `${normalized.slice(0, -'.i18n.yaml'.length)}.md`
if (normalized.endsWith('.md')) return normalized
return `${normalized}.md`
}
/** A parsed `verify-translation-pairing` invocation. */
export interface TranslationPairingCliRequest {
mode: 'check' | 'list' | 'write'
/** `corpus` runs discovery over the whole tree; `pairs` touches only the named anchors. */
scope: 'corpus' | 'pairs'
/** English anchor paths, empty for corpus scope. */
anchors: string[]
}
/**
* Parse and validate `verify-translation-pairing` CLI arguments.
*
* Check accepts optional pair paths; `--write` requires either pair paths or
* `--all` so a bulk re-record is always an explicit choice — a bare
* `--write` would silently bless every drifted pair in the tree, including
* ones the caller never confirmed. `--list` is corpus-only.
*
* @param argv - Arguments after the script name.
* @returns The validated request.
* @throws Error when flags or their combination are invalid.
*/
export function parseTranslationPairingCliArgs(argv: string[]): TranslationPairingCliRequest {
const flags = argv.filter(argument => argument.startsWith('--'))
const anchors = [...new Set(argv.filter(argument => !argument.startsWith('--')).map(pairAnchorOfArgument))].sort()
const unknown = flags.filter(flag => !['--list', '--write', '--all'].includes(flag))
if (unknown.length > 0) throw new Error(`unknown flag(s): ${unknown.join(', ')}`)
const listMode = flags.includes('--list')
const writeMode = flags.includes('--write')
const allMode = flags.includes('--all')
if (listMode && (writeMode || allMode || anchors.length > 0)) {
throw new Error('--list reports the whole corpus and takes no other flags or paths')
}
if (allMode && !writeMode) throw new Error('--all only applies to --write')
if (writeMode) {
if (anchors.length > 0 && allMode) throw new Error('--write takes either pair paths or --all, not both')
if (anchors.length === 0 && !allMode) {
throw new Error('--write requires the pair(s) you confirmed (any file of a pair), or --all to re-record every complete pair; recording pairs you did not review blesses unconfirmed content')
}
return { mode: 'write', scope: allMode ? 'corpus' : 'pairs', anchors }
}
if (listMode) return { mode: 'list', scope: 'corpus', anchors: [] }
return { mode: 'check', scope: anchors.length > 0 ? 'pairs' : 'corpus', anchors }
}
/** The structural surface compared between the two sides of a pair. */
export interface TranslationStructureSignature {
/** Heading depths in document order (h2 -> 2). */

View File

@@ -46,6 +46,26 @@
"symbol": "LlmModelContext",
"source": "packages/llm/llm/src/types.ts"
},
{
"doc": "docs/core-data-structures/core.md",
"symbol": "ReasoningEffortId",
"source": "packages/llm/llm/src/brand.ts"
},
{
"doc": "docs/core-data-structures/core.md",
"symbol": "LlmReasoningEffortInfo",
"source": "packages/llm/llm/src/types.ts"
},
{
"doc": "docs/core-data-structures/core.md",
"symbol": "LlmModelReasoningInfo",
"source": "packages/llm/llm/src/types.ts"
},
{
"doc": "docs/core-data-structures/core.md",
"symbol": "LlmResolvedModelInfo",
"source": "packages/llm/llm/src/types.ts"
},
{
"doc": "docs/core-data-structures/core.md",
"symbol": "GenerateOptions",
@@ -292,6 +312,11 @@
"source": "packages/llm/llm/src/assembler.ts",
"projection": "public-api"
},
{
"doc": "docs/core-data-structures/llm-streaming.md",
"symbol": "PreparedLlmCall",
"source": "packages/llm/llm/src/index.ts"
},
{
"doc": "docs/core-data-structures/llm-streaming.md",
"symbol": "LlmAdapter",

View File

@@ -13,11 +13,15 @@ import {
} from 'node:fs'
import { dirname, resolve } from 'node:path'
import { pathToFileURL } from 'node:url'
import { parseArgs } from 'node:util'
const repositoryRoot = resolve(import.meta.dirname, '..')
const options = parseOptions(process.argv.slice(2))
const packagesRoot = resolve(options.get('--packages-root') ?? repositoryRoot)
const loaderUrl = options.get('--loader-url')
const { values: options } = parseArgs({
args: process.argv.slice(2),
options: { 'packages-root': { type: 'string' }, 'loader-url': { type: 'string' } },
})
const packagesRoot = resolve(options['packages-root'] ?? repositoryRoot)
const loaderUrl = options['loader-url']
?? pathToFileURL(resolve(repositoryRoot, 'vendor/loader/lib/index.js')).href
const failures = []
const manifests = globSync('packages/*/*/package.json', { cwd: packagesRoot }).sort()
@@ -77,21 +81,6 @@ if (failures.length > 0) {
console.log(`verify-built-package-invariants: ${manifests.length} compiled companion(s) passed plain-Node Loader checks.`)
function parseOptions(args) {
const allowed = new Set(['--packages-root', '--loader-url'])
const parsed = new Map()
for (let index = 0; index < args.length; index += 2) {
const name = args[index]
const value = args[index + 1]
if (!allowed.has(name) || value === undefined || value.startsWith('--')) {
throw new Error(`verify-built-package-invariants: expected [--packages-root PATH] [--loader-url URL], got ${JSON.stringify(args)}.`)
}
if (parsed.has(name)) throw new Error(`verify-built-package-invariants: duplicate option ${name}.`)
parsed.set(name, value)
}
return parsed
}
function copyDeclaredLibFiles(packageDir, stagedPackageDir, files) {
for (const pattern of files) {
if (!pattern.startsWith('lib/')) continue

View File

@@ -14,8 +14,8 @@
* pnpm exec tsx scripts/verify-client-domain-graph.ts
*/
import { readdirSync, readFileSync, statSync } from 'node:fs'
import { join, resolve } from 'node:path'
import { globSync, readdirSync, readFileSync, statSync } from 'node:fs'
import { join, resolve, sep } from 'node:path'
const root = resolve(import.meta.dirname, '..')
const CLIENT_DIR = join(root, 'packages/client')
@@ -28,15 +28,11 @@ const ASSEMBLY_FILES = new Set(['apply.ts', 'index.ts', 'index.tsx'])
interface Violation { file: string; imported: string; reason: string }
/** Recursively list .ts/.tsx files under dir (relative paths). */
function listSources(dir: string, prefix = ''): string[] {
const out: string[] = []
for (const name of readdirSync(dir)) {
const full = join(dir, name)
const rel = prefix ? `${prefix}/${name}` : name
if (statSync(full).isDirectory()) out.push(...listSources(full, rel))
else if (/\.tsx?$/.test(name) && !/\.legacy\./.test(name)) out.push(rel)
}
return out
function listSources(dir: string): string[] {
return globSync('**/*.{ts,tsx}', { cwd: dir })
.map(rel => rel.split(sep).join('/'))
.filter(rel => !/\.legacy\./.test(rel.slice(rel.lastIndexOf('/') + 1)))
.sort()
}
/** First path segment of a client-relative file, or '' for top-level files. */

View File

@@ -5,7 +5,7 @@
* outside the check.
*/
import { existsSync, readdirSync } from 'node:fs'
import { existsSync, globSync } from 'node:fs'
import { resolve } from 'node:path'
import {
findReferenceViolations,
@@ -41,12 +41,8 @@ const isExcluded = (p: string): boolean =>
*/
function realPackageNames(): Set<string> {
const names = new Set<string>()
const pkgRoot = resolve(root, 'packages')
for (const group of readdirSync(pkgRoot, { withFileTypes: true })) {
if (!group.isDirectory()) continue
for (const pkg of readdirSync(resolve(pkgRoot, group.name), { withFileTypes: true })) {
if (pkg.isDirectory()) names.add(pkg.name)
}
for (const pkg of globSync('packages/*/*', { cwd: root, withFileTypes: true })) {
if (pkg.isDirectory()) names.add(pkg.name)
}
return names
}

View File

@@ -3,8 +3,9 @@
* peer in its dependency graph. With auto peer installation disabled, a missing
* root peer can otherwise fail only when Cordis loads the packaged plugin.
*/
import { readFile, readdir } from 'node:fs/promises'
import { join, resolve } from 'node:path'
import { globSync } from 'node:fs'
import { readFile } from 'node:fs/promises'
import { resolve } from 'node:path'
import { parseArgs } from 'node:util'
interface PackageManifest {
@@ -72,15 +73,9 @@ if (failures.length > 0) {
console.log(`verify-runtime-closure: ${queue.length} workspace packages form a closed runtime dependency graph.`)
async function loadWorkspacePackages(): Promise<Map<string, WorkspacePackage>> {
const paths: string[] = []
for (const group of await childDirectories(join(root, 'packages'))) {
for (const packageDir of await childDirectories(join(root, 'packages', group))) {
paths.push(join(root, 'packages', group, packageDir, 'package.json'))
}
}
for (const packageDir of await childDirectories(join(root, 'vendor'))) {
paths.push(join(root, 'vendor', packageDir, 'package.json'))
}
const paths = globSync(['packages/*/*/package.json', 'vendor/*/package.json'], { cwd: root })
.sort()
.map(relative => resolve(root, relative))
const result = new Map<string, WorkspacePackage>()
for (const path of paths) {
const manifest = await loadManifest(path)
@@ -89,11 +84,6 @@ async function loadWorkspacePackages(): Promise<Map<string, WorkspacePackage>> {
return result
}
async function childDirectories(path: string): Promise<string[]> {
const entries = await readdir(path, { withFileTypes: true })
return entries.filter(entry => entry.isDirectory()).map(entry => entry.name).sort()
}
async function loadManifest(path: string): Promise<PackageManifest> {
return JSON.parse(await readFile(path, 'utf8')) as PackageManifest
}

View File

@@ -2,8 +2,10 @@
* Enforce complete English/Chinese pairs, matching structure, and recorded git
* blob hashes for every in-scope document. The manifest contains only explicit
* exclusions, which may have neither a counterpart nor a sidecar.
* `--list` reports state and `--write` records both sides after human review.
* Translation quality remains a review responsibility.
* `--list` reports state; `--write <pairs...>` records the named confirmed
* pairs (`--write --all` records every complete pair); a check or write named
* with pair paths touches only those pairs, so update iteration does not pay
* for a corpus scan. Translation quality remains a review responsibility.
* See `docs/i18n/README.md` for the owning contract.
*/
@@ -13,6 +15,7 @@ import { basename, join, resolve, sep } from 'node:path'
import {
linksTo,
parseTranslationMarkdown,
parseTranslationPairingCliArgs,
parseTranslationPairingManifest,
isTranslationScopeFile,
TRANSLATION_SCOPE_GLOB_EXCLUDES,
@@ -21,8 +24,15 @@ import {
} from './translation-pairing.ts'
const root = resolve(import.meta.dirname, '..')
const listMode = process.argv.includes('--list')
const writeMode = process.argv.includes('--write')
let request: ReturnType<typeof parseTranslationPairingCliArgs>
try {
request = parseTranslationPairingCliArgs(process.argv.slice(2))
} catch (error) {
console.error(`verify-translation-pairing: ${error instanceof Error ? error.message : String(error)}`)
process.exit(2)
}
const listMode = request.mode === 'list'
const writeMode = request.mode === 'write'
/** Discover source Markdown and pairing sidecars before applying the corpus predicate. */
const SCOPE_PATTERNS = [
@@ -77,32 +87,67 @@ function renderMeta(source: string, sourceHash: string, zh: string, zhHash: stri
'# Bilingual-pair consistency record (docs/i18n/README.md): the git blob hash of each',
'# side as of the last confirmed-consistent state. Both languages carry equal authority;',
'# after editing either side, bring the other along and re-record with:',
'# pnpm run verify-translation-pairing --write',
`# pnpm run verify-translation-pairing --write ${source}`,
`${basename(source)}: ${sourceHash}`,
`${basename(zh)}: ${zhHash}`,
'',
].join('\n')
}
// Enumerate the scope once.
// Enumerate the scope once: the whole corpus, or exactly the named pairs'
// three files (a named pair whose files are absent is caught by the same
// completeness rules that cover discovered remnants).
const files = new Set<string>()
for (const pattern of SCOPE_PATTERNS) {
for (const match of globSync(pattern, { cwd: root, exclude: TRANSLATION_SCOPE_GLOB_EXCLUDES })) {
const normalized = match.split(sep).join('/')
if (isTranslationScopeFile(normalized)) files.add(normalized)
if (request.scope === 'pairs') {
for (const anchor of request.anchors) {
for (const file of [anchor, ...Object.values(pairPaths(anchor))]) {
if (existsSync(join(root, file))) files.add(file)
}
// A named anchor with no files on disk still enters the source list so
// the check reports it instead of silently passing an empty scope.
if (!existsSync(join(root, anchor))) files.add(anchor)
}
} else {
for (const pattern of SCOPE_PATTERNS) {
for (const match of globSync(pattern, { cwd: root, exclude: TRANSLATION_SCOPE_GLOB_EXCLUDES })) {
const normalized = match.split(sep).join('/')
if (isTranslationScopeFile(normalized)) files.add(normalized)
}
}
}
const translations = [...files].filter(f => f.endsWith('.zh.md')).sort()
const metas = [...files].filter(f => f.endsWith('.i18n.yaml')).sort()
const sources = [...files].filter(f => f.endsWith('.md') && !f.endsWith('.zh.md')).sort()
// --write: (re)record both hashes for every complete pair, creating missing records.
if (request.scope === 'pairs') {
const rejected = request.anchors.filter(anchor => !isTranslationScopeFile(anchor) || isExcluded(anchor))
const absent = request.anchors.filter(anchor => ![anchor, ...Object.values(pairPaths(anchor))].some(file => existsSync(join(root, file))))
if (rejected.length > 0 || absent.length > 0) {
for (const anchor of rejected) {
console.error(`verify-translation-pairing: ${anchor} is not an in-scope pair (excluded or outside the documentation corpus; see docs/i18n/README.md)`)
}
for (const anchor of absent) {
console.error(`verify-translation-pairing: ${anchor} names no pair on disk (none of its three files exist)`)
}
process.exit(2)
}
}
// --write: (re)record both hashes for the requested complete pairs, creating
// missing records. A named pair that cannot be recorded (missing counterpart)
// fails loud; corpus scope (--all) skips pairless sources as before.
if (writeMode) {
let written = 0
for (const source of sources) {
if (isExcluded(source)) continue
const { zh, meta } = pairPaths(source)
if (!existsSync(join(root, zh))) continue
if (!existsSync(join(root, source)) || !existsSync(join(root, zh))) {
if (request.scope === 'pairs') {
console.error(`verify-translation-pairing: cannot record ${source}: missing ${existsSync(join(root, source)) ? zh : source}`)
process.exit(2)
}
continue
}
const record = renderMeta(source, blobHash(readFileSync(join(root, source))), zh, blobHash(readFileSync(join(root, zh))))
if (existsSync(join(root, meta)) && readFileSync(join(root, meta), 'utf8') === record) continue
writeFileSync(join(root, meta), record)
@@ -204,7 +249,9 @@ if (listMode) {
}
if (errors.length === 0) {
console.log(`verify-translation-pairing: ${pairAnchors.size} pair(s) checked across all in-scope documentation, all consistent.`)
console.log(request.scope === 'pairs'
? `verify-translation-pairing: ${pairAnchors.size} named pair(s) consistent; the corpus-wide check still runs in doc-sync.`
: `verify-translation-pairing: ${pairAnchors.size} pair(s) checked across all in-scope documentation, all consistent.`)
process.exit(0)
}

View File

@@ -11,6 +11,7 @@
import { globSync, readFileSync, existsSync } from 'node:fs'
import { resolve, sep } from 'node:path'
import ts from 'typescript'
import { markdownFences } from './markdown.ts'
import { partitionPairedMarkdownDerivatives } from './paired-markdown-derivatives.ts'
import { isArchivedAgentNotePath } from './repo-files.ts'
@@ -81,42 +82,27 @@ function blockSymbol(code: string): string | null {
/** Extract every source-equivalence block from one Markdown file. */
function extractEquivBlocks(docRel: string): EquivBlock[] {
const text = readFileSync(resolve(root, docRel), 'utf8')
const lines = text.split('\n')
const blocks: EquivBlock[] = []
let open: { line: number; body: string[]; projection?: 'public-api' } | null = null
for (let i = 0; i < lines.length; i++) {
const raw = lines[i] ?? ''
const fence = /^```(\s*)(\S.*)?$/.exec(raw)
if (!fence) {
if (open) open.body.push(raw)
continue
for (const fence of markdownFences(readFileSync(resolve(root, docRel), 'utf8'))) {
if (fence.info === 'ts type-equiv public-api') {
throw new Error(`verify-type-equiv: ${docRel}:${fence.line} — use the concise \`ts public-api\` fence`)
}
if (open) {
const code = open.body.join('\n')
const symbol = blockSymbol(code)
if (!symbol) {
throw new Error(`verify-type-equiv: ${docRel}:${open.line} — type-equiv block has no parseable interface/type/class declaration`)
}
blocks.push({
doc: docRel,
line: open.line,
symbol,
code,
...(open.projection === undefined ? {} : { projection: open.projection }),
})
open = null
continue
if (fence.info !== 'ts type-equiv' && fence.info !== 'ts public-api') continue
if (!fence.closed) {
throw new Error(`verify-type-equiv: ${docRel}:${fence.line} — unterminated type-equivalence fence (missing closing \`\`\`)`)
}
const info = (fence[2] ?? '').trim()
if (info === 'ts type-equiv public-api') {
throw new Error(`verify-type-equiv: ${docRel}:${i + 1} — use the concise \`ts public-api\` fence`)
const symbol = blockSymbol(fence.code)
if (symbol === null) {
throw new Error(`verify-type-equiv: ${docRel}:${fence.line} — type-equiv block has no parseable interface/type/class declaration`)
}
if (info === 'ts type-equiv') open = { line: i + 1, body: [] }
if (info === 'ts public-api') open = { line: i + 1, body: [], projection: 'public-api' }
blocks.push({
doc: docRel,
line: fence.line,
symbol,
code: fence.code,
...(fence.info === 'ts public-api' ? { projection: 'public-api' as const } : {}),
})
}
if (open) throw new Error(`verify-type-equiv: ${docRel}:${open.line} — unterminated type-equiv block`)
return blocks
}