Merge remote-tracking branch 'origin/master' into feat/search-presenter

# Conflicts:
#	docs/config-catalog.md
#	docs/cookbook/adding-a-tool.i18n.yaml
#	docs/cookbook/adding-a-tool.md
#	docs/cookbook/adding-a-tool.zh.md
#	docs/cordis-catalog/events.md
#	docs/cordis-catalog/services.md
#	docs/core-data-structures/tools.i18n.yaml
#	docs/core-data-structures/tools.md
#	docs/core-data-structures/tools.zh.md
#	docs/event-producer-consumer.md
#	examples/acp-agent/tests/snapshots/cordis-inspect-jsdoc/session.jsonl
#	packages/cordis/tool-cordis/src/api-catalog.ts
#	packages/core/tools/README.i18n.yaml
#	packages/core/tools/README.md
#	packages/core/tools/README.zh.md
#	packages/core/tools/src/index.ts
#	packages/core/tools/src/presentation.ts
#	packages/fs/tool-fs-search/src/glob.ts
#	packages/fs/tool-fs-search/src/index.ts
#	packages/ui/tui/src/components/transcript.ts
#	packages/ui/tui/tests/tui.spec.ts
This commit is contained in:
Chinesezjc
2026-07-31 11:31:32 +08:00
946 changed files with 28921 additions and 4610 deletions

View File

@@ -3,20 +3,19 @@
* pattern, sorted by modification time. Execution goes through the bash seam
* (`ctx.bash`) with a fixed `rg --files` command — this module owns the
* model-facing schema, argument validation, shell-safe command construction,
* result parsing, retention, and formatting; process concerns (defaulting,
* result parsing, inline sampling, and formatting; process concerns (defaulting,
* scrubbing, kill, backend substitution) stay behind `ctx.bash`.
*
* @module @deepseek-ai/dsh-tool-fs-search/glob
*/
import type { Context } from 'cordis'
import { sep } from 'node:path'
import { defineTool } from '@deepseek-ai/dsh-tools'
import type { GenericCallView, SearchResultView, ToolResult } from '@deepseek-ai/dsh-tools'
import type { RetainedItems } from '@deepseek-ai/dsh-retention'
import type { SpillRef } from '@deepseek-ai/dsh-spill'
import type {} from '@deepseek-ai/dsh-bash'
import type {} from '@deepseek-ai/dsh-system-prompt'
import { retainGlobPaths, runRipgrep, toWorkdirRelative, trySaveFormattedResult } from './search-core.ts'
import { runRipgrep, toWorkdirRelative, trySaveFormattedResult } from './search-core.ts'
import { globSearchMeta, searchViewFromMeta } from './presentation.ts'
import { singleQuote } from './shell-quote.ts'
import { acceptedSurfaceValue } from './surface.ts'
@@ -41,6 +40,8 @@ export const GLOB_VCS_EXCLUDES: readonly string[] = ['.git', '.svn', '.hg', '.bz
/** Resolved glob-tool caps — plugin config after defaulting (see `Config` in index.ts). */
export interface GlobToolCaps {
/** Whether over-cap pages are sampled across top-level entries instead of taking the modification-time head. */
sampleOverCapGlobResults: boolean
/** Max paths retained inline; later paths go to the formatted spill file. */
maxResults: number
/** Max bytes of serialized `presentationMeta`; trailing paths drop past it. */
@@ -101,28 +102,154 @@ export function buildGlobCommand(input: GlobInput): string {
}
/**
* Format the model-facing `glob` result: the retained paths, then — when the
* result was capped — a footer carrying either the formatted-spill recovery
* locator or the could-not-save explanation. The omitted count is a budget fact:
* the search itself completed.
* The inline page of a capped `glob` result, plus how much of the complete
* result's top level it reaches.
*/
export interface GlobSample {
/** Paths to show inline: grouped by top-level entry, modification-time ordered within each group. */
items: string[]
/** Distinct top-level entries the shown paths reach. */
shown: number
/** Distinct top-level entries across the complete result. */
total: number
}
/** Remove the displayed search-root prefix before choosing a top-level group. */
function relativeToSearchRoot(path: string, root: string): string {
if (root === '.') return path.startsWith(`.${sep}`) ? path.slice(2) : path
let rootEnd = root.length
while (rootEnd > 0 && root[rootEnd - 1] === sep) rootEnd -= 1
const trimmedRoot = root.slice(0, rootEnd)
if (trimmedRoot.length === 0) return stripLeadingSeparators(path)
if (path === trimmedRoot) return ''
if (path.startsWith(`${trimmedRoot}${sep}`)) {
return path.slice(trimmedRoot.length + 1)
}
return path
}
/** Strip only separators recognized by the execution platform. */
function stripLeadingSeparators(path: string): string {
let start = 0
while (path[start] === sep) start += 1
return path.slice(start)
}
/**
* The leading path segment of one display path — the top-level entry, relative
* to the search root, that the path sits under. A path with no separator is its
* own top-level entry. Leading separators are stripped first so an absolute path
* (one outside the workdir, which {@link toWorkdirRelative} leaves untouched)
* groups by its first real name instead of collapsing every such path into one
* empty group.
*/
function topLevelSegment(path: string): string {
const trimmed = stripLeadingSeparators(path)
const cut = trimmed.indexOf(sep)
return cut === -1 ? trimmed : trimmed.slice(0, cut)
}
/**
* Choose the inline page of an over-cap result by round-robin across the
* complete result's top-level entries, instead of taking its head.
*
* @param retained - the retention outcome over every discovered path.
* Every top-level entry receives a slot before any receives a second; exhausted
* groups drop out. Group order and order within each group follow `paths`, so a
* flat result reproduces the modification-time head.
*
* @param paths - the complete result, in ripgrep's modification-time order.
* @param maxItems - how many paths the page may hold; the caller has already established it is smaller than `paths`.
* @param root - the search root in the same display-path space as `paths`.
* @returns the page grouped by top-level entry, with the shown/total top-level spread.
*/
export function sampleAcrossTopLevel(paths: readonly string[], maxItems: number, root = '.'): GlobSample {
type ActiveGroup = { key: string; items: string[]; index: number; current: string }
const groups = new Map<string, string[]>()
let active: ActiveGroup[] = []
for (const path of paths) {
const key = topLevelSegment(relativeToSearchRoot(path, root))
const group = groups.get(key)
if (group === undefined) {
const items = [path]
groups.set(key, items)
active.push({ key, items, index: 0, current: path })
} else {
group.push(path)
}
}
const taken = new Map<string, string[]>()
let count = 0
while (active.length > 0 && count < maxItems) {
const nextActive: ActiveGroup[] = []
for (const { key, items, index, current } of active) {
if (count >= maxItems) break
count += 1
const bucket = taken.get(key)
if (bucket === undefined) taken.set(key, [current])
else bucket.push(current)
const nextIndex = index + 1
const nextPath = items[nextIndex]
if (nextPath !== undefined) nextActive.push({ key, items, index: nextIndex, current: nextPath })
}
active = nextActive
}
return { items: [...taken.values()].flat(), shown: taken.size, total: groups.size }
}
/**
* Format a capped sampled page and its complete-result recovery path. A flat
* result keeps the plain footer because its sample is the modification-time head.
*
* @param sample - the inline page and its top-level spread.
* @param seen - how many paths the complete result holds; always more than the page.
* @param spillRef - the saved complete-result reference, or `undefined` when unsaved.
* @returns the model-facing text.
*/
export function formatGlobOutput(retained: RetainedItems<string>, spillRef: SpillRef | undefined): string {
const body = retained.items.join('\n')
if (!retained.truncated) return body
export function formatGlobOutput(sample: GlobSample, seen: number, spillRef: SpillRef | undefined): string {
const basis = sample.total === seen
? '.'
: `, sampled across ${sample.shown} of the ${sample.total} top-level entries this pattern matched instead of taken in modification-time order.`
+ (sample.shown < sample.total ? ' Narrow path to inspect a specific subtree.' : '')
return formatGlobPage(sample.items, seen, spillRef, basis)
}
/** Format one bounded page and the recovery path for its complete sorted result. */
function formatGlobPage(items: readonly string[], seen: number, spillRef: SpillRef | undefined, basis: string): string {
const body = items.join('\n')
const recovery = spillRef !== undefined
? `Full sorted result stored at: ${spillRef.locator}. ${spillRef.retrievalHint}`
: 'The complete result could not be saved; narrow pattern or path to see more.'
return `${body}\n\n(Showing ${retained.kept} of ${retained.seen} paths. ${recovery})`
return `${body}\n\n(Showing ${items.length} of ${seen} paths${basis} ${recovery})`
}
/** Format one already-retained path list for the Native surface. */
function formatRetainedGlob(retained: RetainedItems<string>, spillRef?: SpillRef): string {
if (retained.seen === 0) return 'No files found'
return formatGlobOutput(retained, spillRef)
/** Bound and format one canonical path list for the Native surface relative to its search root. */
function renderGlobPaths(paths: string[], caps: GlobToolCaps, root: string, spillRef?: SpillRef): string {
if (paths.length === 0) return 'No files found'
// A result that fits is shown whole, untouched: modification-time order is the
// tool's contract, and over a complete result it is what answers age questions.
if (paths.length <= caps.maxResults) return paths.join('\n')
if (!caps.sampleOverCapGlobResults) {
return formatGlobPage(paths.slice(0, caps.maxResults), paths.length, spillRef, '.')
}
return formatGlobOutput(sampleAcrossTopLevel(paths, caps.maxResults, root), paths.length, spillRef)
}
/**
* The inline page of paths a completed `glob` card shows, computed the SAME way
* {@link renderGlobPaths} computes its model-facing page so the card and the text
* agree on which paths survived the cap. A result within the cap is shown whole;
* an over-cap result is either the modification-time head or the top-level sample,
* matching the deployment's `sampleOverCapGlobResults`.
*
* @param paths - the complete discovered path list, in modification-time order.
* @param caps - the resolved glob caps (the inline cap and the sampling switch).
* @param root - the search root in the same display-path space as `paths`.
* @returns the inline page and whether the complete result was capped.
*/
function globCardPage(paths: string[], caps: GlobToolCaps, root: string): { items: string[]; truncated: boolean } {
if (paths.length <= caps.maxResults) return { items: paths, truncated: false }
if (!caps.sampleOverCapGlobResults) return { items: paths.slice(0, caps.maxResults), truncated: true }
return { items: sampleAcrossTopLevel(paths, caps.maxResults, root).items, truncated: true }
}
/**
@@ -162,19 +289,32 @@ export function presentGlobResult(_args: { pattern: string; path?: string }, res
* @param caps - the deployment's resolved glob caps (plugin config after defaulting).
*/
export function applyGlobTool(ctx: Context, caps: GlobToolCaps): void {
const overCapGuidance = caps.sampleOverCapGlobResults
? 'while a larger one is sampled across top-level entries, so it spans the tree instead of one subtree.'
: 'while a larger one keeps the modification-time-ordered head.'
ctx.systemPrompt.section({
name: 'tool:glob',
order: 103,
text: 'Use the glob tool — not shell find or ls — to discover files by path pattern. Results are sorted by modification time and include hidden and ignored files.',
text: 'Use the glob tool — not shell find — to discover files by path pattern. A pattern with no "/" matches basenames at any depth, so "*" matches every file in the tree rather than its top level. '
+ `Results are files only, never directories, and include hidden and ignored files: a result that fits comes back in modification-time order, ${overCapGuidance}`,
})
const overCapDescription = caps.sampleOverCapGlobResults
? `a larger result instead returns ${caps.maxResults} paths sampled across top-level entries`
: `a larger result returns the first ${caps.maxResults} paths in modification-time order`
const tool = defineTool({
name: 'glob',
description: 'Find files whose paths match a glob pattern. Returns matching paths sorted by modification time, '
description: 'Find files whose paths match a glob pattern. Returns matching file paths — never directories — '
+ 'including hidden and ignored files (VCS metadata directories are excluded). '
+ `Returns the first ${caps.maxResults} paths inline; a capped result reports where the complete list was saved.`,
+ `Up to ${caps.maxResults} paths come back in modification-time order; ${overCapDescription}, `
+ 'says so, and reports where the complete sorted list was saved. This tool does not enumerate directory entries.',
parameters: {
pattern: { type: 'string', required: true, description: 'Glob pattern to match file paths against (e.g. "**/*.ts", "src/**/*.test.js").' },
pattern: {
type: 'string',
required: true,
description: 'Glob pattern to match file paths against (e.g. "**/*.ts", "src/**/*.test.js"). '
+ 'A pattern with no "/" matches the basename at any depth, so "*" and "*.ts" both search the whole tree; include a separator to anchor the depth.',
},
path: { type: 'string', description: 'Directory to search in. Defaults to the session workspace; a relative path resolves against it.' },
},
timeoutMs: caps.timeoutMs,
@@ -183,16 +323,21 @@ export function applyGlobTool(ctx: Context, caps: GlobToolCaps): void {
type: 'object',
additionalProperties: false,
properties: {
root: { type: 'string', required: true },
paths: { type: 'array', required: true, items: { type: 'string' } },
},
},
render: (_args, value) => [{ type: 'text', text: formatRetainedGlob(retainGlobPaths(value.paths, caps.maxResults)) }],
presentationMeta: (_args, value) => globSearchMeta(retainGlobPaths(value.paths, caps.maxResults), caps.maxMetaBytes),
render: (_args, value) => [{ type: 'text', text: renderGlobPaths(value.paths, caps, value.root) }],
presentationMeta: (_args, value) => {
const page = globCardPage(value.paths, caps, value.root)
return globSearchMeta({ items: page.items, truncated: page.truncated, seen: value.paths.length }, caps.maxMetaBytes)
},
},
async execute(args, exec) {
const input = parseGlobArgs(args)
const run = await runRipgrep(ctx, exec, 'glob', buildGlobCommand(input), caps.rawOutputMaxBytes)
if (run.noMatches) return { paths: [] }
const root = input.path === undefined ? '.' : toWorkdirRelative(input.path, run.workdir)
if (run.noMatches) return { root, paths: [] }
const all: string[] = []
for (const line of run.stdout.split('\n')) {
@@ -200,7 +345,7 @@ export function applyGlobTool(ctx: Context, caps: GlobToolCaps): void {
const displayPath = toWorkdirRelative(line, run.workdir)
all.push(displayPath)
}
return { paths: all }
return { root, paths: all }
},
presentCall: presentGlobCall,
presentResult: presentGlobResult,
@@ -209,14 +354,14 @@ export function applyGlobTool(ctx: Context, caps: GlobToolCaps): void {
ctx.on('tools/post-execute', async (exec, result, next) => {
const decision = await next()
const value = acceptedSurfaceValue(ctx, tool, exec, result, decision) as { paths: string[] } | undefined
const value = acceptedSurfaceValue(ctx, tool, exec, result, decision) as { root: string; paths: string[] } | undefined
if (value === undefined) return decision
const paths = value.paths
if (paths.length <= caps.maxResults) return decision
const spillRef = await trySaveFormattedResult(ctx, exec, 'glob-results.txt', paths.join('\n'))
return {
kind: 'accept',
content: [{ type: 'text', text: formatRetainedGlob(retainGlobPaths(paths, caps.maxResults), spillRef) }],
content: [{ type: 'text', text: renderGlobPaths(paths, caps, value.root, spillRef) }],
...decision.additionalContexts !== undefined ? { additionalContexts: decision.additionalContexts } : {},
}
})

View File

@@ -33,8 +33,8 @@ import { GLOB_MAX_RESULTS, applyGlobTool } from './glob.ts'
import { GREP_MAX_LINE_BYTES, GREP_MAX_MATCHES, applyGrepTool } from './grep.ts'
import { RAW_OUTPUT_MAX_BYTES, SEARCH_META_MAX_BYTES, SEARCH_TIMEOUT_MS } from './search-core.ts'
export { GLOB_MAX_RESULTS, GLOB_VCS_EXCLUDES, applyGlobTool, buildGlobCommand, formatGlobOutput, parseGlobArgs, presentGlobCall, presentGlobResult } from './glob.ts'
export type { GlobInput, GlobToolCaps } from './glob.ts'
export { GLOB_MAX_RESULTS, GLOB_VCS_EXCLUDES, applyGlobTool, buildGlobCommand, formatGlobOutput, parseGlobArgs, presentGlobCall, presentGlobResult, sampleAcrossTopLevel } from './glob.ts'
export type { GlobInput, GlobSample, GlobToolCaps } from './glob.ts'
export {
GREP_MAX_LINE_BYTES,
GREP_MAX_MATCHES,
@@ -67,8 +67,10 @@ export const name = 'tool-fs-search'
/** Services required by the search tool suite (`spillStore` is optional, read via `ctx.get()`). */
export const inject = ['tools', 'systemPrompt', 'bash']
/** Plugin config (all optional — `Config` supplies the defaults). */
/** Plugin config; over-cap glob sampling is an explicit deployment choice and the remaining fields have defaults. */
export interface Config {
/** Whether an over-cap `glob` page is sampled across top-level entries instead of taking the modification-time head. */
sampleOverCapGlobResults: boolean
/** Max paths one `glob` call retains inline; later paths go to the formatted spill file. */
globMaxResults?: number
/** Max flat matches one `grep` call retains inline; later matches go to the formatted spill file. */
@@ -84,6 +86,7 @@ export interface Config {
}
export const Config: z<Config> = z.object({
sampleOverCapGlobResults: z.boolean().required(),
globMaxResults: z.number().default(GLOB_MAX_RESULTS),
grepMaxMatches: z.number().default(GREP_MAX_MATCHES),
grepMaxLineBytes: z.number().default(GREP_MAX_LINE_BYTES),
@@ -150,6 +153,7 @@ export async function apply(ctx: Context, config: Config): Promise<void> {
return
}
applyGlobTool(ctx, {
sampleOverCapGlobResults: resolved.sampleOverCapGlobResults,
maxResults: resolved.globMaxResults,
maxMetaBytes: resolved.searchMetaMaxBytes,
rawOutputMaxBytes: resolved.rawOutputMaxBytes,

View File

@@ -34,6 +34,15 @@ import type {
import type { RetainedItems } from '@deepseek-ai/dsh-retention'
import type { GrepMatch } from './search-core.ts'
/**
* The retention fields a meta projection reads: the retained page, whether the
* complete result was capped, and the pre-cap total. Both a full
* {@link RetainedItems} (from `retainGrepMatches`) and `glob`'s sampled page
* satisfy this structural subset, so a projection consumes either without a fake
* `kept`/`omitted`.
*/
type RetainedPage<T> = Pick<RetainedItems<T>, 'items' | 'truncated' | 'seen'>
/**
* The `grep`/`glob` tools' private `tool/result` `meta` payload: the capped,
* structured search result. Attached opaquely (as `JsonValue`) on the tool result
@@ -118,7 +127,7 @@ function capMetaBytes(meta: SearchMeta, maxMetaBytes: number): SearchMeta {
* @param maxMetaBytes - the serialized-meta byte budget.
* @returns the `matches`-shaped search metadata.
*/
export function grepSearchMeta(retained: RetainedItems<GrepMatch>, maxMetaBytes: number): SearchMeta {
export function grepSearchMeta(retained: RetainedPage<GrepMatch>, maxMetaBytes: number): SearchMeta {
const meta: SearchMeta = {
shape: 'matches',
files: groupMatchesByFile(retained.items),
@@ -138,7 +147,7 @@ export function grepSearchMeta(retained: RetainedItems<GrepMatch>, maxMetaBytes:
* @param maxMetaBytes - the serialized-meta byte budget.
* @returns the `paths`-shaped search metadata.
*/
export function globSearchMeta(retained: RetainedItems<string>, maxMetaBytes: number): SearchMeta {
export function globSearchMeta(retained: RetainedPage<string>, maxMetaBytes: number): SearchMeta {
const meta: SearchMeta = {
shape: 'paths',
paths: retained.items,