Merge remote-tracking branch 'origin/master' into feat/search-presenter
# Conflicts: # docs/config-catalog.md # docs/cookbook/adding-a-tool.i18n.yaml # docs/cookbook/adding-a-tool.md # docs/cookbook/adding-a-tool.zh.md # docs/cordis-catalog/events.md # docs/cordis-catalog/services.md # docs/core-data-structures/tools.i18n.yaml # docs/core-data-structures/tools.md # docs/core-data-structures/tools.zh.md # docs/event-producer-consumer.md # examples/acp-agent/tests/snapshots/cordis-inspect-jsdoc/session.jsonl # packages/cordis/tool-cordis/src/api-catalog.ts # packages/core/tools/README.i18n.yaml # packages/core/tools/README.md # packages/core/tools/README.zh.md # packages/core/tools/src/index.ts # packages/core/tools/src/presentation.ts # packages/fs/tool-fs-search/src/glob.ts # packages/fs/tool-fs-search/src/index.ts # packages/ui/tui/src/components/transcript.ts # packages/ui/tui/tests/tui.spec.ts
This commit is contained in:
@@ -3,20 +3,19 @@
|
||||
* pattern, sorted by modification time. Execution goes through the bash seam
|
||||
* (`ctx.bash`) with a fixed `rg --files` command — this module owns the
|
||||
* model-facing schema, argument validation, shell-safe command construction,
|
||||
* result parsing, retention, and formatting; process concerns (defaulting,
|
||||
* result parsing, inline sampling, and formatting; process concerns (defaulting,
|
||||
* scrubbing, kill, backend substitution) stay behind `ctx.bash`.
|
||||
*
|
||||
* @module @deepseek-ai/dsh-tool-fs-search/glob
|
||||
*/
|
||||
|
||||
import type { Context } from 'cordis'
|
||||
import { sep } from 'node:path'
|
||||
import { defineTool } from '@deepseek-ai/dsh-tools'
|
||||
import type { GenericCallView, SearchResultView, ToolResult } from '@deepseek-ai/dsh-tools'
|
||||
import type { RetainedItems } from '@deepseek-ai/dsh-retention'
|
||||
import type { SpillRef } from '@deepseek-ai/dsh-spill'
|
||||
import type {} from '@deepseek-ai/dsh-bash'
|
||||
import type {} from '@deepseek-ai/dsh-system-prompt'
|
||||
import { retainGlobPaths, runRipgrep, toWorkdirRelative, trySaveFormattedResult } from './search-core.ts'
|
||||
import { runRipgrep, toWorkdirRelative, trySaveFormattedResult } from './search-core.ts'
|
||||
import { globSearchMeta, searchViewFromMeta } from './presentation.ts'
|
||||
import { singleQuote } from './shell-quote.ts'
|
||||
import { acceptedSurfaceValue } from './surface.ts'
|
||||
@@ -41,6 +40,8 @@ export const GLOB_VCS_EXCLUDES: readonly string[] = ['.git', '.svn', '.hg', '.bz
|
||||
|
||||
/** Resolved glob-tool caps — plugin config after defaulting (see `Config` in index.ts). */
|
||||
export interface GlobToolCaps {
|
||||
/** Whether over-cap pages are sampled across top-level entries instead of taking the modification-time head. */
|
||||
sampleOverCapGlobResults: boolean
|
||||
/** Max paths retained inline; later paths go to the formatted spill file. */
|
||||
maxResults: number
|
||||
/** Max bytes of serialized `presentationMeta`; trailing paths drop past it. */
|
||||
@@ -101,28 +102,154 @@ export function buildGlobCommand(input: GlobInput): string {
|
||||
}
|
||||
|
||||
/**
|
||||
* Format the model-facing `glob` result: the retained paths, then — when the
|
||||
* result was capped — a footer carrying either the formatted-spill recovery
|
||||
* locator or the could-not-save explanation. The omitted count is a budget fact:
|
||||
* the search itself completed.
|
||||
* The inline page of a capped `glob` result, plus how much of the complete
|
||||
* result's top level it reaches.
|
||||
*/
|
||||
export interface GlobSample {
|
||||
/** Paths to show inline: grouped by top-level entry, modification-time ordered within each group. */
|
||||
items: string[]
|
||||
/** Distinct top-level entries the shown paths reach. */
|
||||
shown: number
|
||||
/** Distinct top-level entries across the complete result. */
|
||||
total: number
|
||||
}
|
||||
|
||||
/** Remove the displayed search-root prefix before choosing a top-level group. */
|
||||
function relativeToSearchRoot(path: string, root: string): string {
|
||||
if (root === '.') return path.startsWith(`.${sep}`) ? path.slice(2) : path
|
||||
let rootEnd = root.length
|
||||
while (rootEnd > 0 && root[rootEnd - 1] === sep) rootEnd -= 1
|
||||
const trimmedRoot = root.slice(0, rootEnd)
|
||||
if (trimmedRoot.length === 0) return stripLeadingSeparators(path)
|
||||
if (path === trimmedRoot) return ''
|
||||
if (path.startsWith(`${trimmedRoot}${sep}`)) {
|
||||
return path.slice(trimmedRoot.length + 1)
|
||||
}
|
||||
return path
|
||||
}
|
||||
|
||||
/** Strip only separators recognized by the execution platform. */
|
||||
function stripLeadingSeparators(path: string): string {
|
||||
let start = 0
|
||||
while (path[start] === sep) start += 1
|
||||
return path.slice(start)
|
||||
}
|
||||
|
||||
/**
|
||||
* The leading path segment of one display path — the top-level entry, relative
|
||||
* to the search root, that the path sits under. A path with no separator is its
|
||||
* own top-level entry. Leading separators are stripped first so an absolute path
|
||||
* (one outside the workdir, which {@link toWorkdirRelative} leaves untouched)
|
||||
* groups by its first real name instead of collapsing every such path into one
|
||||
* empty group.
|
||||
*/
|
||||
function topLevelSegment(path: string): string {
|
||||
const trimmed = stripLeadingSeparators(path)
|
||||
const cut = trimmed.indexOf(sep)
|
||||
return cut === -1 ? trimmed : trimmed.slice(0, cut)
|
||||
}
|
||||
|
||||
/**
|
||||
* Choose the inline page of an over-cap result by round-robin across the
|
||||
* complete result's top-level entries, instead of taking its head.
|
||||
*
|
||||
* @param retained - the retention outcome over every discovered path.
|
||||
* Every top-level entry receives a slot before any receives a second; exhausted
|
||||
* groups drop out. Group order and order within each group follow `paths`, so a
|
||||
* flat result reproduces the modification-time head.
|
||||
*
|
||||
* @param paths - the complete result, in ripgrep's modification-time order.
|
||||
* @param maxItems - how many paths the page may hold; the caller has already established it is smaller than `paths`.
|
||||
* @param root - the search root in the same display-path space as `paths`.
|
||||
* @returns the page grouped by top-level entry, with the shown/total top-level spread.
|
||||
*/
|
||||
export function sampleAcrossTopLevel(paths: readonly string[], maxItems: number, root = '.'): GlobSample {
|
||||
type ActiveGroup = { key: string; items: string[]; index: number; current: string }
|
||||
const groups = new Map<string, string[]>()
|
||||
let active: ActiveGroup[] = []
|
||||
for (const path of paths) {
|
||||
const key = topLevelSegment(relativeToSearchRoot(path, root))
|
||||
const group = groups.get(key)
|
||||
if (group === undefined) {
|
||||
const items = [path]
|
||||
groups.set(key, items)
|
||||
active.push({ key, items, index: 0, current: path })
|
||||
} else {
|
||||
group.push(path)
|
||||
}
|
||||
}
|
||||
const taken = new Map<string, string[]>()
|
||||
let count = 0
|
||||
while (active.length > 0 && count < maxItems) {
|
||||
const nextActive: ActiveGroup[] = []
|
||||
for (const { key, items, index, current } of active) {
|
||||
if (count >= maxItems) break
|
||||
count += 1
|
||||
const bucket = taken.get(key)
|
||||
if (bucket === undefined) taken.set(key, [current])
|
||||
else bucket.push(current)
|
||||
const nextIndex = index + 1
|
||||
const nextPath = items[nextIndex]
|
||||
if (nextPath !== undefined) nextActive.push({ key, items, index: nextIndex, current: nextPath })
|
||||
}
|
||||
active = nextActive
|
||||
}
|
||||
return { items: [...taken.values()].flat(), shown: taken.size, total: groups.size }
|
||||
}
|
||||
|
||||
/**
|
||||
* Format a capped sampled page and its complete-result recovery path. A flat
|
||||
* result keeps the plain footer because its sample is the modification-time head.
|
||||
*
|
||||
* @param sample - the inline page and its top-level spread.
|
||||
* @param seen - how many paths the complete result holds; always more than the page.
|
||||
* @param spillRef - the saved complete-result reference, or `undefined` when unsaved.
|
||||
* @returns the model-facing text.
|
||||
*/
|
||||
export function formatGlobOutput(retained: RetainedItems<string>, spillRef: SpillRef | undefined): string {
|
||||
const body = retained.items.join('\n')
|
||||
if (!retained.truncated) return body
|
||||
export function formatGlobOutput(sample: GlobSample, seen: number, spillRef: SpillRef | undefined): string {
|
||||
const basis = sample.total === seen
|
||||
? '.'
|
||||
: `, sampled across ${sample.shown} of the ${sample.total} top-level entries this pattern matched instead of taken in modification-time order.`
|
||||
+ (sample.shown < sample.total ? ' Narrow path to inspect a specific subtree.' : '')
|
||||
return formatGlobPage(sample.items, seen, spillRef, basis)
|
||||
}
|
||||
|
||||
/** Format one bounded page and the recovery path for its complete sorted result. */
|
||||
function formatGlobPage(items: readonly string[], seen: number, spillRef: SpillRef | undefined, basis: string): string {
|
||||
const body = items.join('\n')
|
||||
const recovery = spillRef !== undefined
|
||||
? `Full sorted result stored at: ${spillRef.locator}. ${spillRef.retrievalHint}`
|
||||
: 'The complete result could not be saved; narrow pattern or path to see more.'
|
||||
return `${body}\n\n(Showing ${retained.kept} of ${retained.seen} paths. ${recovery})`
|
||||
return `${body}\n\n(Showing ${items.length} of ${seen} paths${basis} ${recovery})`
|
||||
}
|
||||
|
||||
/** Format one already-retained path list for the Native surface. */
|
||||
function formatRetainedGlob(retained: RetainedItems<string>, spillRef?: SpillRef): string {
|
||||
if (retained.seen === 0) return 'No files found'
|
||||
return formatGlobOutput(retained, spillRef)
|
||||
/** Bound and format one canonical path list for the Native surface relative to its search root. */
|
||||
function renderGlobPaths(paths: string[], caps: GlobToolCaps, root: string, spillRef?: SpillRef): string {
|
||||
if (paths.length === 0) return 'No files found'
|
||||
// A result that fits is shown whole, untouched: modification-time order is the
|
||||
// tool's contract, and over a complete result it is what answers age questions.
|
||||
if (paths.length <= caps.maxResults) return paths.join('\n')
|
||||
if (!caps.sampleOverCapGlobResults) {
|
||||
return formatGlobPage(paths.slice(0, caps.maxResults), paths.length, spillRef, '.')
|
||||
}
|
||||
return formatGlobOutput(sampleAcrossTopLevel(paths, caps.maxResults, root), paths.length, spillRef)
|
||||
}
|
||||
|
||||
/**
|
||||
* The inline page of paths a completed `glob` card shows, computed the SAME way
|
||||
* {@link renderGlobPaths} computes its model-facing page so the card and the text
|
||||
* agree on which paths survived the cap. A result within the cap is shown whole;
|
||||
* an over-cap result is either the modification-time head or the top-level sample,
|
||||
* matching the deployment's `sampleOverCapGlobResults`.
|
||||
*
|
||||
* @param paths - the complete discovered path list, in modification-time order.
|
||||
* @param caps - the resolved glob caps (the inline cap and the sampling switch).
|
||||
* @param root - the search root in the same display-path space as `paths`.
|
||||
* @returns the inline page and whether the complete result was capped.
|
||||
*/
|
||||
function globCardPage(paths: string[], caps: GlobToolCaps, root: string): { items: string[]; truncated: boolean } {
|
||||
if (paths.length <= caps.maxResults) return { items: paths, truncated: false }
|
||||
if (!caps.sampleOverCapGlobResults) return { items: paths.slice(0, caps.maxResults), truncated: true }
|
||||
return { items: sampleAcrossTopLevel(paths, caps.maxResults, root).items, truncated: true }
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -162,19 +289,32 @@ export function presentGlobResult(_args: { pattern: string; path?: string }, res
|
||||
* @param caps - the deployment's resolved glob caps (plugin config after defaulting).
|
||||
*/
|
||||
export function applyGlobTool(ctx: Context, caps: GlobToolCaps): void {
|
||||
const overCapGuidance = caps.sampleOverCapGlobResults
|
||||
? 'while a larger one is sampled across top-level entries, so it spans the tree instead of one subtree.'
|
||||
: 'while a larger one keeps the modification-time-ordered head.'
|
||||
ctx.systemPrompt.section({
|
||||
name: 'tool:glob',
|
||||
order: 103,
|
||||
text: 'Use the glob tool — not shell find or ls — to discover files by path pattern. Results are sorted by modification time and include hidden and ignored files.',
|
||||
text: 'Use the glob tool — not shell find — to discover files by path pattern. A pattern with no "/" matches basenames at any depth, so "*" matches every file in the tree rather than its top level. '
|
||||
+ `Results are files only, never directories, and include hidden and ignored files: a result that fits comes back in modification-time order, ${overCapGuidance}`,
|
||||
})
|
||||
|
||||
const overCapDescription = caps.sampleOverCapGlobResults
|
||||
? `a larger result instead returns ${caps.maxResults} paths sampled across top-level entries`
|
||||
: `a larger result returns the first ${caps.maxResults} paths in modification-time order`
|
||||
const tool = defineTool({
|
||||
name: 'glob',
|
||||
description: 'Find files whose paths match a glob pattern. Returns matching paths sorted by modification time, '
|
||||
description: 'Find files whose paths match a glob pattern. Returns matching file paths — never directories — '
|
||||
+ 'including hidden and ignored files (VCS metadata directories are excluded). '
|
||||
+ `Returns the first ${caps.maxResults} paths inline; a capped result reports where the complete list was saved.`,
|
||||
+ `Up to ${caps.maxResults} paths come back in modification-time order; ${overCapDescription}, `
|
||||
+ 'says so, and reports where the complete sorted list was saved. This tool does not enumerate directory entries.',
|
||||
parameters: {
|
||||
pattern: { type: 'string', required: true, description: 'Glob pattern to match file paths against (e.g. "**/*.ts", "src/**/*.test.js").' },
|
||||
pattern: {
|
||||
type: 'string',
|
||||
required: true,
|
||||
description: 'Glob pattern to match file paths against (e.g. "**/*.ts", "src/**/*.test.js"). '
|
||||
+ 'A pattern with no "/" matches the basename at any depth, so "*" and "*.ts" both search the whole tree; include a separator to anchor the depth.',
|
||||
},
|
||||
path: { type: 'string', description: 'Directory to search in. Defaults to the session workspace; a relative path resolves against it.' },
|
||||
},
|
||||
timeoutMs: caps.timeoutMs,
|
||||
@@ -183,16 +323,21 @@ export function applyGlobTool(ctx: Context, caps: GlobToolCaps): void {
|
||||
type: 'object',
|
||||
additionalProperties: false,
|
||||
properties: {
|
||||
root: { type: 'string', required: true },
|
||||
paths: { type: 'array', required: true, items: { type: 'string' } },
|
||||
},
|
||||
},
|
||||
render: (_args, value) => [{ type: 'text', text: formatRetainedGlob(retainGlobPaths(value.paths, caps.maxResults)) }],
|
||||
presentationMeta: (_args, value) => globSearchMeta(retainGlobPaths(value.paths, caps.maxResults), caps.maxMetaBytes),
|
||||
render: (_args, value) => [{ type: 'text', text: renderGlobPaths(value.paths, caps, value.root) }],
|
||||
presentationMeta: (_args, value) => {
|
||||
const page = globCardPage(value.paths, caps, value.root)
|
||||
return globSearchMeta({ items: page.items, truncated: page.truncated, seen: value.paths.length }, caps.maxMetaBytes)
|
||||
},
|
||||
},
|
||||
async execute(args, exec) {
|
||||
const input = parseGlobArgs(args)
|
||||
const run = await runRipgrep(ctx, exec, 'glob', buildGlobCommand(input), caps.rawOutputMaxBytes)
|
||||
if (run.noMatches) return { paths: [] }
|
||||
const root = input.path === undefined ? '.' : toWorkdirRelative(input.path, run.workdir)
|
||||
if (run.noMatches) return { root, paths: [] }
|
||||
|
||||
const all: string[] = []
|
||||
for (const line of run.stdout.split('\n')) {
|
||||
@@ -200,7 +345,7 @@ export function applyGlobTool(ctx: Context, caps: GlobToolCaps): void {
|
||||
const displayPath = toWorkdirRelative(line, run.workdir)
|
||||
all.push(displayPath)
|
||||
}
|
||||
return { paths: all }
|
||||
return { root, paths: all }
|
||||
},
|
||||
presentCall: presentGlobCall,
|
||||
presentResult: presentGlobResult,
|
||||
@@ -209,14 +354,14 @@ export function applyGlobTool(ctx: Context, caps: GlobToolCaps): void {
|
||||
|
||||
ctx.on('tools/post-execute', async (exec, result, next) => {
|
||||
const decision = await next()
|
||||
const value = acceptedSurfaceValue(ctx, tool, exec, result, decision) as { paths: string[] } | undefined
|
||||
const value = acceptedSurfaceValue(ctx, tool, exec, result, decision) as { root: string; paths: string[] } | undefined
|
||||
if (value === undefined) return decision
|
||||
const paths = value.paths
|
||||
if (paths.length <= caps.maxResults) return decision
|
||||
const spillRef = await trySaveFormattedResult(ctx, exec, 'glob-results.txt', paths.join('\n'))
|
||||
return {
|
||||
kind: 'accept',
|
||||
content: [{ type: 'text', text: formatRetainedGlob(retainGlobPaths(paths, caps.maxResults), spillRef) }],
|
||||
content: [{ type: 'text', text: renderGlobPaths(paths, caps, value.root, spillRef) }],
|
||||
...decision.additionalContexts !== undefined ? { additionalContexts: decision.additionalContexts } : {},
|
||||
}
|
||||
})
|
||||
|
||||
@@ -33,8 +33,8 @@ import { GLOB_MAX_RESULTS, applyGlobTool } from './glob.ts'
|
||||
import { GREP_MAX_LINE_BYTES, GREP_MAX_MATCHES, applyGrepTool } from './grep.ts'
|
||||
import { RAW_OUTPUT_MAX_BYTES, SEARCH_META_MAX_BYTES, SEARCH_TIMEOUT_MS } from './search-core.ts'
|
||||
|
||||
export { GLOB_MAX_RESULTS, GLOB_VCS_EXCLUDES, applyGlobTool, buildGlobCommand, formatGlobOutput, parseGlobArgs, presentGlobCall, presentGlobResult } from './glob.ts'
|
||||
export type { GlobInput, GlobToolCaps } from './glob.ts'
|
||||
export { GLOB_MAX_RESULTS, GLOB_VCS_EXCLUDES, applyGlobTool, buildGlobCommand, formatGlobOutput, parseGlobArgs, presentGlobCall, presentGlobResult, sampleAcrossTopLevel } from './glob.ts'
|
||||
export type { GlobInput, GlobSample, GlobToolCaps } from './glob.ts'
|
||||
export {
|
||||
GREP_MAX_LINE_BYTES,
|
||||
GREP_MAX_MATCHES,
|
||||
@@ -67,8 +67,10 @@ export const name = 'tool-fs-search'
|
||||
/** Services required by the search tool suite (`spillStore` is optional, read via `ctx.get()`). */
|
||||
export const inject = ['tools', 'systemPrompt', 'bash']
|
||||
|
||||
/** Plugin config (all optional — `Config` supplies the defaults). */
|
||||
/** Plugin config; over-cap glob sampling is an explicit deployment choice and the remaining fields have defaults. */
|
||||
export interface Config {
|
||||
/** Whether an over-cap `glob` page is sampled across top-level entries instead of taking the modification-time head. */
|
||||
sampleOverCapGlobResults: boolean
|
||||
/** Max paths one `glob` call retains inline; later paths go to the formatted spill file. */
|
||||
globMaxResults?: number
|
||||
/** Max flat matches one `grep` call retains inline; later matches go to the formatted spill file. */
|
||||
@@ -84,6 +86,7 @@ export interface Config {
|
||||
}
|
||||
|
||||
export const Config: z<Config> = z.object({
|
||||
sampleOverCapGlobResults: z.boolean().required(),
|
||||
globMaxResults: z.number().default(GLOB_MAX_RESULTS),
|
||||
grepMaxMatches: z.number().default(GREP_MAX_MATCHES),
|
||||
grepMaxLineBytes: z.number().default(GREP_MAX_LINE_BYTES),
|
||||
@@ -150,6 +153,7 @@ export async function apply(ctx: Context, config: Config): Promise<void> {
|
||||
return
|
||||
}
|
||||
applyGlobTool(ctx, {
|
||||
sampleOverCapGlobResults: resolved.sampleOverCapGlobResults,
|
||||
maxResults: resolved.globMaxResults,
|
||||
maxMetaBytes: resolved.searchMetaMaxBytes,
|
||||
rawOutputMaxBytes: resolved.rawOutputMaxBytes,
|
||||
|
||||
@@ -34,6 +34,15 @@ import type {
|
||||
import type { RetainedItems } from '@deepseek-ai/dsh-retention'
|
||||
import type { GrepMatch } from './search-core.ts'
|
||||
|
||||
/**
|
||||
* The retention fields a meta projection reads: the retained page, whether the
|
||||
* complete result was capped, and the pre-cap total. Both a full
|
||||
* {@link RetainedItems} (from `retainGrepMatches`) and `glob`'s sampled page
|
||||
* satisfy this structural subset, so a projection consumes either without a fake
|
||||
* `kept`/`omitted`.
|
||||
*/
|
||||
type RetainedPage<T> = Pick<RetainedItems<T>, 'items' | 'truncated' | 'seen'>
|
||||
|
||||
/**
|
||||
* The `grep`/`glob` tools' private `tool/result` `meta` payload: the capped,
|
||||
* structured search result. Attached opaquely (as `JsonValue`) on the tool result
|
||||
@@ -118,7 +127,7 @@ function capMetaBytes(meta: SearchMeta, maxMetaBytes: number): SearchMeta {
|
||||
* @param maxMetaBytes - the serialized-meta byte budget.
|
||||
* @returns the `matches`-shaped search metadata.
|
||||
*/
|
||||
export function grepSearchMeta(retained: RetainedItems<GrepMatch>, maxMetaBytes: number): SearchMeta {
|
||||
export function grepSearchMeta(retained: RetainedPage<GrepMatch>, maxMetaBytes: number): SearchMeta {
|
||||
const meta: SearchMeta = {
|
||||
shape: 'matches',
|
||||
files: groupMatchesByFile(retained.items),
|
||||
@@ -138,7 +147,7 @@ export function grepSearchMeta(retained: RetainedItems<GrepMatch>, maxMetaBytes:
|
||||
* @param maxMetaBytes - the serialized-meta byte budget.
|
||||
* @returns the `paths`-shaped search metadata.
|
||||
*/
|
||||
export function globSearchMeta(retained: RetainedItems<string>, maxMetaBytes: number): SearchMeta {
|
||||
export function globSearchMeta(retained: RetainedPage<string>, maxMetaBytes: number): SearchMeta {
|
||||
const meta: SearchMeta = {
|
||||
shape: 'paths',
|
||||
paths: retained.items,
|
||||
|
||||
Reference in New Issue
Block a user