fix: address codex review round 1
- Re-validate redirect targets through validateFetchUrl before following, so a same-origin Location carrying credentials (or a non-http(s)/over-long URL) cannot bypass the transport hygiene a direct request enforces. - Treat only DROPPED bytes as truncation: a body exactly at maxResponseBytes is no longer falsely flagged truncated (which emitted a spurious footer). - Honor the declared response charset: parse the Content-Type charset and decode with it (rejecting unsupported labels as WEB_UNSUPPORTED_CONTENT_TYPE) instead of always assuming UTF-8 and returning replacement characters. - Catalog the web seam vocabulary in docs/core-data-structures/web.md with type-equiv blocks + manifest entries, per the core-data-structures rule.
This commit is contained in:
@@ -18,7 +18,7 @@ export {
|
||||
LocalFetchProvider,
|
||||
} from './provider.ts'
|
||||
export type { LocalFetchLimits } from './provider.ts'
|
||||
export { classifyContentType, isSameOrigin, validateFetchUrl } from './policy.ts'
|
||||
export { classifyContentType, decoderForCharset, isSameOrigin, parseCharset, validateFetchUrl } from './policy.ts'
|
||||
export type { FetchableKind } from './policy.ts'
|
||||
|
||||
/** Default `User-Agent`: an explicit product agent, never a browser disguise. */
|
||||
|
||||
@@ -57,3 +57,29 @@ export function classifyContentType(contentType: string | null): FetchableKind |
|
||||
if (mime === 'application/json' || mime === 'application/xml' || mime.endsWith('+json') || mime.endsWith('+xml')) return 'text'
|
||||
return undefined
|
||||
}
|
||||
|
||||
/**
|
||||
* Extract the `charset` parameter from a response `Content-Type`, lower-cased,
|
||||
* or `undefined` when absent. The provider feeds this label to `TextDecoder`
|
||||
* so a non-UTF-8 response is decoded with its declared encoding rather than
|
||||
* silently mangled into replacement characters.
|
||||
*/
|
||||
export function parseCharset(contentType: string | null): string | undefined {
|
||||
const match = /;\s*charset\s*=\s*"?([^";]+)"?/i.exec(contentType ?? '')
|
||||
return match?.[1]?.trim().toLowerCase()
|
||||
}
|
||||
|
||||
/**
|
||||
* Build a `TextDecoder` for the declared charset, falling back to UTF-8 when
|
||||
* none is declared. Throws {@link WebError} `WEB_UNSUPPORTED_CONTENT_TYPE` when
|
||||
* the label is present but not a charset `TextDecoder` recognizes — better to
|
||||
* fail loudly than return mojibake.
|
||||
*/
|
||||
export function decoderForCharset(charset: string | undefined): TextDecoder {
|
||||
if (charset === undefined) return new TextDecoder('utf-8')
|
||||
try {
|
||||
return new TextDecoder(charset)
|
||||
} catch (error: unknown) {
|
||||
throw new WebError(`unsupported charset "${charset}"`, 'WEB_UNSUPPORTED_CONTENT_TYPE', { cause: error })
|
||||
}
|
||||
}
|
||||
|
||||
@@ -21,7 +21,7 @@
|
||||
|
||||
import { WebError } from '@deepseek-ai/dsh-web'
|
||||
import type { WebFetchBody, WebFetchProvider, WebFetchRequest, WebFetchResult, WebProviderStatus } from '@deepseek-ai/dsh-web'
|
||||
import { classifyContentType, isSameOrigin, validateFetchUrl } from './policy.ts'
|
||||
import { classifyContentType, decoderForCharset, isSameOrigin, parseCharset, validateFetchUrl } from './policy.ts'
|
||||
|
||||
/** Resolved provider limits (the plugin's schemastery Config supplies defaults). */
|
||||
export interface LocalFetchLimits {
|
||||
@@ -92,14 +92,18 @@ export class LocalFetchProvider implements WebFetchProvider {
|
||||
throw new WebError(`redirect response (HTTP ${response.status}) without a Location header`, 'WEB_PROVIDER_ERROR')
|
||||
}
|
||||
const target = resolveRedirect(location, currentUrl)
|
||||
if (!isSameOrigin(target, currentUrl)) {
|
||||
// Re-validate the target against the same transport hygiene a direct
|
||||
// request gets: a redirect must not be a back door to a credentialed,
|
||||
// non-http(s), or over-long URL that validateFetchUrl would reject.
|
||||
const validatedTarget = validateFetchUrl(target.toString(), this.limits.maxUrlLength)
|
||||
if (!isSameOrigin(validatedTarget, currentUrl)) {
|
||||
throw new WebError(
|
||||
`cross-origin redirect to ${target.origin} is not followed automatically; retry against that URL directly`,
|
||||
`cross-origin redirect to ${validatedTarget.origin} is not followed automatically; retry against that URL directly`,
|
||||
'WEB_REDIRECT_BLOCKED',
|
||||
)
|
||||
}
|
||||
await response.body?.cancel()
|
||||
currentUrl = target
|
||||
currentUrl = validatedTarget
|
||||
continue
|
||||
}
|
||||
|
||||
@@ -124,14 +128,18 @@ export class LocalFetchProvider implements WebFetchProvider {
|
||||
|
||||
/** Read, byte-cap, classify, and decode the final response body. */
|
||||
private async readBody(response: Response, finalUrl: URL): Promise<WebFetchResult> {
|
||||
const kind = classifyContentType(response.headers.get('content-type'))
|
||||
const contentType = response.headers.get('content-type')
|
||||
const kind = classifyContentType(contentType)
|
||||
if (kind === undefined) {
|
||||
await response.body?.cancel()
|
||||
throw new WebError(`unsupported content type "${response.headers.get('content-type') ?? 'unknown'}"`, 'WEB_UNSUPPORTED_CONTENT_TYPE')
|
||||
throw new WebError(`unsupported content type "${contentType ?? 'unknown'}"`, 'WEB_UNSUPPORTED_CONTENT_TYPE')
|
||||
}
|
||||
|
||||
// Resolve the decoder BEFORE reading the body so an unsupported charset
|
||||
// fails without consuming the stream.
|
||||
const decoder = decoderForCharset(parseCharset(contentType))
|
||||
const { bytes, truncatedByBytes } = await this.readCapped(response)
|
||||
const decoded = new TextDecoder('utf-8').decode(bytes)
|
||||
const decoded = decoder.decode(bytes)
|
||||
const truncatedByChars = decoded.length > this.limits.maxBodyChars
|
||||
const content = truncatedByChars ? decoded.slice(0, this.limits.maxBodyChars) : decoded
|
||||
const body: WebFetchBody = kind === 'html' ? { kind: 'html', content } : { kind: 'text', content }
|
||||
@@ -173,7 +181,10 @@ export class LocalFetchProvider implements WebFetchProvider {
|
||||
const { done, value } = await reader.read()
|
||||
if (done) break
|
||||
const remaining = this.limits.maxResponseBytes - total
|
||||
if (value.byteLength >= remaining) {
|
||||
// Only DROPPED bytes count as truncation: a chunk that exactly fills the
|
||||
// remaining capacity keeps all its bytes and we read on to observe EOF,
|
||||
// so an exactly-at-cap body is not falsely flagged truncated.
|
||||
if (value.byteLength > remaining) {
|
||||
chunks.push(value.subarray(0, remaining))
|
||||
total += remaining
|
||||
truncatedByBytes = true
|
||||
|
||||
Reference in New Issue
Block a user