mirror of
https://github.com/deepseek-ai/deepseek-harness
synced 2026-08-15 21:04:50 +00:00
- Re-validate redirect targets through validateFetchUrl before following, so a same-origin Location carrying credentials (or a non-http(s)/over-long URL) cannot bypass the transport hygiene a direct request enforces. - Treat only DROPPED bytes as truncation: a body exactly at maxResponseBytes is no longer falsely flagged truncated (which emitted a spurious footer). - Honor the declared response charset: parse the Content-Type charset and decode with it (rejecting unsupported labels as WEB_UNSUPPORTED_CONTENT_TYPE) instead of always assuming UTF-8 and returning replacement characters. - Catalog the web seam vocabulary in docs/core-data-structures/web.md with type-equiv blocks + manifest entries, per the core-data-structures rule.
86 lines
3.5 KiB
TypeScript
86 lines
3.5 KiB
TypeScript
/**
|
|
* URL validation and content-type classification for the local HTTP(S) fetch
|
|
* provider — the pure, network-free half. The provider's `fetch()` composes
|
|
* these with transport (redirect following, byte caps, decoding).
|
|
*
|
|
* @module @deepseek-ai/dsh-web-fetch-local/policy
|
|
*/
|
|
|
|
import { WebError } from '@deepseek-ai/dsh-web'
|
|
|
|
/** The body kinds this provider decodes. */
|
|
export type FetchableKind = 'html' | 'text'
|
|
|
|
/**
|
|
* Validate a request URL against the basic transport hygiene the provider
|
|
* enforces before any network access: http(s) only, no embedded credentials,
|
|
* bounded length. Returns the parsed `URL`. Throws {@link WebError} otherwise.
|
|
* (SSRF / private-network blocking is deferred — see the package RFC.)
|
|
*/
|
|
export function validateFetchUrl(input: string, maxUrlLength: number): URL {
|
|
if (input.length > maxUrlLength) {
|
|
throw new WebError(`URL exceeds the maximum length of ${maxUrlLength}`, 'WEB_INVALID_URL')
|
|
}
|
|
let url: URL
|
|
try {
|
|
url = new URL(input)
|
|
} catch (error: unknown) {
|
|
throw new WebError(`invalid URL: ${input}`, 'WEB_INVALID_URL', { cause: error })
|
|
}
|
|
if (url.protocol !== 'http:' && url.protocol !== 'https:') {
|
|
throw new WebError(`unsupported URL scheme "${url.protocol}" (only http and https are allowed)`, 'WEB_INVALID_URL')
|
|
}
|
|
if (url.username.length > 0 || url.password.length > 0) {
|
|
throw new WebError('credentials in URLs are not allowed', 'WEB_BLOCKED_URL')
|
|
}
|
|
return url
|
|
}
|
|
|
|
/**
|
|
* Two URLs are same-origin when scheme, hostname, and port match. A redirect
|
|
* that crosses origins is refused so each new origin requires a fresh tool call
|
|
* (and thus a fresh provider/permission decision).
|
|
*/
|
|
export function isSameOrigin(a: URL, b: URL): boolean {
|
|
return a.protocol === b.protocol && a.hostname === b.hostname && a.port === b.port
|
|
}
|
|
|
|
/**
|
|
* Classify a response `Content-Type` into a decodable body kind, or `undefined`
|
|
* for an unsupported (e.g. binary) type. `text/html` and `application/xhtml+xml`
|
|
* are `html`; other `text/*` plus a few structured text types are `text`.
|
|
*/
|
|
export function classifyContentType(contentType: string | null): FetchableKind | undefined {
|
|
const mime = (contentType ?? '').replace(/;.*$/s, '').trim().toLowerCase()
|
|
if (mime === 'text/html' || mime === 'application/xhtml+xml') return 'html'
|
|
if (mime.startsWith('text/')) return 'text'
|
|
if (mime === 'application/json' || mime === 'application/xml' || mime.endsWith('+json') || mime.endsWith('+xml')) return 'text'
|
|
return undefined
|
|
}
|
|
|
|
/**
|
|
* Extract the `charset` parameter from a response `Content-Type`, lower-cased,
|
|
* or `undefined` when absent. The provider feeds this label to `TextDecoder`
|
|
* so a non-UTF-8 response is decoded with its declared encoding rather than
|
|
* silently mangled into replacement characters.
|
|
*/
|
|
export function parseCharset(contentType: string | null): string | undefined {
|
|
const match = /;\s*charset\s*=\s*"?([^";]+)"?/i.exec(contentType ?? '')
|
|
return match?.[1]?.trim().toLowerCase()
|
|
}
|
|
|
|
/**
|
|
* Build a `TextDecoder` for the declared charset, falling back to UTF-8 when
|
|
* none is declared. Throws {@link WebError} `WEB_UNSUPPORTED_CONTENT_TYPE` when
|
|
* the label is present but not a charset `TextDecoder` recognizes — better to
|
|
* fail loudly than return mojibake.
|
|
*/
|
|
export function decoderForCharset(charset: string | undefined): TextDecoder {
|
|
if (charset === undefined) return new TextDecoder('utf-8')
|
|
try {
|
|
return new TextDecoder(charset)
|
|
} catch (error: unknown) {
|
|
throw new WebError(`unsupported charset "${charset}"`, 'WEB_UNSUPPORTED_CONTENT_TYPE', { cause: error })
|
|
}
|
|
}
|