/** * URL validation and content-type classification for the local HTTP(S) fetch * provider — the pure, network-free half. The provider's `fetch()` composes * these with transport (redirect following, byte caps, decoding). * * @module @deepseek-ai/dsh-web-fetch-local/policy */ import { WebError } from '@deepseek-ai/dsh-web' /** The body kinds this provider decodes. */ export type FetchableKind = 'html' | 'text' /** * Validate a request URL against the basic transport hygiene the provider * enforces before any network access: http(s) only, no embedded credentials, * bounded length. Returns the parsed `URL`. Throws {@link WebError} otherwise. * (SSRF / private-network blocking is deferred — see the package RFC.) */ export function validateFetchUrl(input: string, maxUrlLength: number): URL { if (input.length > maxUrlLength) { throw new WebError(`URL exceeds the maximum length of ${maxUrlLength}`, 'WEB_INVALID_URL') } let url: URL try { url = new URL(input) } catch (error: unknown) { throw new WebError(`invalid URL: ${input}`, 'WEB_INVALID_URL', { cause: error }) } if (url.protocol !== 'http:' && url.protocol !== 'https:') { throw new WebError(`unsupported URL scheme "${url.protocol}" (only http and https are allowed)`, 'WEB_INVALID_URL') } if (url.username.length > 0 || url.password.length > 0) { throw new WebError('credentials in URLs are not allowed', 'WEB_BLOCKED_URL') } return url } /** * Two URLs are same-origin when scheme, hostname, and port match. A redirect * that crosses origins is refused so each new origin requires a fresh tool call * (and thus a fresh provider/permission decision). */ export function isSameOrigin(a: URL, b: URL): boolean { return a.protocol === b.protocol && a.hostname === b.hostname && a.port === b.port } /** * Classify a response `Content-Type` into a decodable body kind, or `undefined` * for an unsupported (e.g. binary) type. `text/html` and `application/xhtml+xml` * are `html`; other `text/*` plus a few structured text types are `text`. */ export function classifyContentType(contentType: string | null): FetchableKind | undefined { const mime = (contentType ?? '').replace(/;.*$/s, '').trim().toLowerCase() if (mime === 'text/html' || mime === 'application/xhtml+xml') return 'html' if (mime.startsWith('text/')) return 'text' if (mime === 'application/json' || mime === 'application/xml' || mime.endsWith('+json') || mime.endsWith('+xml')) return 'text' return undefined } /** * Extract the `charset` parameter from a response `Content-Type`, lower-cased, * or `undefined` when absent. The provider feeds this label to `TextDecoder` * so a non-UTF-8 response is decoded with its declared encoding rather than * silently mangled into replacement characters. */ export function parseCharset(contentType: string | null): string | undefined { const match = /;\s*charset\s*=\s*"?([^";]+)"?/i.exec(contentType ?? '') return match?.[1]?.trim().toLowerCase() } /** * Build a `TextDecoder` for the declared charset, falling back to UTF-8 when * none is declared. Throws {@link WebError} `WEB_UNSUPPORTED_CONTENT_TYPE` when * the label is present but not a charset `TextDecoder` recognizes — better to * fail loudly than return mojibake. */ export function decoderForCharset(charset: string | undefined): TextDecoder { if (charset === undefined) return new TextDecoder('utf-8') try { return new TextDecoder(charset) } catch (error: unknown) { throw new WebError(`unsupported charset "${charset}"`, 'WEB_UNSUPPORTED_CONTENT_TYPE', { cause: error }) } }