From d08a150a278bdbe9e5133f5f3b3e11fd008d50f0 Mon Sep 17 00:00:00 2001 From: chruffins <23645059+chruffins@users.noreply.github.com> Date: Wed, 30 Sep 2026 16:09:08 +0000 Subject: [PATCH] Generate search content response in Node SDK --- api.md | 2 +- src/resources/search/contents.ts | 148 ++++++++++++++++++------------- src/resources/search/search.ts | 116 +++++++++++++----------- 3 files changed, 152 insertions(+), 114 deletions(-) diff --git a/api.md b/api.md index e05d4237..0d32721d 100644 --- a/api.md +++ b/api.md @@ -715,7 +715,7 @@ Types: Methods: -- client.search.contents.fetch(id, { ...params }) -> void +- client.search.contents.fetch(id, { ...params }) -> Response ## Providers diff --git a/src/resources/search/contents.ts b/src/resources/search/contents.ts index eda9624a..18b57611 100644 --- a/src/resources/search/contents.ts +++ b/src/resources/search/contents.ts @@ -3,7 +3,6 @@ import { APIResource } from '../../core/resource'; import * as SearchAPI from './search'; import { APIPromise } from '../../core/api-promise'; -import { buildHeaders } from '../../internal/headers'; import { RequestOptions } from '../../internal/request-options'; import { path } from '../../internal/utils/path'; @@ -12,21 +11,25 @@ import { path } from '../../internal/utils/path'; */ export class Contents extends APIResource { /** - * Deferred result-content retrieval is reserved but not available in this release. - * Requests return 404 until the retrieval implementation is shipped. X-Request-Id - * identifies this request separately from the search resource. + * Retrieves selected results from a retained search. Provide exactly one of + * result_ids or limit; the latter fetches the top results. Content defaults to + * source:auto. Responses preserve result_ids order. Unknown result IDs are + * rejected before retrieval starts. Missing, expired, or inaccessible searches + * return 404. Once retrieval begins, return one outcome per selected result, + * including timeout entries for work unfinished at the overall deadline. Browser + * retrievals run sequentially in result order, so later results may time out when + * earlier pages are slow. X-Request-Id identifies this request separately from the + * search resource. * * @example * ```ts - * await client.search.contents.fetch('srch_abc123'); + * const response = await client.search.contents.fetch( + * 'srch_abc123', + * ); * ``` */ - fetch(id: string, body: ContentFetchParams, options?: RequestOptions): APIPromise { - return this._client.post(path`/search/${id}/contents`, { - body, - ...options, - headers: buildHeaders([{ Accept: '*/*' }, options?.headers]), - }); + fetch(id: string, body: ContentFetchParams, options?: RequestOptions): APIPromise { + return this._client.post(path`/search/${id}/contents`, { body, ...options }); } } @@ -60,36 +63,37 @@ export namespace FetchRequest { */ export interface Content { /** - * Invalid with source=provider. Supplying browser_id requires source=browser so - * the chosen identity is not bypassed. + * Requires source=auto or source=browser in deferred retrieval. */ browser?: Content.Browser; format?: 'markdown' | 'text'; /** - * Maximum acceptable age of cached page content, measured from origin retrieval. 0 - * forces a live fetch. Governs the Kernel content cache, which is scoped to the - * caller organization and project and separated by retrieval context; fetches - * through a caller-supplied browser_id bypass that cache. Mapped to the provider - * freshness control when source is provider and the provider supports one; - * otherwise provider content age is reported as unknown via fetched_at. + * For source=auto, maximum acceptable age of retained provider content, measured + * from when the search received it from the provider. A value of 0 disables reuse + * of retained content, so every result is fetched through a browser. + * source=provider reuses retained provider content without freshness validation. + * source=browser always fetches through a browser and does not use this age limit. */ max_age_hours?: number; /** - * Per-result Unicode character limit after extraction. + * Per-result Unicode character limit after extraction. Retained provider content + * cannot exceed what was stored at search time; such results report truncated when + * the stored text was already truncated. */ max_chars?: number; /** - * provider uses the search provider's native content retrieval; browser fetches - * each URL through a Kernel browser; auto prefers Kernel browser retrieval and - * falls back to provider-native content when browser retrieval is unavailable or - * unsuitable. Defaults to auto for both inline and deferred retrieval. Deferred - * provider retrieval requires post_hoc capability; an explicit provider source - * without it is a 400. Missing documents produce per-result unavailable outcomes, - * not request failures. + * auto uses retained provider content within max_age_hours; for deferred retrieval + * it falls back to a Kernel browser (caller-supplied or temporary) for results + * without it. Inline retrieval never uses a browser. provider reuses retained + * provider content when available, without freshness validation, and never + * provisions a browser. browser fetches each URL through a Kernel browser, either + * caller-supplied or temporary. No option makes a new provider request. Defaults + * to auto for both inline and deferred retrieval. Missing documents produce + * per-result unavailable outcomes, not request failures. */ source?: 'auto' | 'provider' | 'browser'; @@ -102,17 +106,22 @@ export namespace FetchRequest { export namespace Content { /** - * Invalid with source=provider. Supplying browser_id requires source=browser so - * the chosen identity is not bypassed. + * Requires source=auto or source=browser in deferred retrieval. */ export interface Browser { /** * Existing browser session ID authorized for the caller and selected project. - * Reuses its cookies, proxy, and browser configuration. Kernel does not delete a - * caller-supplied browser. Render mode uses a temporary tab; website activity may - * still change shared cookies and storage. When omitted, Kernel obtains isolated - * browser capacity in the caller's account and releases it after retrieval. That - * capacity is not retained for later interaction. Existing browser quotas apply. + * Reuses its cookies, proxy, and browser configuration; requests follow that + * browser's existing network access behavior, with no additional destination + * allowlist in this endpoint. Kernel does not delete a caller-supplied browser. + * Render mode uses a temporary tab; website activity may still change shared + * cookies and storage. When omitted and any result needs browser retrieval, Kernel + * creates one temporary browser for the request using the dashboard launch + * defaults (headful, stealth, default proxy), tags it with search_id, and deletes + * it when the request finishes. It is billed and counts toward browser concurrency + * like any other browser. A concurrency rejection returns 429 for source=browser; + * for source=auto, results with retained content are still returned and the rest + * report the rejection. */ browser_id?: string; @@ -154,7 +163,9 @@ export namespace Response { url: string; /** - * Kernel cache outcome. Provider-internal cache behavior may be unknown. + * Kernel content cache outcome. Kernel has no content cache yet: responses report + * bypass or unknown, and hit and miss are reserved. Provider-internal cache + * behavior may be unknown. */ cache_status?: 'hit' | 'miss' | 'bypass' | 'unknown'; @@ -172,15 +183,21 @@ export namespace Response { extractor_version?: string; /** - * Origin retrieval time when known, not cache read time. + * When Kernel fetched the content, or received it from the provider for retained + * content. */ fetched_at?: string | null; /** - * Final retrieval URL when known. + * Final retrieval URL after redirects when known. Curl mode follows up to 5 + * redirects. */ final_url?: string; + /** + * Format of text. Plain-text and JSON pages are returned unchanged as text even + * when markdown was requested. + */ format?: 'markdown' | 'text'; /** @@ -189,7 +206,7 @@ export namespace Response { http_status?: number; /** - * Original retrieval method, including on cache hits. + * Original retrieval method. */ method?: 'provider' | 'browser_curl' | 'browser_render'; @@ -200,7 +217,8 @@ export namespace Response { text?: string; /** - * Whether max_chars truncated the extracted content. + * Whether the content was cut short, by max_chars or because the page exceeded the + * 1 MiB read limit. */ truncated?: boolean; } @@ -252,36 +270,37 @@ export namespace ContentFetchParams { */ export interface Content { /** - * Invalid with source=provider. Supplying browser_id requires source=browser so - * the chosen identity is not bypassed. + * Requires source=auto or source=browser in deferred retrieval. */ browser?: Content.Browser; format?: 'markdown' | 'text'; /** - * Maximum acceptable age of cached page content, measured from origin retrieval. 0 - * forces a live fetch. Governs the Kernel content cache, which is scoped to the - * caller organization and project and separated by retrieval context; fetches - * through a caller-supplied browser_id bypass that cache. Mapped to the provider - * freshness control when source is provider and the provider supports one; - * otherwise provider content age is reported as unknown via fetched_at. + * For source=auto, maximum acceptable age of retained provider content, measured + * from when the search received it from the provider. A value of 0 disables reuse + * of retained content, so every result is fetched through a browser. + * source=provider reuses retained provider content without freshness validation. + * source=browser always fetches through a browser and does not use this age limit. */ max_age_hours?: number; /** - * Per-result Unicode character limit after extraction. + * Per-result Unicode character limit after extraction. Retained provider content + * cannot exceed what was stored at search time; such results report truncated when + * the stored text was already truncated. */ max_chars?: number; /** - * provider uses the search provider's native content retrieval; browser fetches - * each URL through a Kernel browser; auto prefers Kernel browser retrieval and - * falls back to provider-native content when browser retrieval is unavailable or - * unsuitable. Defaults to auto for both inline and deferred retrieval. Deferred - * provider retrieval requires post_hoc capability; an explicit provider source - * without it is a 400. Missing documents produce per-result unavailable outcomes, - * not request failures. + * auto uses retained provider content within max_age_hours; for deferred retrieval + * it falls back to a Kernel browser (caller-supplied or temporary) for results + * without it. Inline retrieval never uses a browser. provider reuses retained + * provider content when available, without freshness validation, and never + * provisions a browser. browser fetches each URL through a Kernel browser, either + * caller-supplied or temporary. No option makes a new provider request. Defaults + * to auto for both inline and deferred retrieval. Missing documents produce + * per-result unavailable outcomes, not request failures. */ source?: 'auto' | 'provider' | 'browser'; @@ -294,17 +313,22 @@ export namespace ContentFetchParams { export namespace Content { /** - * Invalid with source=provider. Supplying browser_id requires source=browser so - * the chosen identity is not bypassed. + * Requires source=auto or source=browser in deferred retrieval. */ export interface Browser { /** * Existing browser session ID authorized for the caller and selected project. - * Reuses its cookies, proxy, and browser configuration. Kernel does not delete a - * caller-supplied browser. Render mode uses a temporary tab; website activity may - * still change shared cookies and storage. When omitted, Kernel obtains isolated - * browser capacity in the caller's account and releases it after retrieval. That - * capacity is not retained for later interaction. Existing browser quotas apply. + * Reuses its cookies, proxy, and browser configuration; requests follow that + * browser's existing network access behavior, with no additional destination + * allowlist in this endpoint. Kernel does not delete a caller-supplied browser. + * Render mode uses a temporary tab; website activity may still change shared + * cookies and storage. When omitted and any result needs browser retrieval, Kernel + * creates one temporary browser for the request using the dashboard launch + * defaults (headful, stealth, default proxy), tags it with search_id, and deletes + * it when the request finishes. It is billed and counts toward browser concurrency + * like any other browser. A concurrency rejection returns 429 for source=browser; + * for source=auto, results with retained content are still returned and the rest + * report the rejection. */ browser_id?: string; diff --git a/src/resources/search/search.ts b/src/resources/search/search.ts index e5a91878..bb3cc5cd 100644 --- a/src/resources/search/search.ts +++ b/src/resources/search/search.ts @@ -938,36 +938,37 @@ export interface Request { export namespace Request { export interface SearchContentOptions { /** - * Invalid with source=provider. Supplying browser_id requires source=browser so - * the chosen identity is not bypassed. + * Requires source=auto or source=browser in deferred retrieval. */ browser?: SearchContentOptions.Browser; format?: 'markdown' | 'text'; /** - * Maximum acceptable age of cached page content, measured from origin retrieval. 0 - * forces a live fetch. Governs the Kernel content cache, which is scoped to the - * caller organization and project and separated by retrieval context; fetches - * through a caller-supplied browser_id bypass that cache. Mapped to the provider - * freshness control when source is provider and the provider supports one; - * otherwise provider content age is reported as unknown via fetched_at. + * For source=auto, maximum acceptable age of retained provider content, measured + * from when the search received it from the provider. A value of 0 disables reuse + * of retained content, so every result is fetched through a browser. + * source=provider reuses retained provider content without freshness validation. + * source=browser always fetches through a browser and does not use this age limit. */ max_age_hours?: number; /** - * Per-result Unicode character limit after extraction. + * Per-result Unicode character limit after extraction. Retained provider content + * cannot exceed what was stored at search time; such results report truncated when + * the stored text was already truncated. */ max_chars?: number; /** - * provider uses the search provider's native content retrieval; browser fetches - * each URL through a Kernel browser; auto prefers Kernel browser retrieval and - * falls back to provider-native content when browser retrieval is unavailable or - * unsuitable. Defaults to auto for both inline and deferred retrieval. Deferred - * provider retrieval requires post_hoc capability; an explicit provider source - * without it is a 400. Missing documents produce per-result unavailable outcomes, - * not request failures. + * auto uses retained provider content within max_age_hours; for deferred retrieval + * it falls back to a Kernel browser (caller-supplied or temporary) for results + * without it. Inline retrieval never uses a browser. provider reuses retained + * provider content when available, without freshness validation, and never + * provisions a browser. browser fetches each URL through a Kernel browser, either + * caller-supplied or temporary. No option makes a new provider request. Defaults + * to auto for both inline and deferred retrieval. Missing documents produce + * per-result unavailable outcomes, not request failures. */ source?: 'auto' | 'provider' | 'browser'; @@ -980,17 +981,22 @@ export namespace Request { export namespace SearchContentOptions { /** - * Invalid with source=provider. Supplying browser_id requires source=browser so - * the chosen identity is not bypassed. + * Requires source=auto or source=browser in deferred retrieval. */ export interface Browser { /** * Existing browser session ID authorized for the caller and selected project. - * Reuses its cookies, proxy, and browser configuration. Kernel does not delete a - * caller-supplied browser. Render mode uses a temporary tab; website activity may - * still change shared cookies and storage. When omitted, Kernel obtains isolated - * browser capacity in the caller's account and releases it after retrieval. That - * capacity is not retained for later interaction. Existing browser quotas apply. + * Reuses its cookies, proxy, and browser configuration; requests follow that + * browser's existing network access behavior, with no additional destination + * allowlist in this endpoint. Kernel does not delete a caller-supplied browser. + * Render mode uses a temporary tab; website activity may still change shared + * cookies and storage. When omitted and any result needs browser retrieval, Kernel + * creates one temporary browser for the request using the dashboard launch + * defaults (headful, stealth, default proxy), tags it with search_id, and deletes + * it when the request finishes. It is billed and counts toward browser concurrency + * like any other browser. A concurrency rejection returns 429 for source=browser; + * for source=auto, results with retained content are still returned and the rest + * report the rejection. */ browser_id?: string; @@ -1070,7 +1076,9 @@ export namespace Result { status: 'ok' | 'unavailable' | 'blocked' | 'timeout' | 'unsupported_type' | 'extraction_failed' | 'error'; /** - * Kernel cache outcome. Provider-internal cache behavior may be unknown. + * Kernel content cache outcome. Kernel has no content cache yet: responses report + * bypass or unknown, and hit and miss are reserved. Provider-internal cache + * behavior may be unknown. */ cache_status?: 'hit' | 'miss' | 'bypass' | 'unknown'; @@ -1088,7 +1096,7 @@ export namespace Result { extractor_version?: string; /** - * Origin retrieval time when known, not cache read time. + * When Kernel received the content from the provider. */ fetched_at?: string | null; @@ -1105,7 +1113,7 @@ export namespace Result { http_status?: number; /** - * Original retrieval method, including on cache hits. + * Original retrieval method. */ method?: 'provider' | 'browser_curl' | 'browser_render'; @@ -1254,8 +1262,8 @@ export namespace Strategy { export interface Usage { /** - * Number of result URLs for which a Kernel browser retrieval was attempted, - * excluding cache-only hits. + * Number of result URLs for which a Kernel browser retrieval received a response + * from the target, excluding cache-only hits. */ content_fetches: number; @@ -1398,36 +1406,37 @@ export interface SearchCreateParams { export namespace SearchCreateParams { export interface SearchContentOptions { /** - * Invalid with source=provider. Supplying browser_id requires source=browser so - * the chosen identity is not bypassed. + * Requires source=auto or source=browser in deferred retrieval. */ browser?: SearchContentOptions.Browser; format?: 'markdown' | 'text'; /** - * Maximum acceptable age of cached page content, measured from origin retrieval. 0 - * forces a live fetch. Governs the Kernel content cache, which is scoped to the - * caller organization and project and separated by retrieval context; fetches - * through a caller-supplied browser_id bypass that cache. Mapped to the provider - * freshness control when source is provider and the provider supports one; - * otherwise provider content age is reported as unknown via fetched_at. + * For source=auto, maximum acceptable age of retained provider content, measured + * from when the search received it from the provider. A value of 0 disables reuse + * of retained content, so every result is fetched through a browser. + * source=provider reuses retained provider content without freshness validation. + * source=browser always fetches through a browser and does not use this age limit. */ max_age_hours?: number; /** - * Per-result Unicode character limit after extraction. + * Per-result Unicode character limit after extraction. Retained provider content + * cannot exceed what was stored at search time; such results report truncated when + * the stored text was already truncated. */ max_chars?: number; /** - * provider uses the search provider's native content retrieval; browser fetches - * each URL through a Kernel browser; auto prefers Kernel browser retrieval and - * falls back to provider-native content when browser retrieval is unavailable or - * unsuitable. Defaults to auto for both inline and deferred retrieval. Deferred - * provider retrieval requires post_hoc capability; an explicit provider source - * without it is a 400. Missing documents produce per-result unavailable outcomes, - * not request failures. + * auto uses retained provider content within max_age_hours; for deferred retrieval + * it falls back to a Kernel browser (caller-supplied or temporary) for results + * without it. Inline retrieval never uses a browser. provider reuses retained + * provider content when available, without freshness validation, and never + * provisions a browser. browser fetches each URL through a Kernel browser, either + * caller-supplied or temporary. No option makes a new provider request. Defaults + * to auto for both inline and deferred retrieval. Missing documents produce + * per-result unavailable outcomes, not request failures. */ source?: 'auto' | 'provider' | 'browser'; @@ -1440,17 +1449,22 @@ export namespace SearchCreateParams { export namespace SearchContentOptions { /** - * Invalid with source=provider. Supplying browser_id requires source=browser so - * the chosen identity is not bypassed. + * Requires source=auto or source=browser in deferred retrieval. */ export interface Browser { /** * Existing browser session ID authorized for the caller and selected project. - * Reuses its cookies, proxy, and browser configuration. Kernel does not delete a - * caller-supplied browser. Render mode uses a temporary tab; website activity may - * still change shared cookies and storage. When omitted, Kernel obtains isolated - * browser capacity in the caller's account and releases it after retrieval. That - * capacity is not retained for later interaction. Existing browser quotas apply. + * Reuses its cookies, proxy, and browser configuration; requests follow that + * browser's existing network access behavior, with no additional destination + * allowlist in this endpoint. Kernel does not delete a caller-supplied browser. + * Render mode uses a temporary tab; website activity may still change shared + * cookies and storage. When omitted and any result needs browser retrieval, Kernel + * creates one temporary browser for the request using the dashboard launch + * defaults (headful, stealth, default proxy), tags it with search_id, and deletes + * it when the request finishes. It is billed and counts toward browser concurrency + * like any other browser. A concurrency rejection returns 429 for source=browser; + * for source=auto, results with retained content are still returned and the rest + * report the rejection. */ browser_id?: string;