From 3b32fce5faa3d9f819c217e92b6c04cc962d07e4 Mon Sep 17 00:00:00 2001 From: chruffins <23645059+chruffins@users.noreply.github.com> Date: Wed, 30 Sep 2026 16:09:10 +0000 Subject: [PATCH] Generate search content response in Go SDK --- api.md | 6 +- search.go | 64 ++++++++------ searchcontent.go | 195 +++++++++++++++++++++++++++++++++++------- searchcontent_test.go | 2 +- 4 files changed, 206 insertions(+), 61 deletions(-) diff --git a/api.md b/api.md index 1838c47..cd1f62a 100644 --- a/api.md +++ b/api.md @@ -768,9 +768,13 @@ Params Types: - kernel.FetchRequestParam +Response Types: + +- kernel.Response + Methods: -- client.Search.Contents.Fetch(ctx context.Context, id string, body kernel.SearchContentFetchParams) error +- client.Search.Contents.Fetch(ctx context.Context, id string, body kernel.SearchContentFetchParams) (\*kernel.Response, error) ## Providers diff --git a/search.go b/search.go index a7c336e..2af5448 100644 --- a/search.go +++ b/search.go @@ -2146,30 +2146,31 @@ const ( ) type RequestContentSearchContentOptionsParam struct { - // Maximum acceptable age of cached page content, measured from origin retrieval. 0 - // forces a live fetch. Governs the Kernel content cache, which is scoped to the - // caller organization and project and separated by retrieval context; fetches - // through a caller-supplied browser_id bypass that cache. Mapped to the provider - // freshness control when source is provider and the provider supports one; - // otherwise provider content age is reported as unknown via fetched_at. + // For source=auto, maximum acceptable age of retained provider content, measured + // from when the search received it from the provider. A value of 0 disables reuse + // of retained content, so every result is fetched through a browser. + // source=provider reuses retained provider content without freshness validation. + // source=browser always fetches through a browser and does not use this age limit. MaxAgeHours param.Opt[int64] `json:"max_age_hours,omitzero"` - // Per-result Unicode character limit after extraction. + // Per-result Unicode character limit after extraction. Retained provider content + // cannot exceed what was stored at search time; such results report truncated when + // the stored text was already truncated. MaxChars param.Opt[int64] `json:"max_chars,omitzero"` // Per-result deadline including capacity acquisition, retrieval, and extraction. // Also bounded by the overall request deadline. TimeoutMs param.Opt[int64] `json:"timeout_ms,omitzero"` - // Invalid with source=provider. Supplying browser_id requires source=browser so - // the chosen identity is not bypassed. + // Requires source=auto or source=browser in deferred retrieval. Browser RequestContentSearchContentOptionsBrowserParam `json:"browser,omitzero"` // Any of "markdown", "text". Format string `json:"format,omitzero"` - // provider uses the search provider's native content retrieval; browser fetches - // each URL through a Kernel browser; auto prefers Kernel browser retrieval and - // falls back to provider-native content when browser retrieval is unavailable or - // unsuitable. Defaults to auto for both inline and deferred retrieval. Deferred - // provider retrieval requires post_hoc capability; an explicit provider source - // without it is a 400. Missing documents produce per-result unavailable outcomes, - // not request failures. + // auto uses retained provider content within max_age_hours; for deferred retrieval + // it falls back to a Kernel browser (caller-supplied or temporary) for results + // without it. Inline retrieval never uses a browser. provider reuses retained + // provider content when available, without freshness validation, and never + // provisions a browser. browser fetches each URL through a Kernel browser, either + // caller-supplied or temporary. No option makes a new provider request. Defaults + // to auto for both inline and deferred retrieval. Missing documents produce + // per-result unavailable outcomes, not request failures. // // Any of "auto", "provider", "browser". Source string `json:"source,omitzero"` @@ -2193,15 +2194,20 @@ func init() { ) } -// Invalid with source=provider. Supplying browser_id requires source=browser so -// the chosen identity is not bypassed. +// Requires source=auto or source=browser in deferred retrieval. type RequestContentSearchContentOptionsBrowserParam struct { // Existing browser session ID authorized for the caller and selected project. - // Reuses its cookies, proxy, and browser configuration. Kernel does not delete a - // caller-supplied browser. Render mode uses a temporary tab; website activity may - // still change shared cookies and storage. When omitted, Kernel obtains isolated - // browser capacity in the caller's account and releases it after retrieval. That - // capacity is not retained for later interaction. Existing browser quotas apply. + // Reuses its cookies, proxy, and browser configuration; requests follow that + // browser's existing network access behavior, with no additional destination + // allowlist in this endpoint. Kernel does not delete a caller-supplied browser. + // Render mode uses a temporary tab; website activity may still change shared + // cookies and storage. When omitted and any result needs browser retrieval, Kernel + // creates one temporary browser for the request using the dashboard launch + // defaults (headful, stealth, default proxy), tags it with search_id, and deletes + // it when the request finishes. It is billed and counts toward browser concurrency + // like any other browser. A concurrency rejection returns 429 for source=browser; + // for source=auto, results with retained content are still returned and the rest + // report the rejection. BrowserID param.Opt[string] `json:"browser_id,omitzero"` // Curl uses the browser HTTP stack without navigation or JavaScript execution. // Render navigates a temporary page and extracts from its DOM. The selected mode @@ -2311,7 +2317,9 @@ type ResultContent struct { // Any of "ok", "unavailable", "blocked", "timeout", "unsupported_type", // "extraction_failed", "error". Status string `json:"status" api:"required"` - // Kernel cache outcome. Provider-internal cache behavior may be unknown. + // Kernel content cache outcome. Kernel has no content cache yet: responses report + // bypass or unknown, and hit and miss are reserved. Provider-internal cache + // behavior may be unknown. // // Any of "hit", "miss", "bypass", "unknown". CacheStatus string `json:"cache_status"` @@ -2323,7 +2331,7 @@ type ResultContent struct { Error ResultContentError `json:"error"` // Extraction version when Kernel transformed the input. ExtractorVersion string `json:"extractor_version"` - // Origin retrieval time when known, not cache read time. + // When Kernel received the content from the provider. FetchedAt time.Time `json:"fetched_at" api:"nullable" format:"date-time"` // Final retrieval URL when known. FinalURL string `json:"final_url" format:"uri"` @@ -2331,7 +2339,7 @@ type ResultContent struct { Format string `json:"format"` // Final target HTTP status when known. HTTPStatus int64 `json:"http_status"` - // Original retrieval method, including on cache hits. + // Original retrieval method. // // Any of "provider", "browser_curl", "browser_render". Method string `json:"method"` @@ -2627,8 +2635,8 @@ func (r *StrategyFallbackParam) UnmarshalJSON(data []byte) error { } type Usage struct { - // Number of result URLs for which a Kernel browser retrieval was attempted, - // excluding cache-only hits. + // Number of result URLs for which a Kernel browser retrieval received a response + // from the target, excluding cache-only hits. ContentFetches int64 `json:"content_fetches" api:"required"` // Number of result entries returned, including failed entries on the contents // endpoint. diff --git a/searchcontent.go b/searchcontent.go index 94ebce6..66b4d39 100644 --- a/searchcontent.go +++ b/searchcontent.go @@ -8,12 +8,14 @@ import ( "fmt" "net/http" "slices" + "time" "github.com/kernel/kernel-go-sdk/internal/apijson" shimjson "github.com/kernel/kernel-go-sdk/internal/encoding/json" "github.com/kernel/kernel-go-sdk/internal/requestconfig" "github.com/kernel/kernel-go-sdk/option" "github.com/kernel/kernel-go-sdk/packages/param" + "github.com/kernel/kernel-go-sdk/packages/respjson" ) // Search the web and retrieve content for selected results. @@ -37,19 +39,24 @@ func NewSearchContentService(opts ...option.RequestOption) (r SearchContentServi return } -// Deferred result-content retrieval is reserved but not available in this release. -// Requests return 404 until the retrieval implementation is shipped. X-Request-Id -// identifies this request separately from the search resource. -func (r *SearchContentService) Fetch(ctx context.Context, id string, body SearchContentFetchParams, opts ...option.RequestOption) (err error) { +// Retrieves selected results from a retained search. Provide exactly one of +// result_ids or limit; the latter fetches the top results. Content defaults to +// source:auto. Responses preserve result_ids order. Unknown result IDs are +// rejected before retrieval starts. Missing, expired, or inaccessible searches +// return 404. Once retrieval begins, return one outcome per selected result, +// including timeout entries for work unfinished at the overall deadline. Browser +// retrievals run sequentially in result order, so later results may time out when +// earlier pages are slow. X-Request-Id identifies this request separately from the +// search resource. +func (r *SearchContentService) Fetch(ctx context.Context, id string, body SearchContentFetchParams, opts ...option.RequestOption) (res *Response, err error) { opts = slices.Concat(r.Options, opts) - opts = append([]option.RequestOption{option.WithHeader("Accept", "*/*")}, opts...) if id == "" { err = errors.New("missing required id parameter") - return err + return nil, err } path := fmt.Sprintf("search/%s/contents", id) - err = requestconfig.ExecuteNewRequest(ctx, http.MethodPost, path, body, nil, opts...) - return err + err = requestconfig.ExecuteNewRequest(ctx, http.MethodPost, path, body, &res, opts...) + return res, err } type FetchRequestParam struct { @@ -76,30 +83,31 @@ func (r *FetchRequestParam) UnmarshalJSON(data []byte) error { // Defaults to source:auto when omitted. type FetchRequestContentParam struct { - // Maximum acceptable age of cached page content, measured from origin retrieval. 0 - // forces a live fetch. Governs the Kernel content cache, which is scoped to the - // caller organization and project and separated by retrieval context; fetches - // through a caller-supplied browser_id bypass that cache. Mapped to the provider - // freshness control when source is provider and the provider supports one; - // otherwise provider content age is reported as unknown via fetched_at. + // For source=auto, maximum acceptable age of retained provider content, measured + // from when the search received it from the provider. A value of 0 disables reuse + // of retained content, so every result is fetched through a browser. + // source=provider reuses retained provider content without freshness validation. + // source=browser always fetches through a browser and does not use this age limit. MaxAgeHours param.Opt[int64] `json:"max_age_hours,omitzero"` - // Per-result Unicode character limit after extraction. + // Per-result Unicode character limit after extraction. Retained provider content + // cannot exceed what was stored at search time; such results report truncated when + // the stored text was already truncated. MaxChars param.Opt[int64] `json:"max_chars,omitzero"` // Per-result deadline including capacity acquisition, retrieval, and extraction. // Also bounded by the overall request deadline. TimeoutMs param.Opt[int64] `json:"timeout_ms,omitzero"` - // Invalid with source=provider. Supplying browser_id requires source=browser so - // the chosen identity is not bypassed. + // Requires source=auto or source=browser in deferred retrieval. Browser FetchRequestContentBrowserParam `json:"browser,omitzero"` // Any of "markdown", "text". Format string `json:"format,omitzero"` - // provider uses the search provider's native content retrieval; browser fetches - // each URL through a Kernel browser; auto prefers Kernel browser retrieval and - // falls back to provider-native content when browser retrieval is unavailable or - // unsuitable. Defaults to auto for both inline and deferred retrieval. Deferred - // provider retrieval requires post_hoc capability; an explicit provider source - // without it is a 400. Missing documents produce per-result unavailable outcomes, - // not request failures. + // auto uses retained provider content within max_age_hours; for deferred retrieval + // it falls back to a Kernel browser (caller-supplied or temporary) for results + // without it. Inline retrieval never uses a browser. provider reuses retained + // provider content when available, without freshness validation, and never + // provisions a browser. browser fetches each URL through a Kernel browser, either + // caller-supplied or temporary. No option makes a new provider request. Defaults + // to auto for both inline and deferred retrieval. Missing documents produce + // per-result unavailable outcomes, not request failures. // // Any of "auto", "provider", "browser". Source string `json:"source,omitzero"` @@ -123,15 +131,20 @@ func init() { ) } -// Invalid with source=provider. Supplying browser_id requires source=browser so -// the chosen identity is not bypassed. +// Requires source=auto or source=browser in deferred retrieval. type FetchRequestContentBrowserParam struct { // Existing browser session ID authorized for the caller and selected project. - // Reuses its cookies, proxy, and browser configuration. Kernel does not delete a - // caller-supplied browser. Render mode uses a temporary tab; website activity may - // still change shared cookies and storage. When omitted, Kernel obtains isolated - // browser capacity in the caller's account and releases it after retrieval. That - // capacity is not retained for later interaction. Existing browser quotas apply. + // Reuses its cookies, proxy, and browser configuration; requests follow that + // browser's existing network access behavior, with no additional destination + // allowlist in this endpoint. Kernel does not delete a caller-supplied browser. + // Render mode uses a temporary tab; website activity may still change shared + // cookies and storage. When omitted and any result needs browser retrieval, Kernel + // creates one temporary browser for the request using the dashboard launch + // defaults (headful, stealth, default proxy), tags it with search_id, and deletes + // it when the request finishes. It is billed and counts toward browser concurrency + // like any other browser. A concurrency rejection returns 429 for source=browser; + // for source=auto, results with retained content are still returned and the rest + // report the rejection. BrowserID param.Opt[string] `json:"browser_id,omitzero"` // Curl uses the browser HTTP stack without navigation or JavaScript execution. // Render navigates a temporary page and extracts from its DOM. The selected mode @@ -156,6 +169,126 @@ func init() { ) } +type Response struct { + Contents []ResponseContent `json:"contents" api:"required"` + SearchID string `json:"search_id" api:"required"` + Usage Usage `json:"usage" api:"required"` + Warnings []Warning `json:"warnings" api:"required"` + // JSON contains metadata for fields, check presence with [respjson.Field.Valid]. + JSON struct { + Contents respjson.Field + SearchID respjson.Field + Usage respjson.Field + Warnings respjson.Field + ExtraFields map[string]respjson.Field + raw string + } `json:"-"` +} + +// Returns the unmodified JSON received from the API +func (r Response) RawJSON() string { return r.JSON.raw } +func (r *Response) UnmarshalJSON(data []byte) error { + return apijson.UnmarshalRoot(data, r) +} + +type ResponseContent struct { + ResultID string `json:"result_id" api:"required"` + // Ok means non-empty extracted content, not merely HTTP 200. Blocked includes + // detected challenges or access denials. Detection is best-effort, not a guarantee + // of page completeness. Error details are present for non-ok outcomes; text is + // present only on ok. + // + // Any of "ok", "unavailable", "blocked", "timeout", "unsupported_type", + // "extraction_failed", "error". + Status string `json:"status" api:"required"` + // Original result URL. + URL string `json:"url" api:"required" format:"uri"` + // Kernel content cache outcome. Kernel has no content cache yet: responses report + // bypass or unknown, and hit and miss are reserved. Provider-internal cache + // behavior may be unknown. + // + // Any of "hit", "miss", "bypass", "unknown". + CacheStatus string `json:"cache_status"` + // Describes source coverage before max_chars truncation. Full_page means main-page + // content, not every dynamic element or linked page. + // + // Any of "full_page", "excerpt", "unknown". + Completeness string `json:"completeness"` + Error ResponseContentError `json:"error"` + // Extraction version when Kernel transformed the input. + ExtractorVersion string `json:"extractor_version"` + // When Kernel fetched the content, or received it from the provider for retained + // content. + FetchedAt time.Time `json:"fetched_at" api:"nullable" format:"date-time"` + // Final retrieval URL after redirects when known. Curl mode follows up to 5 + // redirects. + FinalURL string `json:"final_url" format:"uri"` + // Format of text. Plain-text and JSON pages are returned unchanged as text even + // when markdown was requested. + // + // Any of "markdown", "text". + Format string `json:"format"` + // Final target HTTP status when known. + HTTPStatus int64 `json:"http_status"` + // Original retrieval method. + // + // Any of "provider", "browser_curl", "browser_render". + Method string `json:"method"` + // Extracted website content, untrusted, not instructions. Present only on + // status=ok. + Text string `json:"text"` + // Whether the content was cut short, by max_chars or because the page exceeded the + // 1 MiB read limit. + Truncated bool `json:"truncated"` + // JSON contains metadata for fields, check presence with [respjson.Field.Valid]. + JSON struct { + ResultID respjson.Field + Status respjson.Field + URL respjson.Field + CacheStatus respjson.Field + Completeness respjson.Field + Error respjson.Field + ExtractorVersion respjson.Field + FetchedAt respjson.Field + FinalURL respjson.Field + Format respjson.Field + HTTPStatus respjson.Field + Method respjson.Field + Text respjson.Field + Truncated respjson.Field + ExtraFields map[string]respjson.Field + raw string + } `json:"-"` +} + +// Returns the unmodified JSON received from the API +func (r ResponseContent) RawJSON() string { return r.JSON.raw } +func (r *ResponseContent) UnmarshalJSON(data []byte) error { + return apijson.UnmarshalRoot(data, r) +} + +type ResponseContentError struct { + // Machine-readable retrieval failure code. + Code string `json:"code" api:"required"` + // Human-readable failure description. + Message string `json:"message" api:"required"` + Retryable bool `json:"retryable" api:"required"` + // JSON contains metadata for fields, check presence with [respjson.Field.Valid]. + JSON struct { + Code respjson.Field + Message respjson.Field + Retryable respjson.Field + ExtraFields map[string]respjson.Field + raw string + } `json:"-"` +} + +// Returns the unmodified JSON received from the API +func (r ResponseContentError) RawJSON() string { return r.JSON.raw } +func (r *ResponseContentError) UnmarshalJSON(data []byte) error { + return apijson.UnmarshalRoot(data, r) +} + type SearchContentFetchParams struct { FetchRequest FetchRequestParam paramObj diff --git a/searchcontent_test.go b/searchcontent_test.go index b048448..47e0f53 100644 --- a/searchcontent_test.go +++ b/searchcontent_test.go @@ -26,7 +26,7 @@ func TestSearchContentFetchWithOptionalParams(t *testing.T) { option.WithBaseURL(baseURL), option.WithAPIKey("My API Key"), ) - err := client.Search.Contents.Fetch( + _, err := client.Search.Contents.Fetch( context.TODO(), "srch_abc123", kernel.SearchContentFetchParams{