Skip to content

Commit 8bbdc36

Browse files
author
stlc-bot
committed
feat(highlights): preserve Markdown structure in scrape excerpts (#1236)
Stainless-Generated-From: ea0af0b
1 parent 5768952 commit 8bbdc36

1 file changed

Lines changed: 24 additions & 23 deletions

File tree

‎web.go‎

Lines changed: 24 additions & 23 deletions
Original file line numberDiff line numberDiff line change
@@ -87,21 +87,21 @@ func (r *WebService) MapURLs(ctx context.Context, query WebMapURLsParams, opts .
8787
// is shared with Markdown, parsed fields, product data, highlights, and JSON
8888
// extraction. Cached outputs can come from different visits within maxAgeMs; use 0
8989
// for a fresh capture. HTML-only requests use the existing fast acquisition path.
90-
// Highlights return the plain-text passages most relevant to
91-
// highlightsParams.query. Requests with at least one successful output cost one
92-
// base credit, including cache hits, or two with browser actions. All-failed
93-
// responses are unbilled except missing pages, which retain the base price and the
94-
// one-credit product charge when product was requested. Highlights add 3 credits
95-
// when passages are returned. JSON extraction runs an LLM over nonempty page
96-
// Markdown and adds four credits only when its result is returned successfully.
97-
// PDF OCR adds one credit per recovered page on fresh extraction. Product adds one
98-
// credit when its successful result is returned, plus six if that result used the
99-
// specialized model. Original response bytes and screenshots are limited to 20 MiB
100-
// each, screenshots to 40 megapixels, and the combined response to 60 MiB. An
101-
// oversized output has success: false and data: null. If the combined response
102-
// exceeds its limit, the largest outputs are marked failed until the remaining
103-
// outputs fit. Valid captured pieces may still be cached when omitted to meet the
104-
// response size limit.
90+
// Highlights return Markdown excerpts most relevant to highlightsParams.query.
91+
// Requests with at least one successful output cost one base credit, including
92+
// cache hits, or two with browser actions. All-failed responses are unbilled
93+
// except missing pages, which retain the base price and the one-credit product
94+
// charge when product was requested. Highlights add 3 credits when passages are
95+
// returned. JSON extraction runs an LLM over nonempty page Markdown and adds four
96+
// credits only when its result is returned successfully. PDF OCR adds one credit
97+
// per recovered page on fresh extraction. Product adds one credit when its
98+
// successful result is returned, plus six if that result used the specialized
99+
// model. Original response bytes and screenshots are limited to 20 MiB each,
100+
// screenshots to 40 megapixels, and the combined response to 60 MiB. An oversized
101+
// output has success: false and data: null. If the combined response exceeds its
102+
// limit, the largest outputs are marked failed until the remaining outputs fit.
103+
// Valid captured pieces may still be cached when omitted to meet the response size
104+
// limit.
105105
func (r *WebService) Scrape(ctx context.Context, body WebScrapeParams, opts ...option.RequestOption) (res *WebScrapeResponse, err error) {
106106
opts = slices.Concat(r.options, opts)
107107
path := "web/scrape"
@@ -1090,9 +1090,9 @@ type WebScrapeResponse struct {
10901090
// cache-controlled fetch contributing to the output was a hit; age_ms is the
10911091
// oldest contributing hit.
10921092
CacheMetadata WebScrapeResponseCacheMetadata `json:"cache_metadata" api:"required"`
1093-
// Relevant passages for your question or topic, in page order. A heading in square
1094-
// brackets is included when needed to interpret a passage. Empty when the page has
1095-
// no text.
1093+
// Relevant Markdown excerpts for your question or topic, in page order. Headings
1094+
// in square brackets supply necessary context; ellipses mark omitted portions.
1095+
// Empty when the page has no text.
10961096
Highlights WebScrapeResponseHighlights `json:"highlights" api:"required"`
10971097
// Rendered HTML after content filters.
10981098
HTML WebScrapeResponseHTML `json:"html" api:"required"`
@@ -1222,9 +1222,9 @@ func (r *WebScrapeResponseCacheMetadata) UnmarshalJSON(data []byte) error {
12221222
return apijson.UnmarshalRoot(data, r)
12231223
}
12241224

1225-
// Relevant passages for your question or topic, in page order. A heading in square
1226-
// brackets is included when needed to interpret a passage. Empty when the page has
1227-
// no text.
1225+
// Relevant Markdown excerpts for your question or topic, in page order. Headings
1226+
// in square brackets supply necessary context; ellipses mark omitted portions.
1227+
// Empty when the page has no text.
12281228
type WebScrapeResponseHighlights struct {
12291229
Data []string `json:"data" api:"required"`
12301230
Requested bool `json:"requested" api:"required"`
@@ -2851,8 +2851,9 @@ func (r *WebScrapeParams) UnmarshalJSON(data []byte) error {
28512851
type WebScrapeParamsFormats struct {
28522852
// The original HTTP response body.
28532853
Bytes param.Opt[bool] `json:"bytes,omitzero"`
2854-
// Relevant passages for your question or topic, with headings included when needed
2855-
// for context. Adds 3 credits when passages are returned.
2854+
// Relevant Markdown excerpts for your question or topic, preserving code, lists,
2855+
// and tables, with headings included when needed for context. Adds 3 credits when
2856+
// passages are returned.
28562857
Highlights param.Opt[bool] `json:"highlights,omitzero"`
28572858
// Rendered HTML.
28582859
HTML param.Opt[bool] `json:"html,omitzero"`

0 commit comments

Comments
 (0)