From a52d60a841fc36c38c1a9d792ec55ca7161094c3 Mon Sep 17 00:00:00 2001 From: "stainless-app[bot]" <142633134+stainless-app[bot]@users.noreply.github.com> Date: Sat, 4 Apr 2026 21:35:10 +0000 Subject: [PATCH 1/4] codegen metadata --- .stats.yml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/.stats.yml b/.stats.yml index a0614e16..2b073939 100644 --- a/.stats.yml +++ b/.stats.yml @@ -1,4 +1,4 @@ configured_endpoints: 20 -openapi_spec_url: https://storage.googleapis.com/stainless-sdk-openapi-specs/context-dev%2Fcontext.dev-56a21db16ac3a797f86daca01a2ea115a0365db4b2f9d8accdec4f4d3ee2eb83.yml -openapi_spec_hash: bfcef090896da96023c4485a3d69e350 +openapi_spec_url: https://storage.googleapis.com/stainless-sdk-openapi-specs/context-dev%2Fcontext.dev-97cdb78dc0d72e9df643a89660f2b0c9687f12c6e4d93f7767f6cfc1b4f2e4c7.yml +openapi_spec_hash: 92fc94fd8865fabe78c2667490ca3884 config_hash: 38268bb88fc4dcbb8f2f94dd138b5910 From b8cd2443bf04824d5d769a555c140997c92c3ef4 Mon Sep 17 00:00:00 2001 From: "stainless-app[bot]" <142633134+stainless-app[bot]@users.noreply.github.com> Date: Sat, 4 Apr 2026 21:41:15 +0000 Subject: [PATCH 2/4] codegen metadata --- .stats.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.stats.yml b/.stats.yml index 2b073939..3542b112 100644 --- a/.stats.yml +++ b/.stats.yml @@ -1,4 +1,4 @@ configured_endpoints: 20 openapi_spec_url: https://storage.googleapis.com/stainless-sdk-openapi-specs/context-dev%2Fcontext.dev-97cdb78dc0d72e9df643a89660f2b0c9687f12c6e4d93f7767f6cfc1b4f2e4c7.yml openapi_spec_hash: 92fc94fd8865fabe78c2667490ca3884 -config_hash: 38268bb88fc4dcbb8f2f94dd138b5910 +config_hash: 51eb368cba05800d9497df4fa318828e From 8e8fcc26f2fbbb2bdcca9713fa3b9f8518303586 Mon Sep 17 00:00:00 2001 From: "stainless-app[bot]" <142633134+stainless-app[bot]@users.noreply.github.com> Date: Sat, 4 Apr 2026 21:42:36 +0000 Subject: [PATCH 3/4] feat(api): manual updates --- .stats.yml | 4 +- lib/context_dev.rb | 2 + lib/context_dev/models.rb | 2 + .../models/web_web_crawl_md_params.rb | 92 +++++++ .../models/web_web_crawl_md_response.rb | 121 +++++++++ lib/context_dev/resources/web.rb | 43 ++++ rbi/context_dev/models.rbi | 2 + .../models/web_web_crawl_md_params.rbi | 137 +++++++++++ .../models/web_web_crawl_md_response.rbi | 230 ++++++++++++++++++ rbi/context_dev/resources/web.rbi | 43 ++++ sig/context_dev/models.rbs | 2 + .../models/web_web_crawl_md_params.rbs | 82 +++++++ .../models/web_web_crawl_md_response.rbs | 116 +++++++++ sig/context_dev/resources/web.rbs | 13 + test/context_dev/resources/web_test.rb | 17 ++ 15 files changed, 904 insertions(+), 2 deletions(-) create mode 100644 lib/context_dev/models/web_web_crawl_md_params.rb create mode 100644 lib/context_dev/models/web_web_crawl_md_response.rb create mode 100644 rbi/context_dev/models/web_web_crawl_md_params.rbi create mode 100644 rbi/context_dev/models/web_web_crawl_md_response.rbi create mode 100644 sig/context_dev/models/web_web_crawl_md_params.rbs create mode 100644 sig/context_dev/models/web_web_crawl_md_response.rbs diff --git a/.stats.yml b/.stats.yml index 3542b112..a13a02a8 100644 --- a/.stats.yml +++ b/.stats.yml @@ -1,4 +1,4 @@ -configured_endpoints: 20 +configured_endpoints: 21 openapi_spec_url: https://storage.googleapis.com/stainless-sdk-openapi-specs/context-dev%2Fcontext.dev-97cdb78dc0d72e9df643a89660f2b0c9687f12c6e4d93f7767f6cfc1b4f2e4c7.yml openapi_spec_hash: 92fc94fd8865fabe78c2667490ca3884 -config_hash: 51eb368cba05800d9497df4fa318828e +config_hash: 682b89b02a20f5d1c13e2c91ecbcf5ce diff --git a/lib/context_dev.rb b/lib/context_dev.rb index ca1d98a3..3358e6ee 100644 --- a/lib/context_dev.rb +++ b/lib/context_dev.rb @@ -84,6 +84,8 @@ require_relative "context_dev/models/utility_prefetch_response" require_relative "context_dev/models/web_screenshot_params" require_relative "context_dev/models/web_screenshot_response" +require_relative "context_dev/models/web_web_crawl_md_params" +require_relative "context_dev/models/web_web_crawl_md_response" require_relative "context_dev/models/web_web_scrape_html_params" require_relative "context_dev/models/web_web_scrape_html_response" require_relative "context_dev/models/web_web_scrape_images_params" diff --git a/lib/context_dev/models.rb b/lib/context_dev/models.rb index 45dea9e9..9be91f2b 100644 --- a/lib/context_dev/models.rb +++ b/lib/context_dev/models.rb @@ -71,6 +71,8 @@ module ContextDev WebScreenshotParams = ContextDev::Models::WebScreenshotParams + WebWebCrawlMdParams = ContextDev::Models::WebWebCrawlMdParams + WebWebScrapeHTMLParams = ContextDev::Models::WebWebScrapeHTMLParams WebWebScrapeImagesParams = ContextDev::Models::WebWebScrapeImagesParams diff --git a/lib/context_dev/models/web_web_crawl_md_params.rb b/lib/context_dev/models/web_web_crawl_md_params.rb new file mode 100644 index 00000000..49ea3480 --- /dev/null +++ b/lib/context_dev/models/web_web_crawl_md_params.rb @@ -0,0 +1,92 @@ +# frozen_string_literal: true + +module ContextDev + module Models + # @see ContextDev::Resources::Web#web_crawl_md + class WebWebCrawlMdParams < ContextDev::Internal::Type::BaseModel + extend ContextDev::Internal::Type::RequestParameters::Converter + include ContextDev::Internal::Type::RequestParameters + + # @!attribute url + # The starting URL for the crawl (must include http:// or https:// protocol) + # + # @return [String] + required :url, String + + # @!attribute follow_subdomains + # When true, follow links on subdomains of the starting URL's domain (e.g. + # docs.example.com when starting from example.com). www and apex are always + # treated as equivalent. + # + # @return [Boolean, nil] + optional :follow_subdomains, ContextDev::Internal::Type::Boolean, api_name: :followSubdomains + + # @!attribute include_images + # Include image references in the Markdown output + # + # @return [Boolean, nil] + optional :include_images, ContextDev::Internal::Type::Boolean, api_name: :includeImages + + # @!attribute include_links + # Preserve hyperlinks in the Markdown output + # + # @return [Boolean, nil] + optional :include_links, ContextDev::Internal::Type::Boolean, api_name: :includeLinks + + # @!attribute max_depth + # Maximum link depth from the starting URL (0 = only the starting page) + # + # @return [Integer, nil] + optional :max_depth, Integer, api_name: :maxDepth + + # @!attribute max_pages + # Maximum number of pages to crawl. Hard cap: 500. + # + # @return [Integer, nil] + optional :max_pages, Integer, api_name: :maxPages + + # @!attribute shorten_base64_images + # Truncate base64-encoded image data in the Markdown output + # + # @return [Boolean, nil] + optional :shorten_base64_images, ContextDev::Internal::Type::Boolean, api_name: :shortenBase64Images + + # @!attribute url_regex + # Regex pattern. Only URLs matching this pattern will be followed and scraped. + # + # @return [String, nil] + optional :url_regex, String, api_name: :urlRegex + + # @!attribute use_main_content_only + # Extract only the main content, stripping headers, footers, sidebars, and + # navigation + # + # @return [Boolean, nil] + optional :use_main_content_only, ContextDev::Internal::Type::Boolean, api_name: :useMainContentOnly + + # @!method initialize(url:, follow_subdomains: nil, include_images: nil, include_links: nil, max_depth: nil, max_pages: nil, shorten_base64_images: nil, url_regex: nil, use_main_content_only: nil, request_options: {}) + # Some parameter documentations has been truncated, see + # {ContextDev::Models::WebWebCrawlMdParams} for more details. + # + # @param url [String] The starting URL for the crawl (must include http:// or https:// protocol) + # + # @param follow_subdomains [Boolean] When true, follow links on subdomains of the starting URL's domain (e.g. docs.ex + # + # @param include_images [Boolean] Include image references in the Markdown output + # + # @param include_links [Boolean] Preserve hyperlinks in the Markdown output + # + # @param max_depth [Integer] Maximum link depth from the starting URL (0 = only the starting page) + # + # @param max_pages [Integer] Maximum number of pages to crawl. Hard cap: 500. + # + # @param shorten_base64_images [Boolean] Truncate base64-encoded image data in the Markdown output + # + # @param url_regex [String] Regex pattern. Only URLs matching this pattern will be followed and scraped. + # + # @param use_main_content_only [Boolean] Extract only the main content, stripping headers, footers, sidebars, and navigat + # + # @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}] + end + end +end diff --git a/lib/context_dev/models/web_web_crawl_md_response.rb b/lib/context_dev/models/web_web_crawl_md_response.rb new file mode 100644 index 00000000..45f09d72 --- /dev/null +++ b/lib/context_dev/models/web_web_crawl_md_response.rb @@ -0,0 +1,121 @@ +# frozen_string_literal: true + +module ContextDev + module Models + # @see ContextDev::Resources::Web#web_crawl_md + class WebWebCrawlMdResponse < ContextDev::Internal::Type::BaseModel + # @!attribute metadata + # + # @return [ContextDev::Models::WebWebCrawlMdResponse::Metadata] + required :metadata, -> { ContextDev::Models::WebWebCrawlMdResponse::Metadata } + + # @!attribute results + # + # @return [Array] + required :results, + -> { ContextDev::Internal::Type::ArrayOf[ContextDev::Models::WebWebCrawlMdResponse::Result] } + + # @!method initialize(metadata:, results:) + # @param metadata [ContextDev::Models::WebWebCrawlMdResponse::Metadata] + # @param results [Array] + + # @see ContextDev::Models::WebWebCrawlMdResponse#metadata + class Metadata < ContextDev::Internal::Type::BaseModel + # @!attribute max_crawl_depth + # Maximum crawl depth reached during the crawl + # + # @return [Integer] + required :max_crawl_depth, Integer, api_name: :maxCrawlDepth + + # @!attribute num_failed + # Number of pages that failed to crawl + # + # @return [Integer] + required :num_failed, Integer, api_name: :numFailed + + # @!attribute num_succeeded + # Number of pages successfully crawled + # + # @return [Integer] + required :num_succeeded, Integer, api_name: :numSucceeded + + # @!attribute num_urls + # Total number of URLs crawled + # + # @return [Integer] + required :num_urls, Integer, api_name: :numUrls + + # @!method initialize(max_crawl_depth:, num_failed:, num_succeeded:, num_urls:) + # @param max_crawl_depth [Integer] Maximum crawl depth reached during the crawl + # + # @param num_failed [Integer] Number of pages that failed to crawl + # + # @param num_succeeded [Integer] Number of pages successfully crawled + # + # @param num_urls [Integer] Total number of URLs crawled + end + + class Result < ContextDev::Internal::Type::BaseModel + # @!attribute markdown + # Extracted page content as Markdown (empty string on failure) + # + # @return [String] + required :markdown, String + + # @!attribute metadata + # + # @return [ContextDev::Models::WebWebCrawlMdResponse::Result::Metadata] + required :metadata, -> { ContextDev::Models::WebWebCrawlMdResponse::Result::Metadata } + + # @!method initialize(markdown:, metadata:) + # @param markdown [String] Extracted page content as Markdown (empty string on failure) + # + # @param metadata [ContextDev::Models::WebWebCrawlMdResponse::Result::Metadata] + + # @see ContextDev::Models::WebWebCrawlMdResponse::Result#metadata + class Metadata < ContextDev::Internal::Type::BaseModel + # @!attribute crawl_depth + # Depth relative to the start URL. 0 = start URL, 1 = one link away. + # + # @return [Integer] + required :crawl_depth, Integer, api_name: :crawlDepth + + # @!attribute status_code + # HTTP status code of the response + # + # @return [Integer] + required :status_code, Integer, api_name: :statusCode + + # @!attribute success + # true if the page was fetched and parsed successfully + # + # @return [Boolean] + required :success, ContextDev::Internal::Type::Boolean + + # @!attribute title + # The page's content (empty string if unavailable) + # + # @return [String] + required :title, String + + # @!attribute url + # The URL that was fetched + # + # @return [String] + required :url, String + + # @!method initialize(crawl_depth:, status_code:, success:, title:, url:) + # @param crawl_depth [Integer] Depth relative to the start URL. 0 = start URL, 1 = one link away. + # + # @param status_code [Integer] HTTP status code of the response + # + # @param success [Boolean] true if the page was fetched and parsed successfully + # + # @param title [String] The page's <title> content (empty string if unavailable) + # + # @param url [String] The URL that was fetched + end + end + end + end +end diff --git a/lib/context_dev/resources/web.rb b/lib/context_dev/resources/web.rb index ee6d675c..ef458c53 100644 --- a/lib/context_dev/resources/web.rb +++ b/lib/context_dev/resources/web.rb @@ -38,6 +38,49 @@ def screenshot(params) ) end + # Some parameter documentations has been truncated, see + # {ContextDev::Models::WebWebCrawlMdParams} for more details. + # + # Performs a crawl starting from a given URL, extracts page content as Markdown, + # and returns results for all crawled pages. Only follows links within the same + # domain as the starting URL. Costs 1 credit per successful page crawled. + # + # @overload web_crawl_md(url:, follow_subdomains: nil, include_images: nil, include_links: nil, max_depth: nil, max_pages: nil, shorten_base64_images: nil, url_regex: nil, use_main_content_only: nil, request_options: {}) + # + # @param url [String] The starting URL for the crawl (must include http:// or https:// protocol) + # + # @param follow_subdomains [Boolean] When true, follow links on subdomains of the starting URL's domain (e.g. docs.ex + # + # @param include_images [Boolean] Include image references in the Markdown output + # + # @param include_links [Boolean] Preserve hyperlinks in the Markdown output + # + # @param max_depth [Integer] Maximum link depth from the starting URL (0 = only the starting page) + # + # @param max_pages [Integer] Maximum number of pages to crawl. Hard cap: 500. + # + # @param shorten_base64_images [Boolean] Truncate base64-encoded image data in the Markdown output + # + # @param url_regex [String] Regex pattern. Only URLs matching this pattern will be followed and scraped. + # + # @param use_main_content_only [Boolean] Extract only the main content, stripping headers, footers, sidebars, and navigat + # + # @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil] + # + # @return [ContextDev::Models::WebWebCrawlMdResponse] + # + # @see ContextDev::Models::WebWebCrawlMdParams + def web_crawl_md(params) + parsed, options = ContextDev::WebWebCrawlMdParams.dump_request(params) + @client.request( + method: :post, + path: "web/crawl", + body: parsed, + model: ContextDev::Models::WebWebCrawlMdResponse, + options: options + ) + end + # Scrapes the given URL and returns the raw HTML content of the page. # # @overload web_scrape_html(url:, request_options: {}) diff --git a/rbi/context_dev/models.rbi b/rbi/context_dev/models.rbi index 3bcbeafe..3932e7df 100644 --- a/rbi/context_dev/models.rbi +++ b/rbi/context_dev/models.rbi @@ -37,6 +37,8 @@ module ContextDev WebScreenshotParams = ContextDev::Models::WebScreenshotParams + WebWebCrawlMdParams = ContextDev::Models::WebWebCrawlMdParams + WebWebScrapeHTMLParams = ContextDev::Models::WebWebScrapeHTMLParams WebWebScrapeImagesParams = ContextDev::Models::WebWebScrapeImagesParams diff --git a/rbi/context_dev/models/web_web_crawl_md_params.rbi b/rbi/context_dev/models/web_web_crawl_md_params.rbi new file mode 100644 index 00000000..4559335d --- /dev/null +++ b/rbi/context_dev/models/web_web_crawl_md_params.rbi @@ -0,0 +1,137 @@ +# typed: strong + +module ContextDev + module Models + class WebWebCrawlMdParams < ContextDev::Internal::Type::BaseModel + extend ContextDev::Internal::Type::RequestParameters::Converter + include ContextDev::Internal::Type::RequestParameters + + OrHash = + T.type_alias do + T.any(ContextDev::WebWebCrawlMdParams, ContextDev::Internal::AnyHash) + end + + # The starting URL for the crawl (must include http:// or https:// protocol) + sig { returns(String) } + attr_accessor :url + + # When true, follow links on subdomains of the starting URL's domain (e.g. + # docs.example.com when starting from example.com). www and apex are always + # treated as equivalent. + sig { returns(T.nilable(T::Boolean)) } + attr_reader :follow_subdomains + + sig { params(follow_subdomains: T::Boolean).void } + attr_writer :follow_subdomains + + # Include image references in the Markdown output + sig { returns(T.nilable(T::Boolean)) } + attr_reader :include_images + + sig { params(include_images: T::Boolean).void } + attr_writer :include_images + + # Preserve hyperlinks in the Markdown output + sig { returns(T.nilable(T::Boolean)) } + attr_reader :include_links + + sig { params(include_links: T::Boolean).void } + attr_writer :include_links + + # Maximum link depth from the starting URL (0 = only the starting page) + sig { returns(T.nilable(Integer)) } + attr_reader :max_depth + + sig { params(max_depth: Integer).void } + attr_writer :max_depth + + # Maximum number of pages to crawl. Hard cap: 500. + sig { returns(T.nilable(Integer)) } + attr_reader :max_pages + + sig { params(max_pages: Integer).void } + attr_writer :max_pages + + # Truncate base64-encoded image data in the Markdown output + sig { returns(T.nilable(T::Boolean)) } + attr_reader :shorten_base64_images + + sig { params(shorten_base64_images: T::Boolean).void } + attr_writer :shorten_base64_images + + # Regex pattern. Only URLs matching this pattern will be followed and scraped. + sig { returns(T.nilable(String)) } + attr_reader :url_regex + + sig { params(url_regex: String).void } + attr_writer :url_regex + + # Extract only the main content, stripping headers, footers, sidebars, and + # navigation + sig { returns(T.nilable(T::Boolean)) } + attr_reader :use_main_content_only + + sig { params(use_main_content_only: T::Boolean).void } + attr_writer :use_main_content_only + + sig do + params( + url: String, + follow_subdomains: T::Boolean, + include_images: T::Boolean, + include_links: T::Boolean, + max_depth: Integer, + max_pages: Integer, + shorten_base64_images: T::Boolean, + url_regex: String, + use_main_content_only: T::Boolean, + request_options: ContextDev::RequestOptions::OrHash + ).returns(T.attached_class) + end + def self.new( + # The starting URL for the crawl (must include http:// or https:// protocol) + url:, + # When true, follow links on subdomains of the starting URL's domain (e.g. + # docs.example.com when starting from example.com). www and apex are always + # treated as equivalent. + follow_subdomains: nil, + # Include image references in the Markdown output + include_images: nil, + # Preserve hyperlinks in the Markdown output + include_links: nil, + # Maximum link depth from the starting URL (0 = only the starting page) + max_depth: nil, + # Maximum number of pages to crawl. Hard cap: 500. + max_pages: nil, + # Truncate base64-encoded image data in the Markdown output + shorten_base64_images: nil, + # Regex pattern. Only URLs matching this pattern will be followed and scraped. + url_regex: nil, + # Extract only the main content, stripping headers, footers, sidebars, and + # navigation + use_main_content_only: nil, + request_options: {} + ) + end + + sig do + override.returns( + { + url: String, + follow_subdomains: T::Boolean, + include_images: T::Boolean, + include_links: T::Boolean, + max_depth: Integer, + max_pages: Integer, + shorten_base64_images: T::Boolean, + url_regex: String, + use_main_content_only: T::Boolean, + request_options: ContextDev::RequestOptions + } + ) + end + def to_hash + end + end + end +end diff --git a/rbi/context_dev/models/web_web_crawl_md_response.rbi b/rbi/context_dev/models/web_web_crawl_md_response.rbi new file mode 100644 index 00000000..59c19a19 --- /dev/null +++ b/rbi/context_dev/models/web_web_crawl_md_response.rbi @@ -0,0 +1,230 @@ +# typed: strong + +module ContextDev + module Models + class WebWebCrawlMdResponse < ContextDev::Internal::Type::BaseModel + OrHash = + T.type_alias do + T.any( + ContextDev::Models::WebWebCrawlMdResponse, + ContextDev::Internal::AnyHash + ) + end + + sig { returns(ContextDev::Models::WebWebCrawlMdResponse::Metadata) } + attr_reader :metadata + + sig do + params( + metadata: ContextDev::Models::WebWebCrawlMdResponse::Metadata::OrHash + ).void + end + attr_writer :metadata + + sig do + returns(T::Array[ContextDev::Models::WebWebCrawlMdResponse::Result]) + end + attr_accessor :results + + sig do + params( + metadata: ContextDev::Models::WebWebCrawlMdResponse::Metadata::OrHash, + results: + T::Array[ContextDev::Models::WebWebCrawlMdResponse::Result::OrHash] + ).returns(T.attached_class) + end + def self.new(metadata:, results:) + end + + sig do + override.returns( + { + metadata: ContextDev::Models::WebWebCrawlMdResponse::Metadata, + results: T::Array[ContextDev::Models::WebWebCrawlMdResponse::Result] + } + ) + end + def to_hash + end + + class Metadata < ContextDev::Internal::Type::BaseModel + OrHash = + T.type_alias do + T.any( + ContextDev::Models::WebWebCrawlMdResponse::Metadata, + ContextDev::Internal::AnyHash + ) + end + + # Maximum crawl depth reached during the crawl + sig { returns(Integer) } + attr_accessor :max_crawl_depth + + # Number of pages that failed to crawl + sig { returns(Integer) } + attr_accessor :num_failed + + # Number of pages successfully crawled + sig { returns(Integer) } + attr_accessor :num_succeeded + + # Total number of URLs crawled + sig { returns(Integer) } + attr_accessor :num_urls + + sig do + params( + max_crawl_depth: Integer, + num_failed: Integer, + num_succeeded: Integer, + num_urls: Integer + ).returns(T.attached_class) + end + def self.new( + # Maximum crawl depth reached during the crawl + max_crawl_depth:, + # Number of pages that failed to crawl + num_failed:, + # Number of pages successfully crawled + num_succeeded:, + # Total number of URLs crawled + num_urls: + ) + end + + sig do + override.returns( + { + max_crawl_depth: Integer, + num_failed: Integer, + num_succeeded: Integer, + num_urls: Integer + } + ) + end + def to_hash + end + end + + class Result < ContextDev::Internal::Type::BaseModel + OrHash = + T.type_alias do + T.any( + ContextDev::Models::WebWebCrawlMdResponse::Result, + ContextDev::Internal::AnyHash + ) + end + + # Extracted page content as Markdown (empty string on failure) + sig { returns(String) } + attr_accessor :markdown + + sig do + returns(ContextDev::Models::WebWebCrawlMdResponse::Result::Metadata) + end + attr_reader :metadata + + sig do + params( + metadata: + ContextDev::Models::WebWebCrawlMdResponse::Result::Metadata::OrHash + ).void + end + attr_writer :metadata + + sig do + params( + markdown: String, + metadata: + ContextDev::Models::WebWebCrawlMdResponse::Result::Metadata::OrHash + ).returns(T.attached_class) + end + def self.new( + # Extracted page content as Markdown (empty string on failure) + markdown:, + metadata: + ) + end + + sig do + override.returns( + { + markdown: String, + metadata: + ContextDev::Models::WebWebCrawlMdResponse::Result::Metadata + } + ) + end + def to_hash + end + + class Metadata < ContextDev::Internal::Type::BaseModel + OrHash = + T.type_alias do + T.any( + ContextDev::Models::WebWebCrawlMdResponse::Result::Metadata, + ContextDev::Internal::AnyHash + ) + end + + # Depth relative to the start URL. 0 = start URL, 1 = one link away. + sig { returns(Integer) } + attr_accessor :crawl_depth + + # HTTP status code of the response + sig { returns(Integer) } + attr_accessor :status_code + + # true if the page was fetched and parsed successfully + sig { returns(T::Boolean) } + attr_accessor :success + + # The page's <title> content (empty string if unavailable) + sig { returns(String) } + attr_accessor :title + + # The URL that was fetched + sig { returns(String) } + attr_accessor :url + + sig do + params( + crawl_depth: Integer, + status_code: Integer, + success: T::Boolean, + title: String, + url: String + ).returns(T.attached_class) + end + def self.new( + # Depth relative to the start URL. 0 = start URL, 1 = one link away. + crawl_depth:, + # HTTP status code of the response + status_code:, + # true if the page was fetched and parsed successfully + success:, + # The page's <title> content (empty string if unavailable) + title:, + # The URL that was fetched + url: + ) + end + + sig do + override.returns( + { + crawl_depth: Integer, + status_code: Integer, + success: T::Boolean, + title: String, + url: String + } + ) + end + def to_hash + end + end + end + end + end +end diff --git a/rbi/context_dev/resources/web.rbi b/rbi/context_dev/resources/web.rbi index 6afbab52..bc0964cd 100644 --- a/rbi/context_dev/resources/web.rbi +++ b/rbi/context_dev/resources/web.rbi @@ -38,6 +38,49 @@ module ContextDev ) end + # Performs a crawl starting from a given URL, extracts page content as Markdown, + # and returns results for all crawled pages. Only follows links within the same + # domain as the starting URL. Costs 1 credit per successful page crawled. + sig do + params( + url: String, + follow_subdomains: T::Boolean, + include_images: T::Boolean, + include_links: T::Boolean, + max_depth: Integer, + max_pages: Integer, + shorten_base64_images: T::Boolean, + url_regex: String, + use_main_content_only: T::Boolean, + request_options: ContextDev::RequestOptions::OrHash + ).returns(ContextDev::Models::WebWebCrawlMdResponse) + end + def web_crawl_md( + # The starting URL for the crawl (must include http:// or https:// protocol) + url:, + # When true, follow links on subdomains of the starting URL's domain (e.g. + # docs.example.com when starting from example.com). www and apex are always + # treated as equivalent. + follow_subdomains: nil, + # Include image references in the Markdown output + include_images: nil, + # Preserve hyperlinks in the Markdown output + include_links: nil, + # Maximum link depth from the starting URL (0 = only the starting page) + max_depth: nil, + # Maximum number of pages to crawl. Hard cap: 500. + max_pages: nil, + # Truncate base64-encoded image data in the Markdown output + shorten_base64_images: nil, + # Regex pattern. Only URLs matching this pattern will be followed and scraped. + url_regex: nil, + # Extract only the main content, stripping headers, footers, sidebars, and + # navigation + use_main_content_only: nil, + request_options: {} + ) + end + # Scrapes the given URL and returns the raw HTML content of the page. sig do params( diff --git a/sig/context_dev/models.rbs b/sig/context_dev/models.rbs index 07f73437..537fbf3d 100644 --- a/sig/context_dev/models.rbs +++ b/sig/context_dev/models.rbs @@ -31,6 +31,8 @@ module ContextDev class WebScreenshotParams = ContextDev::Models::WebScreenshotParams + class WebWebCrawlMdParams = ContextDev::Models::WebWebCrawlMdParams + class WebWebScrapeHTMLParams = ContextDev::Models::WebWebScrapeHTMLParams class WebWebScrapeImagesParams = ContextDev::Models::WebWebScrapeImagesParams diff --git a/sig/context_dev/models/web_web_crawl_md_params.rbs b/sig/context_dev/models/web_web_crawl_md_params.rbs new file mode 100644 index 00000000..28144efb --- /dev/null +++ b/sig/context_dev/models/web_web_crawl_md_params.rbs @@ -0,0 +1,82 @@ +module ContextDev + module Models + type web_web_crawl_md_params = + { + url: String, + follow_subdomains: bool, + include_images: bool, + include_links: bool, + max_depth: Integer, + max_pages: Integer, + :shorten_base64_images => bool, + url_regex: String, + use_main_content_only: bool + } + & ContextDev::Internal::Type::request_parameters + + class WebWebCrawlMdParams < ContextDev::Internal::Type::BaseModel + extend ContextDev::Internal::Type::RequestParameters::Converter + include ContextDev::Internal::Type::RequestParameters + + attr_accessor url: String + + attr_reader follow_subdomains: bool? + + def follow_subdomains=: (bool) -> bool + + attr_reader include_images: bool? + + def include_images=: (bool) -> bool + + attr_reader include_links: bool? + + def include_links=: (bool) -> bool + + attr_reader max_depth: Integer? + + def max_depth=: (Integer) -> Integer + + attr_reader max_pages: Integer? + + def max_pages=: (Integer) -> Integer + + attr_reader shorten_base64_images: bool? + + def shorten_base64_images=: (bool) -> bool + + attr_reader url_regex: String? + + def url_regex=: (String) -> String + + attr_reader use_main_content_only: bool? + + def use_main_content_only=: (bool) -> bool + + def initialize: ( + url: String, + ?follow_subdomains: bool, + ?include_images: bool, + ?include_links: bool, + ?max_depth: Integer, + ?max_pages: Integer, + ?shorten_base64_images: bool, + ?url_regex: String, + ?use_main_content_only: bool, + ?request_options: ContextDev::request_opts + ) -> void + + def to_hash: -> { + url: String, + follow_subdomains: bool, + include_images: bool, + include_links: bool, + max_depth: Integer, + max_pages: Integer, + :shorten_base64_images => bool, + url_regex: String, + use_main_content_only: bool, + request_options: ContextDev::RequestOptions + } + end + end +end diff --git a/sig/context_dev/models/web_web_crawl_md_response.rbs b/sig/context_dev/models/web_web_crawl_md_response.rbs new file mode 100644 index 00000000..a290b1e3 --- /dev/null +++ b/sig/context_dev/models/web_web_crawl_md_response.rbs @@ -0,0 +1,116 @@ +module ContextDev + module Models + type web_web_crawl_md_response = + { + metadata: ContextDev::Models::WebWebCrawlMdResponse::Metadata, + results: ::Array[ContextDev::Models::WebWebCrawlMdResponse::Result] + } + + class WebWebCrawlMdResponse < ContextDev::Internal::Type::BaseModel + attr_accessor metadata: ContextDev::Models::WebWebCrawlMdResponse::Metadata + + attr_accessor results: ::Array[ContextDev::Models::WebWebCrawlMdResponse::Result] + + def initialize: ( + metadata: ContextDev::Models::WebWebCrawlMdResponse::Metadata, + results: ::Array[ContextDev::Models::WebWebCrawlMdResponse::Result] + ) -> void + + def to_hash: -> { + metadata: ContextDev::Models::WebWebCrawlMdResponse::Metadata, + results: ::Array[ContextDev::Models::WebWebCrawlMdResponse::Result] + } + + type metadata = + { + max_crawl_depth: Integer, + num_failed: Integer, + num_succeeded: Integer, + num_urls: Integer + } + + class Metadata < ContextDev::Internal::Type::BaseModel + attr_accessor max_crawl_depth: Integer + + attr_accessor num_failed: Integer + + attr_accessor num_succeeded: Integer + + attr_accessor num_urls: Integer + + def initialize: ( + max_crawl_depth: Integer, + num_failed: Integer, + num_succeeded: Integer, + num_urls: Integer + ) -> void + + def to_hash: -> { + max_crawl_depth: Integer, + num_failed: Integer, + num_succeeded: Integer, + num_urls: Integer + } + end + + type result = + { + markdown: String, + metadata: ContextDev::Models::WebWebCrawlMdResponse::Result::Metadata + } + + class Result < ContextDev::Internal::Type::BaseModel + attr_accessor markdown: String + + attr_accessor metadata: ContextDev::Models::WebWebCrawlMdResponse::Result::Metadata + + def initialize: ( + markdown: String, + metadata: ContextDev::Models::WebWebCrawlMdResponse::Result::Metadata + ) -> void + + def to_hash: -> { + markdown: String, + metadata: ContextDev::Models::WebWebCrawlMdResponse::Result::Metadata + } + + type metadata = + { + crawl_depth: Integer, + status_code: Integer, + success: bool, + title: String, + url: String + } + + class Metadata < ContextDev::Internal::Type::BaseModel + attr_accessor crawl_depth: Integer + + attr_accessor status_code: Integer + + attr_accessor success: bool + + attr_accessor title: String + + attr_accessor url: String + + def initialize: ( + crawl_depth: Integer, + status_code: Integer, + success: bool, + title: String, + url: String + ) -> void + + def to_hash: -> { + crawl_depth: Integer, + status_code: Integer, + success: bool, + title: String, + url: String + } + end + end + end + end +end diff --git a/sig/context_dev/resources/web.rbs b/sig/context_dev/resources/web.rbs index ccddc725..eabf7dd1 100644 --- a/sig/context_dev/resources/web.rbs +++ b/sig/context_dev/resources/web.rbs @@ -9,6 +9,19 @@ module ContextDev ?request_options: ContextDev::request_opts ) -> ContextDev::Models::WebScreenshotResponse + def web_crawl_md: ( + url: String, + ?follow_subdomains: bool, + ?include_images: bool, + ?include_links: bool, + ?max_depth: Integer, + ?max_pages: Integer, + ?shorten_base64_images: bool, + ?url_regex: String, + ?use_main_content_only: bool, + ?request_options: ContextDev::request_opts + ) -> ContextDev::Models::WebWebCrawlMdResponse + def web_scrape_html: ( url: String, ?request_options: ContextDev::request_opts diff --git a/test/context_dev/resources/web_test.rb b/test/context_dev/resources/web_test.rb index 81a1103d..5c4e22ea 100644 --- a/test/context_dev/resources/web_test.rb +++ b/test/context_dev/resources/web_test.rb @@ -23,6 +23,23 @@ def test_screenshot_required_params end end + def test_web_crawl_md_required_params + skip("Mock server tests are disabled") + + response = @context_dev.web.web_crawl_md(url: "https://example.com") + + assert_pattern do + response => ContextDev::Models::WebWebCrawlMdResponse + end + + assert_pattern do + response => { + metadata: ContextDev::Models::WebWebCrawlMdResponse::Metadata, + results: ^(ContextDev::Internal::Type::ArrayOf[ContextDev::Models::WebWebCrawlMdResponse::Result]) + } + end + end + def test_web_scrape_html_required_params skip("Mock server tests are disabled") From 6a5e1e321664bb81ffd8f1d8c955e8e9b541b0b4 Mon Sep 17 00:00:00 2001 From: "stainless-app[bot]" <142633134+stainless-app[bot]@users.noreply.github.com> Date: Sat, 4 Apr 2026 21:42:50 +0000 Subject: [PATCH 4/4] release: 1.3.0 --- .release-please-manifest.json | 2 +- CHANGELOG.md | 8 ++++++++ Gemfile.lock | 2 +- README.md | 2 +- lib/context_dev/version.rb | 2 +- 5 files changed, 12 insertions(+), 4 deletions(-) diff --git a/.release-please-manifest.json b/.release-please-manifest.json index d0ab6645..2a8f4ffd 100644 --- a/.release-please-manifest.json +++ b/.release-please-manifest.json @@ -1,3 +1,3 @@ { - ".": "1.2.0" + ".": "1.3.0" } \ No newline at end of file diff --git a/CHANGELOG.md b/CHANGELOG.md index ec0f68af..84862991 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,5 +1,13 @@ # Changelog +## 1.3.0 (2026-04-04) + +Full Changelog: [v1.2.0...v1.3.0](https://github.com/context-dot-dev/context-ruby-sdk/compare/v1.2.0...v1.3.0) + +### Features + +* **api:** manual updates ([8e8fcc2](https://github.com/context-dot-dev/context-ruby-sdk/commit/8e8fcc26f2fbbb2bdcca9713fa3b9f8518303586)) + ## 1.2.0 (2026-04-03) Full Changelog: [v1.1.0...v1.2.0](https://github.com/context-dot-dev/context-ruby-sdk/compare/v1.1.0...v1.2.0) diff --git a/Gemfile.lock b/Gemfile.lock index 56504d77..3f888a83 100644 --- a/Gemfile.lock +++ b/Gemfile.lock @@ -11,7 +11,7 @@ GIT PATH remote: . specs: - context.dev (1.2.0) + context.dev (1.3.0) cgi connection_pool diff --git a/README.md b/README.md index e82cc712..582e15fa 100644 --- a/README.md +++ b/README.md @@ -26,7 +26,7 @@ To use this gem, install via Bundler by adding the following to your application <!-- x-release-please-start-version --> ```ruby -gem "context.dev", "~> 1.2.0" +gem "context.dev", "~> 1.3.0" ``` <!-- x-release-please-end --> diff --git a/lib/context_dev/version.rb b/lib/context_dev/version.rb index 103fb3e4..cd0f065e 100644 --- a/lib/context_dev/version.rb +++ b/lib/context_dev/version.rb @@ -1,5 +1,5 @@ # frozen_string_literal: true module ContextDev - VERSION = "1.2.0" + VERSION = "1.3.0" end