From ea754f3369c731dd8ba95d89ba065efa798dc27e Mon Sep 17 00:00:00 2001 From: Yahia Bakour Date: Thu, 8 Oct 2026 12:48:11 -0400 Subject: [PATCH 1/4] docs: lead README with Context.dev capabilities --- README.md | 43 ++++++++++++++++++++++++++++++++++++++++++- 1 file changed, 42 insertions(+), 1 deletion(-) diff --git a/README.md b/README.md index 08d23b4e..e279aa13 100644 --- a/README.md +++ b/README.md @@ -2,7 +2,7 @@ [![NPM version]()](https://npmjs.org/package/context.dev) ![npm bundle size](https://img.shields.io/bundlephobia/minzip/context.dev) -This library provides convenient access to the Context Dev REST API from server-side TypeScript or JavaScript. +Context.dev is a web scraping API for AI agents and LLMs. This SDK turns any URL into clean, LLM-ready markdown, crawls whole sites, searches the web, takes screenshots and extracts structured JSON against a schema you define, all with one API key. Proxies, JavaScript rendering and anti-bot handling run on Context.dev's side, so there is no headless browser to host. The REST API documentation can be found on [docs.context.dev](https://docs.context.dev/). The full API of this library can be found in [api.md](api.md). @@ -31,6 +31,47 @@ const brand = await client.brand.retrieve({ domain: 'REPLACE_ME', type: 'by_doma console.log(brand.request_id); ``` +### Extract structured JSON + +`jsonParams.schema` accepts JSON Schema; install Zod 4 (`npm install zod@^4`) to build a schema with Zod and convert it with `z.toJSONSchema`. + +```ts +import ContextDev from 'context.dev'; +import { z } from 'zod'; + +const client = new ContextDev(); +const pageSchema = z.object({ + title: z.string().nullable(), + description: z.string().nullable(), +}); + +const page = await client.web.scrape({ + url: 'https://example.com', + formats: { json: true }, + jsonParams: { schema: z.toJSONSchema(pageSchema) }, +}); + +console.log(page.json.data); +``` + +## What you can do + +| Task | Method | +| --- | --- | +| Scrape a URL to markdown, HTML, JSON or a screenshot | `client.web.scrape` | +| Crawl a site and get every page as markdown | `client.web.webCrawlMd` | +| Map every URL on a domain | `client.web.mapUrls` | +| Search the web | `client.web.search` | +| Take a screenshot of a page | `client.web.screenshot` | +| Parse PDFs and documents | `client.parse.handle` | +| Run thousands of URLs as a batch | `client.batch.submit` | +| Watch a page for changes | `client.monitors.create` | +| Look up a company's logo, colors and brand data | `client.brand.retrieve` | + +## Use it from an AI agent + +Context.dev also ships as a plugin for [Claude](https://github.com/context-dot-dev/claude-plugin), [Cursor](https://github.com/context-dot-dev/cursor-plugin) and [Gemini CLI](https://github.com/context-dot-dev/gemini-cli-context), and as tools for [LangChain](https://github.com/context-dot-dev/langchain-context) and [Haystack](https://github.com/context-dot-dev/context-haystack). + ### Request & Response types This library includes TypeScript definitions for all request params and response fields. You may import and use them like so: From dadbc3a86d3e4edbafb725a2b56191c23871d096 Mon Sep 17 00:00:00 2001 From: Yahia Bakour Date: Thu, 8 Oct 2026 12:48:33 -0400 Subject: [PATCH 2/4] docs: format capability table --- README.md | 22 +++++++++++----------- 1 file changed, 11 insertions(+), 11 deletions(-) diff --git a/README.md b/README.md index e279aa13..bdb26377 100644 --- a/README.md +++ b/README.md @@ -56,17 +56,17 @@ console.log(page.json.data); ## What you can do -| Task | Method | -| --- | --- | -| Scrape a URL to markdown, HTML, JSON or a screenshot | `client.web.scrape` | -| Crawl a site and get every page as markdown | `client.web.webCrawlMd` | -| Map every URL on a domain | `client.web.mapUrls` | -| Search the web | `client.web.search` | -| Take a screenshot of a page | `client.web.screenshot` | -| Parse PDFs and documents | `client.parse.handle` | -| Run thousands of URLs as a batch | `client.batch.submit` | -| Watch a page for changes | `client.monitors.create` | -| Look up a company's logo, colors and brand data | `client.brand.retrieve` | +| Task | Method | +| ---------------------------------------------------- | ------------------------ | +| Scrape a URL to markdown, HTML, JSON or a screenshot | `client.web.scrape` | +| Crawl a site and get every page as markdown | `client.web.webCrawlMd` | +| Map every URL on a domain | `client.web.mapUrls` | +| Search the web | `client.web.search` | +| Take a screenshot of a page | `client.web.screenshot` | +| Parse PDFs and documents | `client.parse.handle` | +| Run thousands of URLs as a batch | `client.batch.submit` | +| Watch a page for changes | `client.monitors.create` | +| Look up a company's logo, colors and brand data | `client.brand.retrieve` | ## Use it from an AI agent From 2628c22cdc48e011546ba0e0ef45d2e3f7fe8b53 Mon Sep 17 00:00:00 2001 From: Yahia Bakour Date: Thu, 8 Oct 2026 12:51:09 -0400 Subject: [PATCH 3/4] docs: remove Stainless README attribution --- README.md | 2 -- 1 file changed, 2 deletions(-) diff --git a/README.md b/README.md index bdb26377..3ada9d90 100644 --- a/README.md +++ b/README.md @@ -6,8 +6,6 @@ Context.dev is a web scraping API for AI agents and LLMs. This SDK turns any URL The REST API documentation can be found on [docs.context.dev](https://docs.context.dev/). The full API of this library can be found in [api.md](api.md). -It is generated with [Stainless](https://www.stainless.com/). - ## Installation ```sh From e9b2400177404d9ff436ab86c75135d29c046568 Mon Sep 17 00:00:00 2001 From: Yahia Bakour Date: Thu, 8 Oct 2026 13:03:44 -0400 Subject: [PATCH 4/4] docs: demonstrate scraping formats throughout README --- README.md | 102 ++++++++++++++++++++++++++++++++++++++---------------- 1 file changed, 72 insertions(+), 30 deletions(-) diff --git a/README.md b/README.md index 3ada9d90..399620e7 100644 --- a/README.md +++ b/README.md @@ -14,19 +14,25 @@ npm install context.dev ## Usage +Set `CONTEXT_DEV_API_KEY` to your API key; the client reads it automatically. + +### Scrape markdown and HTML + The full API of this library can be found in [api.md](api.md). -```js +```ts import ContextDev from 'context.dev'; -const client = new ContextDev({ - apiKey: process.env['CONTEXT_DEV_API_KEY'], // This is the default and can be omitted -}); +const client = new ContextDev(); -const brand = await client.brand.retrieve({ domain: 'REPLACE_ME', type: 'by_domain' }); +const page = await client.web.scrape({ + url: 'https://example.com', + formats: { markdown: true, html: true }, +}); -console.log(brand.request_id); +console.log(page.markdown.data); +console.log(page.html.data); ``` ### Extract structured JSON @@ -52,19 +58,54 @@ const page = await client.web.scrape({ console.log(page.json.data); ``` +### Extract relevant highlights + +Return the passages that answer a question about the page. + +```ts +import ContextDev from 'context.dev'; + +const client = new ContextDev(); + +const page = await client.web.scrape({ + url: 'https://example.com', + formats: { highlights: true }, + highlightsParams: { query: 'What is this domain used for?' }, +}); + +console.log(page.highlights.data); +``` + +### Take a screenshot + +The screenshot is returned as a base64 image data URL. + +```ts +import ContextDev from 'context.dev'; + +const client = new ContextDev(); + +const page = await client.web.scrape({ + url: 'https://example.com', + formats: { screenshot: true }, +}); + +console.log(page.screenshot.data); +``` + ## What you can do -| Task | Method | -| ---------------------------------------------------- | ------------------------ | -| Scrape a URL to markdown, HTML, JSON or a screenshot | `client.web.scrape` | -| Crawl a site and get every page as markdown | `client.web.webCrawlMd` | -| Map every URL on a domain | `client.web.mapUrls` | -| Search the web | `client.web.search` | -| Take a screenshot of a page | `client.web.screenshot` | -| Parse PDFs and documents | `client.parse.handle` | -| Run thousands of URLs as a batch | `client.batch.submit` | -| Watch a page for changes | `client.monitors.create` | -| Look up a company's logo, colors and brand data | `client.brand.retrieve` | +| Task | Method | +| ---------------------------------------------------------------- | ------------------------ | +| Scrape a URL to markdown, HTML, JSON, highlights or a screenshot | `client.web.scrape` | +| Crawl a site and get every page as markdown | `client.web.webCrawlMd` | +| Map every URL on a domain | `client.web.mapUrls` | +| Search the web | `client.web.search` | +| Take a screenshot of a page | `client.web.screenshot` | +| Parse PDFs and documents | `client.parse.handle` | +| Run thousands of URLs as a batch | `client.batch.submit` | +| Watch a page for changes | `client.monitors.create` | +| Look up a company's logo, colors and brand data | `client.brand.retrieve` | ## Use it from an AI agent @@ -82,8 +123,8 @@ const client = new ContextDev({ apiKey: process.env['CONTEXT_DEV_API_KEY'], // This is the default and can be omitted }); -const params: ContextDev.BrandRetrieveParams = { domain: 'REPLACE_ME', type: 'by_domain' }; -const brand: ContextDev.BrandRetrieveResponse = await client.brand.retrieve(params); +const params: ContextDev.WebScrapeParams = { url: 'https://example.com', formats: { markdown: true } }; +const page: ContextDev.WebScrapeResponse = await client.web.scrape(params); ``` Documentation for each method, request param, and response field are available in docstrings and will appear on hover in most modern editors. @@ -96,8 +137,8 @@ a subclass of `APIError` will be thrown: ```ts -const brand = await client.brand - .retrieve({ domain: 'REPLACE_ME', type: 'by_domain' }) +const page = await client.web + .scrape({ url: 'https://example.com', formats: { markdown: true } }) .catch(async (err) => { if (err instanceof ContextDev.APIError) { console.log(err.status); // 400 @@ -138,7 +179,7 @@ const client = new ContextDev({ }); // Or, configure per-request: -await client.brand.retrieve({ domain: 'REPLACE_ME', type: 'by_domain' }, { +await client.web.scrape({ url: 'https://example.com', formats: { markdown: true } }, { maxRetries: 5, }); ``` @@ -155,7 +196,7 @@ const client = new ContextDev({ }); // Override per-request: -await client.brand.retrieve({ domain: 'REPLACE_ME', type: 'by_domain' }, { +await client.web.scrape({ url: 'https://example.com', formats: { markdown: true } }, { timeout: 5 * 1000, }); ``` @@ -178,17 +219,17 @@ Unlike `.asResponse()` this method consumes the body, returning once it is parse ```ts const client = new ContextDev(); -const response = await client.brand - .retrieve({ domain: 'REPLACE_ME', type: 'by_domain' }) +const response = await client.web + .scrape({ url: 'https://example.com', formats: { markdown: true } }) .asResponse(); console.log(response.headers.get('X-My-Header')); console.log(response.statusText); // access the underlying Response object -const { data: brand, response: raw } = await client.brand - .retrieve({ domain: 'REPLACE_ME', type: 'by_domain' }) +const { data: page, response: raw } = await client.web + .scrape({ url: 'https://example.com', formats: { markdown: true } }) .withResponse(); console.log(raw.headers.get('X-My-Header')); -console.log(brand.request_id); +console.log(page.markdown.data); ``` ### Logging @@ -268,8 +309,9 @@ parameter. This library doesn't validate at runtime that the request matches the send will be sent as-is. ```ts -client.brand.retrieve({ - // ... +client.web.scrape({ + url: 'https://example.com', + formats: { markdown: true }, // @ts-expect-error baz is not yet public baz: 'undocumented option', });