From 9e2a32094d3ec51fea384516f1bb3593bb50eaa4 Mon Sep 17 00:00:00 2001 From: Yahia Bakour Date: Thu, 8 Oct 2026 12:48:27 -0400 Subject: [PATCH 1/3] docs: lead README with Context.dev capabilities --- README.md | 48 +++++++++++++++++++++++++++++++++++++++++++++++- 1 file changed, 47 insertions(+), 1 deletion(-) diff --git a/README.md b/README.md index 8a09db3..269161c 100644 --- a/README.md +++ b/README.md @@ -1,6 +1,6 @@ # Context Dev PHP API library -The Context Dev PHP library provides convenient access to the Context Dev REST API from any PHP 8.1.0+ application. +Context.dev is a web scraping API for AI agents and LLMs. This SDK turns any URL into clean, LLM-ready markdown, crawls whole sites, searches the web, takes screenshots and extracts structured JSON against a schema you define, all with one API key. Proxies, JavaScript rendering and anti-bot handling run on Context.dev's side, so there is no headless browser to host. It is generated with [Stainless](https://www.stainless.com/). @@ -35,6 +35,52 @@ $brand = $client->brand->retrieve(domain: 'REPLACE_ME', type: 'by_domain'); var_dump($brand->request_id); ``` +### Extract structured JSON + +```php +web->scrape( + url: 'https://example.com', + formats: ['json' => true], + jsonParams: [ + 'schema' => [ + 'type' => 'object', + 'properties' => [ + 'title' => ['type' => ['string', 'null']], + 'description' => ['type' => ['string', 'null']], + ], + 'required' => ['title', 'description'], + 'additionalProperties' => false, + ], + ], +); + +var_dump($page->json->data); +``` + +## What you can do + +| Task | Method | +| --- | --- | +| Scrape a URL to markdown, HTML, JSON or a screenshot | `$client->web->scrape` | +| Crawl a site and get every page as markdown | `$client->web->webCrawlMd` | +| Map every URL on a domain | `$client->web->mapUrls` | +| Search the web | `$client->web->search` | +| Take a screenshot of a page | `$client->web->screenshot` | +| Parse PDFs and documents | `$client->parse->handle` | +| Run thousands of URLs as a batch | `$client->batch->submit` | +| Watch a page for changes | `$client->monitors->create` | +| Look up a company's logo, colors and brand data | `$client->brand->retrieve` | + +## Use it from an AI agent + +Context.dev also ships as a plugin for [Claude](https://github.com/context-dot-dev/claude-plugin), [Cursor](https://github.com/context-dot-dev/cursor-plugin) and [Gemini CLI](https://github.com/context-dot-dev/gemini-cli-context), and as tools for [LangChain](https://github.com/context-dot-dev/langchain-context) and [Haystack](https://github.com/context-dot-dev/context-haystack). + ### Value Objects It is recommended to use the static `with` constructor `Dog::with(name: "Joey")` From d8aad0e6868885592d5fa8f6a06d499420c9046e Mon Sep 17 00:00:00 2001 From: Yahia Bakour Date: Thu, 8 Oct 2026 12:51:23 -0400 Subject: [PATCH 2/3] docs: remove Stainless README attribution --- README.md | 2 -- 1 file changed, 2 deletions(-) diff --git a/README.md b/README.md index 269161c..f092343 100644 --- a/README.md +++ b/README.md @@ -2,8 +2,6 @@ Context.dev is a web scraping API for AI agents and LLMs. This SDK turns any URL into clean, LLM-ready markdown, crawls whole sites, searches the web, takes screenshots and extracts structured JSON against a schema you define, all with one API key. Proxies, JavaScript rendering and anti-bot handling run on Context.dev's side, so there is no headless browser to host. -It is generated with [Stainless](https://www.stainless.com/). - ## Documentation The REST API documentation can be found on [docs.context.dev](https://docs.context.dev/). From d9a2e77f6365f971cca2a2a4698d7a7b8d8e2df5 Mon Sep 17 00:00:00 2001 From: Yahia Bakour Date: Thu, 8 Oct 2026 13:04:01 -0400 Subject: [PATCH 3/3] docs: demonstrate scraping formats throughout README --- README.md | 67 ++++++++++++++++++++++++++++++++++++++++++++++--------- 1 file changed, 57 insertions(+), 10 deletions(-) diff --git a/README.md b/README.md index f092343..8e43e27 100644 --- a/README.md +++ b/README.md @@ -18,6 +18,10 @@ composer require "context-dev/context-dev-php 2.24.0" ## Usage +Set `CONTEXT_DEV_API_KEY` to your API key; the client reads it automatically. + +### Scrape markdown and HTML + This library uses named parameters to specify optional arguments. Parameters with a default value must be set by name. @@ -26,11 +30,15 @@ Parameters with a default value must be set by name. use ContextDev\Client; -$client = new Client(apiKey: getenv('CONTEXT_DEV_API_KEY') ?: 'My API Key'); +$client = new Client(); -$brand = $client->brand->retrieve(domain: 'REPLACE_ME', type: 'by_domain'); +$page = $client->web->scrape( + url: 'https://example.com', + formats: ['markdown' => true, 'html' => true], +); -var_dump($brand->request_id); +echo $page->markdown->data, PHP_EOL; +echo $page->html->data, PHP_EOL; ``` ### Extract structured JSON @@ -61,11 +69,50 @@ $page = $client->web->scrape( var_dump($page->json->data); ``` +### Extract relevant highlights + +Return the passages that answer a question about the page. + +```php +web->scrape( + url: 'https://example.com', + formats: ['highlights' => true], + highlightsParams: ['query' => 'What is this domain used for?'], +); + +var_dump($page->highlights->data); +``` + +### Take a screenshot + +The screenshot is returned as a base64 image data URL. + +```php +web->scrape( + url: 'https://example.com', + formats: ['screenshot' => true], +); + +var_dump($page->screenshot->data); +``` + ## What you can do | Task | Method | | --- | --- | -| Scrape a URL to markdown, HTML, JSON or a screenshot | `$client->web->scrape` | +| Scrape a URL to markdown, HTML, JSON, highlights or a screenshot | `$client->web->scrape` | | Crawl a site and get every page as markdown | `$client->web->webCrawlMd` | | Map every URL on a domain | `$client->web->mapUrls` | | Search the web | `$client->web->search` | @@ -98,7 +145,7 @@ use ContextDev\Core\Exceptions\RateLimitException; use ContextDev\Core\Exceptions\APIStatusException; try { - $brand = $client->brand->retrieve(domain: 'REPLACE_ME', type: 'by_domain'); + $page = $client->web->scrape(url: 'https://example.com', formats: ['markdown' => true]); } catch (APIConnectionException $e) { echo "The server could not be reached", PHP_EOL; var_dump($e->getPrevious()); @@ -143,8 +190,8 @@ use ContextDev\Client; $client = new Client(requestOptions: ['maxRetries' => 0]); // Or, configure per-request: -$result = $client->brand->retrieve( - domain: 'REPLACE_ME', type: 'by_domain', requestOptions: ['maxRetries' => 5] +$result = $client->web->scrape( + url: 'https://example.com', formats: ['markdown' => true], requestOptions: ['maxRetries' => 5] ); ``` @@ -161,9 +208,9 @@ Note: the `extra*` parameters of the same name overrides the documented paramete ```php brand->retrieve( - domain: 'REPLACE_ME', - type: 'by_domain', +$page = $client->web->scrape( + url: 'https://example.com', + formats: ['markdown' => true], requestOptions: [ 'extraQueryParams' => ['my_query_parameter' => 'value'], 'extraBodyParams' => ['my_body_parameter' => 'value'],