From 80704bed25679dd7ad33d2033fdeb93681dda409 Mon Sep 17 00:00:00 2001 From: Pray Apostel Date: Wed, 26 Aug 2026 08:45:51 +0700 Subject: [PATCH 1/4] Update repository URLs for mrscraper-com --- .claude-plugin/marketplace.json | 2 +- README.md | 6 +++--- plugins/mrscraper/.claude-plugin/plugin.json | 2 +- 3 files changed, 5 insertions(+), 5 deletions(-) diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index 617aa34..11318bc 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -16,7 +16,7 @@ "name": "MrScraper" }, "homepage": "https://docs.mrscraper.com/docs/getting-started/mcp-server", - "repository": "https://github.com/pray-mrscraper/mrscraper-claude-plugin", + "repository": "https://github.com/mrscraper-com/mrscraper-claude-plugin", "license": "MIT", "category": "productivity", "tags": [ diff --git a/README.md b/README.md index 243f8ee..c4e0b39 100644 --- a/README.md +++ b/README.md @@ -8,16 +8,16 @@ hosted MrScraper MCP server for Claude. Paste this GitHub `owner/repo` value into Claude's **Add marketplace** dialog: ```text -pray-mrscraper/mrscraper-claude-plugin +mrscraper-com/mrscraper-claude-plugin ``` If the dialog specifically expects a Git repository URL, use -`https://github.com/pray-mrscraper/mrscraper-claude-plugin.git`. +`https://github.com/mrscraper-com/mrscraper-claude-plugin.git`. From Claude Code CLI, add the marketplace and install the plugin: ```bash -claude plugin marketplace add pray-mrscraper/mrscraper-claude-plugin +claude plugin marketplace add mrscraper-com/mrscraper-claude-plugin claude plugin install mrscraper@mrscraper-claude ``` diff --git a/plugins/mrscraper/.claude-plugin/plugin.json b/plugins/mrscraper/.claude-plugin/plugin.json index 6fbd26b..a2f19aa 100644 --- a/plugins/mrscraper/.claude-plugin/plugin.json +++ b/plugins/mrscraper/.claude-plugin/plugin.json @@ -9,7 +9,7 @@ "url": "https://github.com/mrscraper-com" }, "homepage": "https://docs.mrscraper.com/docs/getting-started/mcp-server", - "repository": "https://github.com/pray-mrscraper/mrscraper-claude-plugin", + "repository": "https://github.com/mrscraper-com/mrscraper-claude-plugin", "license": "MIT", "keywords": [ "mrscraper", From e2567ad5a46088cbf9be8cf7dbf0befbcb3d2e36 Mon Sep 17 00:00:00 2001 From: Pray Apostel Date: Wed, 26 Aug 2026 12:02:53 +0700 Subject: [PATCH 2/4] docs: document supported scraper controls --- .../mrscraper/skills/mrscraper-fetch/SKILL.md | 20 +++++++++-- .../skills/mrscraper-scrape/SKILL.md | 6 ++++ plugins/mrscraper/skills/mrscraper/SKILL.md | 36 +++++++++++++++++-- 3 files changed, 56 insertions(+), 6 deletions(-) diff --git a/plugins/mrscraper/skills/mrscraper-fetch/SKILL.md b/plugins/mrscraper/skills/mrscraper-fetch/SKILL.md index 3f92430..3594a7d 100644 --- a/plugins/mrscraper/skills/mrscraper-fetch/SKILL.md +++ b/plugins/mrscraper/skills/mrscraper-fetch/SKILL.md @@ -1,7 +1,7 @@ --- name: mrscraper-fetch description: | - Fetch HTML from a known public URL with MrScraper, with optional browser rendering, locale routing, selector waits, homepage navigation, resource blocking, retries, token limits, and page-load timeouts. Use when the user wants to read, summarize, cite, inspect, archive, or flexibly analyze a page. Use mrscraper-scrape for backend LLM extraction of defined fields or structured records, and mrscraper-serp when no target URL is known. + Fetch HTML from a known public URL with MrScraper, with optional browser rendering, real-device Super Mode, locale routing, selector waits, homepage navigation, resource blocking, retries, token limits, and page-load timeouts. Use when the user wants to read, summarize, cite, inspect, archive, or flexibly analyze a page. Use mrscraper-scrape for backend LLM extraction of defined fields or structured records, and mrscraper-serp when no target URL is known. --- # Fetch Page Content with MrScraper MCP @@ -75,6 +75,18 @@ Use browser rendering when the page depends on JavaScript: "browser_rendering": true } +If ordinary browser rendering still cannot load the page, route it through a +real device with Super Mode: + + { + "url": "https://example.com/products", + "browser_rendering": true, + "super_mode": true + } + +Use super_mode only after ordinary browser rendering fails. It requires +browser_rendering=true. + Wait for delayed content with a CSS selector: { @@ -111,6 +123,7 @@ Bound resource use for a browser-rendered page: | --- | --- | --- | --- | | url | required | Query url | Absolute HTTP or HTTPS target URL. | | browser_rendering | false | Query browserRendering | Execute page JavaScript. | +| super_mode | false | Query super | Route browser rendering through a real device; requires browser_rendering=true. | | geo_code | omitted | Query geoCode | Route through an ISO 3166-1 alpha-2 country. | | wait_for_selector | omitted | Query waitForSelector | Wait for a CSS selector; requires browser_rendering=true. | | home_page | false | Query homePage | Visit the site root before the target page. | @@ -129,9 +142,10 @@ incomplete or missing dynamic content: 1. Inspect the initial response. 2. Retry once with browser_rendering=true when JavaScript is relevant. -3. Add wait_for_selector, geo_code, or home_page only when the target requires +3. Add super_mode only when ordinary browser rendering still fails. +4. Add wait_for_selector, geo_code, or home_page only when the target requires that behavior. -4. Inspect the revised result before considering another retry. +5. Inspect the revised result before considering another retry. Do not repeat identical calls. Treat wait_for_selector as a CSS selector, not a duration. Browser rendering loads a page; it does not click controls, submit diff --git a/plugins/mrscraper/skills/mrscraper-scrape/SKILL.md b/plugins/mrscraper/skills/mrscraper-scrape/SKILL.md index a81c6ae..ae3b33f 100644 --- a/plugins/mrscraper/skills/mrscraper-scrape/SKILL.md +++ b/plugins/mrscraper/skills/mrscraper-scrape/SKILL.md @@ -35,6 +35,10 @@ The default agent is general. For map, omit prompt, schema_prompt, and proxy_country. For general and listing, omit map-only controls. max_pages is accepted by listing and map but not general. +agent selects the extraction workflow. mode independently selects the backend +execution tier: Cheap or Super. Omit mode to preserve the backend default, and +select Super only when the extraction requires the stronger mode. + ## Step 2 — Define the Extraction Write a prompt that names the fields or records the user needs and preserves @@ -45,6 +49,7 @@ Detail-page example: { "url": "https://example.com/product", "agent": "general", + "mode": "Super", "prompt": "Extract name, price, availability, description, and image URLs. Preserve source values and omit unavailable fields." } @@ -97,6 +102,7 @@ Validate locally when strict compliance is required. | --- | --- | --- | --- | | url | required | Body url | Absolute HTTP or HTTPS starting URL. | | agent | general | Body agent | Select general, listing, or map. | +| mode | service default | Body mode | Select Cheap or Super execution without changing the agent. | | prompt | required for general/listing | Body message | Describe the fields or records to extract. | | proxy_country | omitted | Body proxyCountry | Route general/listing through a country. | | max_pages | service default | Body maxPages | Bound listing or map pages. | diff --git a/plugins/mrscraper/skills/mrscraper/SKILL.md b/plugins/mrscraper/skills/mrscraper/SKILL.md index a04db5b..ad9452b 100644 --- a/plugins/mrscraper/skills/mrscraper/SKILL.md +++ b/plugins/mrscraper/skills/mrscraper/SKILL.md @@ -153,8 +153,22 @@ Single AI example: } Single AI optionally accepts max_depth, max_pages, limit, include_patterns, -and exclude_patterns. When omitted, the tool uses 2, 50, 1000, an empty include -expression, and an empty exclude expression respectively. +exclude_patterns, proxy_country, max_retry, and timeout. Omit any control that +the user did not request so the saved scraper or backend default remains in +effect; the MCP server does not inject replacement defaults. + +Explicit-control example: + + { + "target": "https://example.com/products", + "type": "ai", + "scraper_id": "SCRAPER_UUID", + "proxy_country": "ID", + "max_retry": 4, + "timeout": 120 + } + +These controls are rejected by manual and bulk reruns. Bulk example: @@ -187,6 +201,15 @@ Use results when the exact result UUID is unknown: "page": 1 } +Apply exact backend filters when the user knows identifying result fields: + + { + "scraper_id": "SCRAPER_UUID", + "status": "Finished", + "type": "Rerun-AI", + "url": "https://example.com/product" + } + | Input | Default | Use | | --- | --- | --- | | sort_field | updatedAt | Stored-result sort key. | @@ -197,14 +220,21 @@ Use results when the exact result UUID is unknown: | date_range_column | omitted | Date column used by start_at and end_at. | | start_at | omitted | Inclusive ISO 8601 range start. | | end_at | omitted | Inclusive ISO 8601 range end. | +| scraper_id | omitted | Exact saved scraper UUID filter. | +| status | omitted | Exact Draft, Finished, Running, Failed, or Cancelled filter. | +| type | omitted | Exact result type filter, such as AI or Rerun-AI. | +| url | omitted | Exact stored target URL filter. | Use result when the UUID is known: { - "result_id": "RESULT_UUID" + "result_id": "RESULT_UUID", + "include_html": false } A result_id may also be the bulkResultId returned by an asynchronous rerun. +include_html defaults to true. Set it to false for a smaller response while +polling status or when extracted data is sufficient. ## Step 7 — Review Account Usage From 69177371fe94dfa69edc1ee25110e716a2dac23b Mon Sep 17 00:00:00 2001 From: Pray Apostel Date: Wed, 26 Aug 2026 12:05:24 +0700 Subject: [PATCH 3/4] Prepare Claude plugin for community publishing --- .claude-plugin/marketplace.json | 10 +- .github/workflows/validate.yml | 31 +++++ CHANGELOG.md | 24 ++++ PUBLISHING.md | 72 ++++++++++ README.md | 130 ++++++++++++++---- SECURITY.md | 12 ++ plugins/mrscraper/.claude-plugin/plugin.json | 7 +- plugins/mrscraper/README.md | 20 +++ .../mrscraper/skills/mrscraper-fetch/SKILL.md | 58 ++++++-- .../skills/mrscraper-scrape/SKILL.md | 72 ++++++---- .../mrscraper/skills/mrscraper-serp/SKILL.md | 11 +- plugins/mrscraper/skills/mrscraper/SKILL.md | 67 +++++---- 12 files changed, 412 insertions(+), 102 deletions(-) create mode 100644 .github/workflows/validate.yml create mode 100644 CHANGELOG.md create mode 100644 PUBLISHING.md create mode 100644 SECURITY.md create mode 100644 plugins/mrscraper/README.md diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index 11318bc..eca7d36 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -3,17 +3,19 @@ "name": "mrscraper-claude", "owner": { "name": "MrScraper", + "email": "support@mrscraper.com", "url": "https://github.com/mrscraper-com" }, - "description": "MrScraper MCP skills and tools for Claude.", + "description": "Fetch and preserve page content, extract structured web data, discover sources, and work with saved MrScraper runs from Claude.", "plugins": [ { "name": "mrscraper", "source": "./plugins/mrscraper", - "description": "Fetch page content, extract structured web data, discover sources through Google, and work with saved MrScraper runs.", - "version": "0.1.1", + "description": "Connect Claude to MrScraper for raw page fetching, managed extraction, Google discovery, and saved scraper workflows.", "author": { - "name": "MrScraper" + "name": "MrScraper", + "email": "support@mrscraper.com", + "url": "https://mrscraper.com" }, "homepage": "https://docs.mrscraper.com/docs/getting-started/mcp-server", "repository": "https://github.com/mrscraper-com/mrscraper-claude-plugin", diff --git a/.github/workflows/validate.yml b/.github/workflows/validate.yml new file mode 100644 index 0000000..54364fa --- /dev/null +++ b/.github/workflows/validate.yml @@ -0,0 +1,31 @@ +name: Validate Claude plugin + +on: + pull_request: + push: + branches: + - main + +permissions: + contents: read + +jobs: + validate: + runs-on: ubuntu-latest + steps: + - name: Check out repository + uses: actions/checkout@v4 + + - name: Set up Node.js + uses: actions/setup-node@v4 + with: + node-version: 22 + + - name: Install Claude Code + run: npm install --global @anthropic-ai/claude-code@2.1.246 + + - name: Validate marketplace + run: claude plugin validate . --strict + + - name: Validate plugin + run: claude plugin validate plugins/mrscraper --strict diff --git a/CHANGELOG.md b/CHANGELOG.md new file mode 100644 index 0000000..07f30a5 --- /dev/null +++ b/CHANGELOG.md @@ -0,0 +1,24 @@ +# Changelog + +All notable changes to the MrScraper Claude plugin are documented here. The +project follows [Semantic Versioning](https://semver.org/). + +## 0.1.2 - 2026-08-26 + +- Strengthened fetch-first routing, including reusable local extraction for + large same-layout page sets. +- Added guidance for fetch Super Mode and managed extraction execution modes. +- Added saved-rerun controls, exact result filters, and compact result polling. +- Replaced placeholder targets with public Scrape This Site examples. +- Added publication, data-handling, support, and security documentation. +- Moved version ownership exclusively to the plugin manifest. + +## 0.1.1 - 2026-08-24 + +- Moved the hosted MrScraper connection to Claude's native `.mcp.json` plugin + configuration. + +## 0.1.0 - 2026-08-24 + +- Added the initial MrScraper Claude marketplace, MCP connection, and skill + pack. diff --git a/PUBLISHING.md b/PUBLISHING.md new file mode 100644 index 0000000..5b0fff3 --- /dev/null +++ b/PUBLISHING.md @@ -0,0 +1,72 @@ +# Publishing MrScraper for Claude + +Use this checklist for a public MrScraper Claude plugin release. + +## Release gate + +- [ ] Merge only reviewed changes into `main`. +- [ ] Keep the release version in + `plugins/mrscraper/.claude-plugin/plugin.json`; do not duplicate it in + the marketplace entry. +- [ ] Record user-visible changes in `CHANGELOG.md`. +- [ ] Confirm that the repository contains no credentials or private target + data. +- [ ] Run both strict validations from a clean checkout: + + ```bash + claude plugin validate . --strict + claude plugin validate plugins/mrscraper --strict + ``` + +- [ ] Confirm the GitHub validation workflow passes. +- [ ] Install the plugin from the public repository in a clean Claude profile, + complete OAuth, and confirm all seven MCP tools load. +- [ ] Run the smoke-test prompt below. +- [ ] Create the matching Git tag and GitHub release for an explicit semantic + version. + +## Submission details + +Keep these values ready for Anthropic's community marketplace form: + +| Field | Value | +| --- | --- | +| Plugin name | `mrscraper` | +| Display name | `MrScraper` | +| Repository | `https://github.com/mrscraper-com/mrscraper-claude-plugin` | +| Plugin directory | `plugins/mrscraper` | +| Homepage | `https://docs.mrscraper.com/docs/getting-started/mcp-server` | +| Support | `support@mrscraper.com` | +| Privacy policy | `https://mrscraper.com/privacy-policy` | +| Acceptable use policy | `https://mrscraper.com/acceptable-use-policy` | +| MCP endpoint | `https://mcp.mrscraper.com/mcp` | +| Authentication | OAuth 2.1 browser authorization | + +Suggested summary: + +> Connect Claude to MrScraper for raw page fetching, managed structured +> extraction, Google discovery, and reusable saved scraper workflows. The +> included skills preserve source responses with a fetch-first workflow before +> adding backend extraction when it is explicitly requested or clearly useful. + +The hosted service receives target URLs and tool inputs, including extraction +instructions. Results are returned to Claude for the user's requested task. The +OAuth connection can request `scrape:read`, `scrape:write`, and `account:read` +access. The plugin contains no static credentials. + +Smoke-test prompt: + +```text +Fetch https://www.scrapethissite.com/pages/simple/ and summarize the page. +``` + +## Submit to Anthropic + +After the release commit is on the public `main` branch, use Anthropic's +[Console submission form](https://platform.claude.com/plugins/submit). A Team +or Enterprise organization owner or directory manager can instead use the +[Claude organization form](https://claude.ai/admin-settings/directory/submissions/plugins/new). + +Anthropic reviews third-party plugins for the `claude-community` marketplace. +Approved entries are pinned to a repository commit, and the public catalog may +take until its next nightly sync to show the plugin. diff --git a/README.md b/README.md index c4e0b39..bcee9d4 100644 --- a/README.md +++ b/README.md @@ -1,29 +1,50 @@ # MrScraper for Claude -This marketplace packages MrScraper's MCP-oriented skills together with the -hosted MrScraper MCP server for Claude. +MrScraper connects Claude to a hosted MCP server for fetching public web pages, +managed structured extraction, Google discovery, saved scraper reruns, stored +results, and account usage. The plugin also includes four focused skills that +help Claude choose the right workflow and preserve raw page content. -## Add the marketplace +## What this plugin adds -Paste this GitHub `owner/repo` value into Claude's **Add marketplace** dialog: +- The hosted Streamable HTTP endpoint at `https://mcp.mrscraper.com/mcp`. +- OAuth 2.1 authentication through Claude; no credential is stored in this + repository. +- A fetch-first workflow that keeps raw page responses available for analysis, + verification, and follow-up transformations. +- Managed extraction, site mapping, Google SERP discovery, saved reruns, and + stored-result tools when those capabilities are useful. -```text -mrscraper-com/mrscraper-claude-plugin -``` +## Requirements + +- A current version of [Claude Code](https://code.claude.com/docs/en/setup). +- A [MrScraper](https://app.mrscraper.com) account. +- Permission to access and process the target content. -If the dialog specifically expects a Git repository URL, use -`https://github.com/mrscraper-com/mrscraper-claude-plugin.git`. +## Install -From Claude Code CLI, add the marketplace and install the plugin: +Add this repository as a marketplace and install the plugin: ```bash claude plugin marketplace add mrscraper-com/mrscraper-claude-plugin claude plugin install mrscraper@mrscraper-claude ``` -The plugin connects to `https://mcp.mrscraper.com/mcp`. When prompted, finish -the OAuth authorization flow in your browser. Do not paste OAuth tokens or API -keys into chat. +In Claude's **Add marketplace** dialog, paste: + +```text +mrscraper-com/mrscraper-claude-plugin +``` + +If the dialog expects a full Git repository URL, use: + +```text +https://github.com/mrscraper-com/mrscraper-claude-plugin.git +``` + +Restart Claude or reload plugins after installation. Open `/mcp` if Claude asks +you to authenticate, then complete the MrScraper OAuth flow in your browser. +Never paste OAuth tokens or API keys into chat. ## Try it @@ -31,22 +52,83 @@ keys into chat. Fetch https://www.scrapethissite.com/pages/simple/ and summarize the page. ``` -The included skills encourage `fetch` as the normal first step when a URL is -known because it preserves the page response for flexible agent-led analysis. -Use `scrape` when you specifically need backend structured extraction, -pagination, schema-shaped output, or a reusable saved scraper. +The connection should expose these MCP tools: + +| Tool | Purpose | +| --- | --- | +| `fetch` | Retrieve and preserve a known page's raw response. | +| `scrape` | Run managed structured extraction or bounded site mapping. | +| `serp` | Discover public pages through Google. | +| `rerun` | Reuse a saved AI or manual scraper configuration. | +| `results` | Browse and filter stored result records. | +| `result` | Retrieve one stored result or poll an asynchronous run. | +| `status` | Inspect subscription usage and request outcomes. | + +## Fetch-first routing + +When a public URL is already known, the skills direct Claude to fetch it first +and treat the raw response as the source of truth. Claude can read, summarize, +compare, or derive structured output locally without losing details to an +early extraction prompt. -## Contents +This preference can still be faster for large same-layout sets. For roughly +100 known pages, Claude can safely fetch pages concurrently, retain every raw +response, and apply one reusable local extractor instead of requesting 100 +separate backend-LLM extractions. Use `scrape` when managed extraction is +explicitly requested or has a clear benefit after the page structure and +desired schema are understood. -- `plugins/mrscraper/.claude-plugin/plugin.json`: Claude plugin manifest -- `plugins/mrscraper/skills/`: MCP-oriented MrScraper skill pack -- `.claude-plugin/marketplace.json`: repository marketplace catalog +## Data and permissions -## Development validation +The plugin sends MCP tool inputs—including target URLs and extraction +instructions—to MrScraper's hosted service. Page responses and tool results are +then available to Claude for the requested task. Use it only with public or +otherwise authorized content, and follow the target site's requirements. + +Review MrScraper's [MCP documentation](https://docs.mrscraper.com/docs/getting-started/mcp-server), +[Privacy Policy](https://mrscraper.com/privacy-policy), and +[Acceptable Use Policy](https://mrscraper.com/acceptable-use-policy) before use. + +## Update or remove + +```bash +claude plugin marketplace update mrscraper-claude +claude plugin update mrscraper@mrscraper-claude +claude plugin uninstall mrscraper@mrscraper-claude +``` + +## Support and security + +- Product help: [MrScraper MCP documentation](https://docs.mrscraper.com/docs/getting-started/mcp-server) +- Bugs and feature requests: [GitHub Issues](https://github.com/mrscraper-com/mrscraper-claude-plugin/issues) +- Account help: [support@mrscraper.com](mailto:support@mrscraper.com) +- Security reports: see [SECURITY.md](SECURITY.md) + +## Development + +The repository is both an installable Claude marketplace and the source of the +`mrscraper` plugin: + +```text +.claude-plugin/marketplace.json +plugins/mrscraper/ +├── .claude-plugin/plugin.json +├── .mcp.json +└── skills/ +``` + +Validate both manifests and all plugin components before a release: ```bash claude plugin validate . --strict +claude plugin validate plugins/mrscraper --strict ``` -This repository is for marketplace testing before the equivalent packaging is -integrated into the official MrScraper repositories. +The plugin uses semantic versions from its plugin manifest. Bump that version +for every release and record user-visible changes in [CHANGELOG.md](CHANGELOG.md). +Use [PUBLISHING.md](PUBLISHING.md) for the release and Anthropic submission +checklist. + +## License + +[MIT](LICENSE) diff --git a/SECURITY.md b/SECURITY.md new file mode 100644 index 0000000..5c5dc48 --- /dev/null +++ b/SECURITY.md @@ -0,0 +1,12 @@ +# Security Policy + +Please report suspected vulnerabilities privately to +[support@mrscraper.com](mailto:support@mrscraper.com). Include the affected +component, reproduction steps, and potential impact. Do not open a public issue +for an unpatched vulnerability or include credentials, tokens, customer data, +or private target content in a report. + +The plugin repository contains no MrScraper credentials. Authentication is +handled by Claude's OAuth 2.1 MCP connection. If a credential may have been +exposed, revoke it through the relevant account or client immediately and then +contact support. diff --git a/plugins/mrscraper/.claude-plugin/plugin.json b/plugins/mrscraper/.claude-plugin/plugin.json index a2f19aa..b20e4b5 100644 --- a/plugins/mrscraper/.claude-plugin/plugin.json +++ b/plugins/mrscraper/.claude-plugin/plugin.json @@ -2,11 +2,12 @@ "$schema": "https://json.schemastore.org/claude-code-plugin-manifest.json", "name": "mrscraper", "displayName": "MrScraper", - "version": "0.1.1", - "description": "MrScraper MCP workflows for web fetching, structured extraction, Google discovery, and saved runs.", + "version": "0.1.2", + "description": "Connect Claude to MrScraper for raw page fetching, managed extraction, Google discovery, and saved scraper workflows.", "author": { "name": "MrScraper", - "url": "https://github.com/mrscraper-com" + "email": "support@mrscraper.com", + "url": "https://mrscraper.com" }, "homepage": "https://docs.mrscraper.com/docs/getting-started/mcp-server", "repository": "https://github.com/mrscraper-com/mrscraper-claude-plugin", diff --git a/plugins/mrscraper/README.md b/plugins/mrscraper/README.md new file mode 100644 index 0000000..cc61fff --- /dev/null +++ b/plugins/mrscraper/README.md @@ -0,0 +1,20 @@ +# MrScraper + +This Claude plugin connects to the hosted MrScraper MCP server and adds focused +skills for raw page fetching, managed structured extraction, Google discovery, +saved scraper reruns, stored results, and account usage. + +When a public URL is known, fetch is the normal first content-acquisition step. +It preserves the page response so Claude can inspect every available detail, +answer follow-up questions, and apply reusable local extraction logic. Even for +roughly 100 same-layout pages, concurrent fetches followed by one local batch +extractor can be faster and more complete than repeating backend-LLM extraction +for every page. Use scrape when managed extraction is explicitly requested or +still offers a clear benefit after the source and output schema are understood. + +The plugin connects to `https://mcp.mrscraper.com/mcp` and authenticates through +OAuth 2.1. A MrScraper account is required. Do not paste OAuth tokens or API +keys into Claude. + +See the repository [README](../../README.md) for installation, usage, data +handling, update, and support instructions. diff --git a/plugins/mrscraper/skills/mrscraper-fetch/SKILL.md b/plugins/mrscraper/skills/mrscraper-fetch/SKILL.md index 3594a7d..c24af11 100644 --- a/plugins/mrscraper/skills/mrscraper-fetch/SKILL.md +++ b/plugins/mrscraper/skills/mrscraper-fetch/SKILL.md @@ -1,13 +1,15 @@ --- name: mrscraper-fetch description: | - Fetch HTML from a known public URL with MrScraper, with optional browser rendering, real-device Super Mode, locale routing, selector waits, homepage navigation, resource blocking, retries, token limits, and page-load timeouts. Use when the user wants to read, summarize, cite, inspect, archive, or flexibly analyze a page. Use mrscraper-scrape for backend LLM extraction of defined fields or structured records, and mrscraper-serp when no target URL is known. + Retrieve raw HTML from a known public URL with MrScraper. Use for reading, inspecting, summarizing, archiving, or analyzing a page, and as the default content-acquisition step before agent-led or local extraction. Supports browser rendering, real-device Super Mode, and page-load controls; use mrscraper-serp when no target URL is known. --- # Fetch Page Content with MrScraper MCP -Use the fetch tool supplied by the MrScraper MCP server when the user already -has a URL and needs that page's response. Use +Use the fetch tool supplied by the MrScraper MCP server for the first exploration +whenever the user already has a public URL. Keep using the raw response for +summarization, comparison, transformation, and structured output instead of +handing the page to another LLM by default. Use [mrscraper](../mrscraper/SKILL.md) for connection and authentication checks, saved runs, account status, or broader routing. @@ -20,7 +22,9 @@ geo-specific country, selector waits allow delayed content to appear, and homepage navigation establishes a normal navigation path before loading the target. -Page-loading controls can help render a site that does not work with a basic request. +Use fetch only for content the user is authorized to access and in accordance +with the site's requirements. Page-loading controls can help render a site that +does not work with a basic request. Start with the URL alone and add only the controls the target needs. Plan-token usage is based on runtime and bandwidth: one token per 30 seconds and one token @@ -35,14 +39,34 @@ maintained calculation. Confirm the target URL and what the user wants: +- Preserve a raw source before extraction, transformation, or comparison. - Read, summarize, cite, or inspect the page. - Check whether specific text appears. - Archive the response. +- Produce fields, JSON, tables, or other structured output with local logic. +- Verify or supplement a managed scrape or saved result. - Load JavaScript-rendered or geo-sensitive content. -Use [mrscraper-scrape](../mrscraper-scrape/SKILL.md) when the requested outcome -needs backend extraction of a structured record or set of fields. Use -[mrscraper-serp](../mrscraper-serp/SKILL.md) when discovery must happen first. +Do not call scrape merely because the requested output is structured. Fetch the +page, understand its layout, and transform the saved raw content locally. Use +[mrscraper-scrape](../mrscraper-scrape/SKILL.md) only when the user explicitly +requests managed extraction, or after fetch-led exploration has produced a +stable output schema and managed extraction still offers a concrete benefit. +Use [mrscraper-serp](../mrscraper-serp/SKILL.md) when discovery must happen +first. + +### Prefer reusable local extraction + +When many pages share a layout: + +1. Fetch representative pages and inspect their raw content. +2. Define one local extraction schema and implementation. +3. Fetch the remaining pages, in parallel when safe and proportional. +4. Apply the same local extractor to every saved response. + +For roughly 100 same-layout pages, concurrent fetches followed by one local +batch extraction are often faster than 100 separate backend-LLM extractions. +This also preserves every raw input for later recovery or schema changes. ## Step 2 — Run the Fetch @@ -50,7 +74,7 @@ The client may namespace the tool name; select fetch from the MrScraper MCP provider. Start with one call containing only the required URL: { - "url": "https://example.com" + "url": "https://www.scrapethissite.com/pages/simple/" } The result is available in MCP structuredContent and as formatted JSON text: @@ -71,7 +95,7 @@ data. Non-JSON page bodies are preserved exactly. Use browser rendering when the page depends on JavaScript: { - "url": "https://example.com/products", + "url": "https://www.scrapethissite.com/pages/ajax-javascript/#2015", "browser_rendering": true } @@ -79,7 +103,7 @@ If ordinary browser rendering still cannot load the page, route it through a real device with Super Mode: { - "url": "https://example.com/products", + "url": "https://www.scrapethissite.com/pages/ajax-javascript/#2015", "browser_rendering": true, "super_mode": true } @@ -90,15 +114,15 @@ browser_rendering=true. Wait for delayed content with a CSS selector: { - "url": "https://example.com/products", + "url": "https://www.scrapethissite.com/pages/ajax-javascript/#2015", "browser_rendering": true, - "wait_for_selector": ".product-card" + "wait_for_selector": ".film" } Use geographic routing or homepage navigation when the target requires it: { - "url": "https://example.com/product", + "url": "https://www.scrapethissite.com/pages/simple/", "browser_rendering": true, "geo_code": "ID", "home_page": true @@ -109,7 +133,7 @@ Use geographic routing for geo-specific content. Bound resource use for a browser-rendered page: { - "url": "https://example.com", + "url": "https://www.scrapethissite.com/pages/ajax-javascript/#2015", "browser_rendering": true, "block_resources": true, "max_retries": 3, @@ -157,3 +181,9 @@ Answer the user's request from data. Keep the full envelope when headers or diagnostics matter. When the user requests an archive, save the successful response or its data value with the environment's local file-writing capability; fetch itself has no output-path input. + +Keep fetched content as the source of truth for later steps. Build summaries, +tables, JSON transformations, and extraction scripts from that raw content. If +fetch fails and another MrScraper workflow can still complete the task, disclose +that the raw response was not preserved and do not present the narrower result +as exhaustive source content. diff --git a/plugins/mrscraper/skills/mrscraper-scrape/SKILL.md b/plugins/mrscraper/skills/mrscraper-scrape/SKILL.md index ae3b33f..1fff002 100644 --- a/plugins/mrscraper/skills/mrscraper-scrape/SKILL.md +++ b/plugins/mrscraper/skills/mrscraper-scrape/SKILL.md @@ -1,27 +1,36 @@ --- name: mrscraper-scrape description: | - Extract structured data from an authorized public URL with MrScraper using the general, listing, or map agent. The general and listing modes use an LLM to read page HTML and produce structured output, so prefer mrscraper-fetch when the current agent can work directly from the HTML more quickly or flexibly. Use scrape for defined fields, repeated records, paginated listings, schema-shaped JSON, reusable extraction configurations, or bounded site URL discovery. Use mrscraper-serp when no target URL is known. + Run MrScraper's general, listing, or map agents for managed structured extraction or bounded URL discovery within a known site. Use when managed output is explicitly requested or justified after source-page inspection; use mrscraper-fetch for initial acquisition of known pages and mrscraper-serp when no starting URL is known. --- # Extract Structured Data with MrScraper MCP -Use the scrape tool supplied by the MrScraper MCP server when the user wants -defined fields, records, or site URLs from a known page or website. Use +Do not use scrape for the first exploration of a known page. Start with +[mrscraper-fetch](../mrscraper-fetch/SKILL.md), inspect the complete raw +response, and prefer local analysis or a reusable extractor. Use scrape when +the user explicitly wants MrScraper-managed extraction, or after the agent +understands the website and has defined a stable output schema. Use [mrscraper](../mrscraper/SKILL.md) for connection troubleshooting, saved runs, account status, or broader routing. -For general and listing, MrScraper retrieves the page HTML and asks an LLM to -interpret it according to prompt. This adds model-processing time and can -narrow the result to what the prompt requested. Prefer -[mrscraper-fetch](../mrscraper-fetch/SKILL.md) when the current agent can read -the HTML and perform the requested summarization, filtering, transformation, or -ad hoc analysis itself. Fetch is usually faster, preserves the full source for -follow-up questions, and avoids an extra extraction-model pass. +For general and listing, MrScraper retrieves page HTML and asks a backend LLM to +interpret it according to prompt. This adds model-processing time, can narrow +the result to the requested fields, and repeats model work across many pages. -Choose scrape when backend structured extraction is valuable: the user needs -defined or repeated records, listing pagination, schema guidance, map -discovery, or a saved scraper that can be rerun. +Before using general or listing, confirm all of the following: + +1. The target or representative pages have already been fetched and inspected. +2. The site structure and required output fields are understood. +3. A stable output schema has been defined. +4. Managed extraction still offers a concrete benefit over local code, or the + user explicitly requested it. + +If these conditions are not met, return to fetch. Requiring JSON, a table, or +named fields is not by itself a reason to call scrape. For large same-layout +sets—even roughly 100 pages—safe concurrent fetches plus one reusable local +extractor are often faster and preserve every raw response. The map agent is +separate URL discovery functionality and does not require an extraction schema. ## Step 1 — Choose an Agent @@ -41,36 +50,37 @@ select Super only when the extraction requires the stronger mode. ## Step 2 — Define the Extraction -Write a prompt that names the fields or records the user needs and preserves -source values. Do not ask the model to infer unavailable values. +The examples below assume the decision gate above has been satisfied. Write a +prompt from the already-understood page structure and output schema, preserve +source values, and do not ask the model to infer unavailable values. Detail-page example: { - "url": "https://example.com/product", + "url": "https://www.scrapethissite.com/pages/simple/", "agent": "general", "mode": "Super", - "prompt": "Extract name, price, availability, description, and image URLs. Preserve source values and omit unavailable fields." + "prompt": "Extract each country's name, capital, population, and area. Preserve source values and omit unavailable fields." } Repeated-listing example: { - "url": "https://example.com/products", + "url": "https://www.scrapethissite.com/pages/forms/?page_num=1", "agent": "listing", - "prompt": "Extract every product's name, price, availability, and URL.", + "prompt": "Extract each hockey team's name, year, wins, losses, and win percentage.", "max_pages": 5 } Site-map example: { - "url": "https://example.com", + "url": "https://www.scrapethissite.com/", "agent": "map", "max_depth": 2, "max_pages": 50, "limit": 1000, - "include_patterns": "/products/" + "include_patterns": "/pages/" } ## Step 3 — Add Shape Guidance When Useful @@ -80,14 +90,24 @@ local JSON Schema file, read and parse it with local file tools, confirm its root is an object, and pass that object: { - "url": "https://example.com/product", + "url": "https://www.scrapethissite.com/pages/simple/", "agent": "general", - "prompt": "Extract the product details.", + "prompt": "Extract every country's name, capital, population, and area.", "schema_prompt": { "type": "object", "properties": { - "name": { "type": "string" }, - "price": { "type": "string" } + "countries": { + "type": "array", + "items": { + "type": "object", + "properties": { + "name": { "type": "string" }, + "capital": { "type": "string" }, + "population": { "type": "string" }, + "area": { "type": "string" } + } + } + } } } } @@ -159,7 +179,7 @@ Use the MrScraper rerun tool to apply the saved configuration to the same or another URL: { - "target": "https://example.com/another-product", + "target": "https://www.scrapethissite.com/pages/forms/?page_num=2", "type": "ai", "scraper_id": "SCRAPER_UUID" } diff --git a/plugins/mrscraper/skills/mrscraper-serp/SKILL.md b/plugins/mrscraper/skills/mrscraper-serp/SKILL.md index 94741eb..ed016f8 100644 --- a/plugins/mrscraper/skills/mrscraper-serp/SKILL.md +++ b/plugins/mrscraper/skills/mrscraper-serp/SKILL.md @@ -1,7 +1,7 @@ --- name: mrscraper-serp description: | - Discover public pages through Google with MrScraper using a search query or Google search URL. Use when the user starts with a topic, product, company, or question but has no target URL; asks to search Google or inspect a result page; or needs relevant URLs before page reading or structured extraction. Supports country, language, pagination, JSON or HTML output, JavaScript rendering, and request timeouts. Use mrscraper-fetch for flexible agent-led work from page HTML or mrscraper-scrape for defined structured extraction once a target URL is known. + Discover public pages through Google with MrScraper using a search query or Google search URL. Use when no target URL is known or the user asks to inspect Google results. After discovery, preserve selected pages with mrscraper-fetch before adding managed extraction through mrscraper-scrape. --- # Discover Pages with MrScraper MCP SERP @@ -98,10 +98,11 @@ envelope with local file-writing capability; serp has no output-path input. Select only URLs relevant to the user's goal, then: -- Load [mrscraper-fetch](../mrscraper-fetch/SKILL.md) to read, summarize, cite, - inspect, archive, or flexibly analyze selected pages. -- Load [mrscraper-scrape](../mrscraper-scrape/SKILL.md) to extract requested - fields or repeated records. +- Load [mrscraper-fetch](../mrscraper-fetch/SKILL.md) to preserve and inspect + each selected page before deriving answers, comparisons, or structured output. +- Add [mrscraper-scrape](../mrscraper-scrape/SKILL.md) only when the user + explicitly requests managed extraction or it still offers a clear benefit + after the page structure and output schema are understood. Run independent follow-up URLs in parallel when the environment supports safe parallel execution. Keep the number of pages proportional to the requested diff --git a/plugins/mrscraper/skills/mrscraper/SKILL.md b/plugins/mrscraper/skills/mrscraper/SKILL.md index ad9452b..793d9ff 100644 --- a/plugins/mrscraper/skills/mrscraper/SKILL.md +++ b/plugins/mrscraper/skills/mrscraper/SKILL.md @@ -1,7 +1,7 @@ --- name: mrscraper description: | - Connect, authenticate, route, and troubleshoot MrScraper MCP, and use its saved-scraper, stored-result, and account tools. Use when an agent needs to set up or verify the MrScraper MCP connection, choose the correct web-data tool, rerun an AI or manual scraper, inspect stored results, check subscription usage, or handle work spanning multiple MrScraper capabilities. Route known-URL page reading and flexible agent-led analysis to mrscraper-fetch, defined structured extraction to mrscraper-scrape, and query-first Google discovery to mrscraper-serp. + Connect, authenticate, route, and troubleshoot MrScraper MCP, and use saved scrapers, stored results, and account tools. For known public URLs, acquire and preserve raw content with mrscraper-fetch first; use mrscraper-scrape only for explicitly requested or clearly beneficial managed extraction, and mrscraper-serp for query-first discovery. --- # MrScraper MCP @@ -51,7 +51,7 @@ API-key or bearer-token fallback from these skills. Verify end-to-end access with one small fetch: { - "url": "https://example.com" + "url": "https://www.scrapethissite.com/pages/simple/" } A successful page response confirms that the connection can authenticate and @@ -62,32 +62,47 @@ setup is complete. | User outcome | MCP tool or skill | | --- | --- | -| Read, summarize, cite, inspect, archive, or flexibly analyze a known public URL | [mrscraper-fetch](../mrscraper-fetch/SKILL.md), using fetch | -| Ask MrScraper's extraction LLM for defined fields, listings, records, tables, or site URLs | [mrscraper-scrape](../mrscraper-scrape/SKILL.md), using scrape | -| Discover relevant pages from a Google query | [mrscraper-serp](../mrscraper-serp/SKILL.md), using serp | +| Acquire and preserve raw content from known public URLs for any downstream task | [mrscraper-fetch](../mrscraper-fetch/SKILL.md), preferably first | +| Add managed extraction, repeated records, site mapping, schema guidance, or a reusable scraper | [mrscraper-scrape](../mrscraper-scrape/SKILL.md), normally after or alongside fetch | +| Discover relevant pages from a Google query | [mrscraper-serp](../mrscraper-serp/SKILL.md), then fetch selected pages | | Reproduce a prior scrape or run an existing AI/manual scraper on new URLs | rerun | | List or retrieve stored results | results or result | | Check account usage or domain request outcomes | status | -When “scrape this page” is ambiguous, choose from the requested output: +### Fetch-first principle -- Page content, reading, summarization, or analysis the current agent can do - from HTML → fetch. -- Defined fields, repeated records, schema-shaped JSON, or a reusable saved - extraction → scrape. -- Relevant URLs from a topic or query → serp. +For agent-led work, always use fetch for the first exploration of a known public +URL. The raw response preserves complete page context for later questions, +validation, and custom processing. Requesting JSON, a table, or named fields is +not by itself a reason to use scrape; derive those outputs locally from the raw +content when practical. -The general and listing modes of scrape retrieve page content and use an -extraction LLM to interpret it. That adds model-processing time and commits the -result to the extraction prompt. Prefer fetch when the current agent can -inspect the HTML and perform the requested analysis or transformation itself: -it is usually faster, preserves the source content, and leaves more flexibility -for follow-up questions. Use scrape when the user benefits from backend -structured extraction, listing pagination, map discovery, schema guidance, or -a reusable saved scraper. +The general and listing scrape modes send page content through a backend LLM. +Their output is shaped by the extraction prompt and may omit information. +Applying them across many pages also repeats model work that the agent can often +replace with one reusable local extractor. -For discovery-first work, call serp, select relevant URLs, and then load the -fetch or scrape skill for those pages. +Prefer this workflow for pages with a shared layout: + +1. Fetch representative pages and inspect their raw content. +2. Understand the shared page structure and required fields. +3. Define a stable output schema and reusable local extractor. +4. Fetch the remaining pages, in parallel when safe. +5. Apply the same extractor to every saved response. + +For roughly 100 same-layout pages, that means 100 fetches followed by one local +batch extraction instead of 100 separate backend-LLM extractions. This is often +faster, preserves every raw input, and allows fields to be recovered later if +the schema changes. + +Use scrape only when the user explicitly requests managed extraction or when it +still offers a clear benefit after fetch-led exploration. Even then, retain the +fetched source and verify important values against it. The map agent is separate +URL-discovery functionality; use it for bounded site mapping, then fetch the +pages whose content matters. + +For discovery-first work, call serp, select relevant URLs, fetch those pages, +and add managed extraction only when a derived structured view is useful. ## Step 4 — Handle Output and Artifacts @@ -147,7 +162,7 @@ Choose scraper type and target count independently: Single AI example: { - "target": "https://example.com/product", + "target": "https://www.scrapethissite.com/pages/simple/", "type": "ai", "scraper_id": "SCRAPER_UUID" } @@ -160,7 +175,7 @@ effect; the MCP server does not inject replacement defaults. Explicit-control example: { - "target": "https://example.com/products", + "target": "https://www.scrapethissite.com/pages/forms/?page_num=1", "type": "ai", "scraper_id": "SCRAPER_UUID", "proxy_country": "ID", @@ -173,7 +188,7 @@ These controls are rejected by manual and bulk reruns. Bulk example: { - "target": "https://a.example,https://b.example", + "target": "https://www.scrapethissite.com/pages/simple/,https://www.scrapethissite.com/pages/forms/?page_num=1", "type": "ai", "bulk": true, "id": "SCRAPER_UUID" @@ -207,7 +222,7 @@ Apply exact backend filters when the user knows identifying result fields: "scraper_id": "SCRAPER_UUID", "status": "Finished", "type": "Rerun-AI", - "url": "https://example.com/product" + "url": "https://www.scrapethissite.com/pages/simple/" } | Input | Default | Use | @@ -245,7 +260,7 @@ Call status with an empty object for subscription and token usage: Add domain request outcomes and a time range only when needed: { - "domain": "example.com", + "domain": "www.scrapethissite.com", "from": "7d", "to": "now" } From cecca90c4892750bba51e398396c3a46a7a4b97e Mon Sep 17 00:00:00 2001 From: Pray Apostel Date: Wed, 26 Aug 2026 13:39:55 +0700 Subject: [PATCH 4/4] docs: explain independent fetch loading modes --- .../mrscraper/skills/mrscraper-fetch/SKILL.md | 58 +++++++++++++------ 1 file changed, 41 insertions(+), 17 deletions(-) diff --git a/plugins/mrscraper/skills/mrscraper-fetch/SKILL.md b/plugins/mrscraper/skills/mrscraper-fetch/SKILL.md index c24af11..4600116 100644 --- a/plugins/mrscraper/skills/mrscraper-fetch/SKILL.md +++ b/plugins/mrscraper/skills/mrscraper-fetch/SKILL.md @@ -20,7 +20,8 @@ fetch calls MrScraper's Browser rendering executes page JavaScript, locale routing selects a geo-specific country, selector waits allow delayed content to appear, and homepage navigation establishes a normal navigation path before loading the -target. +target. Super Mode selects real-device routing independently of browser +rendering. Use fetch only for content the user is authorized to access and in accordance with the site's requirements. Page-loading controls can help render a site that @@ -92,6 +93,16 @@ data. Non-JSON page bodies are preserved exactly. ## Step 3 — Choose Page-Loading Options +browser_rendering and super_mode are independent axes. All four combinations +can return different content or failures for the same URL: + +| browser_rendering | super_mode | Loading path | +| --- | --- | --- | +| false | false | Standard routing with the non-browser loader. | +| true | false | Standard routing with browser loading and JavaScript. | +| false | true | Real-device routing with the non-browser loader. | +| true | true | Real-device routing with browser loading and JavaScript. | + Use browser rendering when the page depends on JavaScript: { @@ -99,8 +110,15 @@ Use browser rendering when the page depends on JavaScript: "browser_rendering": true } -If ordinary browser rendering still cannot load the page, route it through a -real device with Super Mode: +Use Super Mode with the non-browser loader when routing may be the problem but +browser loading is unnecessary or returns a worse response: + + { + "url": "https://www.scrapethissite.com/pages/simple/", + "super_mode": true + } + +Use both controls for real-device browser loading: { "url": "https://www.scrapethissite.com/pages/ajax-javascript/#2015", @@ -108,8 +126,9 @@ real device with Super Mode: "super_mode": true } -Use super_mode only after ordinary browser rendering fails. It requires -browser_rendering=true. +Browser rendering is not a strictly stronger mode. Some sites fail or return +worse content with browser_rendering=true but load successfully when it is +false. Super Mode does not enable browser rendering. Wait for delayed content with a CSS selector: @@ -147,7 +166,7 @@ Bound resource use for a browser-rendered page: | --- | --- | --- | --- | | url | required | Query url | Absolute HTTP or HTTPS target URL. | | browser_rendering | false | Query browserRendering | Execute page JavaScript. | -| super_mode | false | Query super | Route browser rendering through a real device; requires browser_rendering=true. | +| super_mode | false | Query super | Select real-device routing independently of browser rendering. | | geo_code | omitted | Query geoCode | Route through an ISO 3166-1 alpha-2 country. | | wait_for_selector | omitted | Query waitForSelector | Wait for a CSS selector; requires browser_rendering=true. | | home_page | false | Query homePage | Visit the site root before the target page. | @@ -161,19 +180,24 @@ never a tool input. ## Step 4 — Inspect and Retry Deliberately -Start with the simplest call that can load the page. If the response is -incomplete or missing dynamic content: +Start with both controls false unless the task already establishes a +requirement. If the response fails, is blocked, incomplete, or missing dynamic +content: 1. Inspect the initial response. -2. Retry once with browser_rendering=true when JavaScript is relevant. -3. Add super_mode only when ordinary browser rendering still fails. -4. Add wait_for_selector, geo_code, or home_page only when the target requires - that behavior. -5. Inspect the revised result before considering another retry. - -Do not repeat identical calls. Treat wait_for_selector as a CSS selector, not a -duration. Browser rendering loads a page; it does not click controls, submit -forms, or provide an authenticated interactive browser session. +2. Change one axis at a time: browser_rendering for JavaScript, or super_mode + when routing may be the problem. +3. If browser loading fails or returns worse content, retry the same super_mode + value with browser_rendering=false. +4. Try the remaining untested combinations when the response is still unusable. +5. Add wait_for_selector, geo_code, or home_page only when evidence shows that + the target requires it. +6. Stop after a usable response unless the user requests a comparison. + +Do not repeat an identical combination. Treat wait_for_selector as a CSS +selector, not a duration. Browser rendering loads a page; it does not click +controls, submit forms, or provide an authenticated interactive browser +session. ## Step 5 — Deliver the Result