mirror of
https://github.com/karust/openserp.git
synced 2026-08-24 17:00:06 +08:00
- Add structured `search [engine] [query]` CLI: --limit/--lang/--region/--site/--file, --format (json|text|markdown|ndjson), --extract N, --search-timeout; Envelope and route logs to stderr with a --quiet default (fixes stdout pollution) - Unify engines behind a single engineSpec registry (CLI + serve share it) - Unify the extract knob to bool-or-int `extract=N` (drop extract_top); CLI and HTTP share core batch extraction, raw/rendered fetch, and clamp helpers - Engines: Ecosia CF captcha detection (raw + browser), Yandex progressive-result wait, Google PAA poll + Has() existence probes, Bing title/desc attribute fallbacks - Proxy: rotate challenged proxies out of the tag pool for one retry (X-Proxy-Attempts); browser health-ping skip window; opt-in WaitStable
29 lines
984 B
JavaScript
29 lines
984 B
JavaScript
import { OpenSERP } from "@openserp/sdk";
|
|
|
|
// Hosted API instead? Get a key at https://openserp.org/dashboard/keys:
|
|
// const client = new OpenSERP({ apiKey: "<YOUR_API_TOKEN>", timeoutMs: 60_000 });
|
|
const client = new OpenSERP({ baseUrl: "http://localhost:7000", timeoutMs: 60_000 });
|
|
|
|
// `extract: N` fetches the top N pages (max 5) and returns their cleaned
|
|
// content alongside each result, so you get the page text in a single request.
|
|
// `extract: true` is shorthand for the top result.
|
|
const { results } = await client.search({
|
|
engine: "ecosia",
|
|
text: "what is a serp api",
|
|
extract: 2,
|
|
extractMode: "auto",
|
|
});
|
|
|
|
for (const item of results) {
|
|
console.log(`${item.rank}. ${item.title}`);
|
|
console.log(` ${item.url}`);
|
|
|
|
// When extracted, `item.extracted.content` holds the page body. Trim it to
|
|
// a short preview here.
|
|
const content = item.extracted?.content;
|
|
if (content) {
|
|
console.log(` ${content.slice(0, 300).trim()}…`);
|
|
}
|
|
console.log();
|
|
}
|