mirror of
https://github.com/superdesigndev/treg.git
synced 2026-10-02 03:24:35 +08:00
feat(catalog): route web search, extract, and map
This commit is contained in:
@@ -2617,6 +2617,19 @@ adapters:
|
||||
miss: "coalesce(output.data.suggestions, []) == []"
|
||||
|
||||
# ---- open web -----------------------------------------------------------------------------
|
||||
tinyfish.web.search:
|
||||
accepts: [[q]]
|
||||
in: {q: queryParams.query}
|
||||
const: {queryParams.domain_type: web}
|
||||
out: {results: results, count: "coalesce(total_results, len(results))"}
|
||||
miss: "coalesce(results, []) == []"
|
||||
keenable.web.search:
|
||||
accepts: [[q]]
|
||||
in: {q: body.query}
|
||||
in_expr: {body.max_results: limit}
|
||||
const: {body.mode: pro}
|
||||
out: {results: results, count: "len(results)"}
|
||||
miss: "coalesce(results, []) == []"
|
||||
tavily.web.search:
|
||||
accepts: [[q]]
|
||||
in: {q: body.query}
|
||||
@@ -2631,20 +2644,68 @@ adapters:
|
||||
const: {body.type: auto}
|
||||
out: {results: results, count: "len(results)"}
|
||||
miss: "coalesce(results, []) == []"
|
||||
olostep.web.search:
|
||||
accepts: [[q]]
|
||||
in: {q: body.query}
|
||||
in_expr: {body.limit: limit}
|
||||
const: {body.fast_mode: true}
|
||||
out: {results: result.links, count: "len(result.links)"}
|
||||
miss: "coalesce(result.links, []) == []"
|
||||
|
||||
tinyfish.web.fetch:
|
||||
accepts: [[url]]
|
||||
in_expr: {body.urls: "list(url)"}
|
||||
test_identity: {url: "https://example.com"}
|
||||
const: {body.format: markdown, body.links: true, body.ttl: 0}
|
||||
out: {pages: results, count: "len(results)"}
|
||||
miss: "coalesce(results, []) == []"
|
||||
exa.web.contents.get:
|
||||
accepts: [[urls]]
|
||||
in: {urls: body.urls}
|
||||
accepts: [[url]]
|
||||
in_expr: {body.urls: "list(url)"}
|
||||
test_identity: {url: "https://en.wikipedia.org/wiki/Hierarchical_navigable_small_world"}
|
||||
const: {body.text: true}
|
||||
out: {pages: results, count: "len(results)"}
|
||||
miss: "coalesce(results, []) == []"
|
||||
olostep.web.scrape:
|
||||
accepts: [[url]]
|
||||
in: {url: body.url_to_scrape}
|
||||
const: {body.formats: [markdown], body.max_age: 604800}
|
||||
out: {pages: "list(result)", count: "len(list(result))"}
|
||||
miss: "result == null"
|
||||
tavily.web.extract:
|
||||
accepts: [[url]]
|
||||
in_expr: {body.urls: "list(url)"}
|
||||
test_identity: {url: "https://example.com"}
|
||||
const: {body.extract_depth: basic}
|
||||
out: {pages: results, count: "len(results)"}
|
||||
miss: "coalesce(results, []) == []"
|
||||
keenable.web.fetch:
|
||||
accepts: [[url]]
|
||||
in: {url: queryParams.url}
|
||||
const: {queryParams.max_chars: 500, queryParams.live: false}
|
||||
out: {pages: "list(.)", count: "len(list(.))"}
|
||||
miss: ". == null"
|
||||
|
||||
anyapi.web.map:
|
||||
accepts: [[url]]
|
||||
in: {url: body.url}
|
||||
accepts: [[url], [url, q]]
|
||||
in: {url: body.url, q: body.search}
|
||||
in_expr: {body.limit: limit}
|
||||
test_identity: {url: "https://www.iana.org", q: domain}
|
||||
out: {results: output.data.results, count: "len(output.data.results)"}
|
||||
miss: "coalesce(output.data.results, []) == []"
|
||||
tavily.web.map:
|
||||
accepts: [[url], [url, q]]
|
||||
in: {url: body.url, q: body.instructions}
|
||||
in_expr: {body.limit: limit}
|
||||
out: {results: results, count: "len(results)"}
|
||||
miss: "coalesce(results, []) == []"
|
||||
olostep.web.map.search:
|
||||
accepts: [[url, q]]
|
||||
in: {url: body.url, q: body.search_query}
|
||||
in_expr: {body.top_n: limit}
|
||||
test_identity: {url: "https://docs.olostep.com", q: "billing and credits"}
|
||||
out: {results: urls, count: "len(urls)"}
|
||||
miss: "coalesce(urls, []) == []"
|
||||
|
||||
anyapi.web.crawl:
|
||||
accepts: [[url]]
|
||||
|
||||
@@ -1109,9 +1109,9 @@ contracts:
|
||||
idempotent: true
|
||||
|
||||
web.extract:
|
||||
summary: "Extract clean content from one or more web pages — provider-native rows in `pages`"
|
||||
summary: "Extract clean content from one web page — provider-native rows in `pages`"
|
||||
identity:
|
||||
- {urls: list}
|
||||
- {url: url}
|
||||
filters: {}
|
||||
output:
|
||||
pages: {type: list, required: true, note: "the provider's own extracted page objects; raw is the same body"}
|
||||
@@ -1123,6 +1123,7 @@ contracts:
|
||||
summary: "Discover the URLs on a website — provider-native map rows in `results`"
|
||||
identity:
|
||||
- {url: url}
|
||||
- {url: url, q: str}
|
||||
filters:
|
||||
limit: {type: int, default: 10, note: "maximum discovered URLs"}
|
||||
output:
|
||||
|
||||
+11
-9
@@ -140,19 +140,21 @@ def test_openmart_tools_are_direct_only_not_routed():
|
||||
assert "openmart.companies.enrich" not in cat.by_id["treg.companies.enrich"]["routed_children"]
|
||||
|
||||
|
||||
def test_tavily_routes_only_search_and_keeps_result_settled_tools_direct():
|
||||
def test_tavily_routes_synchronous_web_tools_and_keeps_crawl_direct():
|
||||
cat = catalog_store.load()
|
||||
assert "tavily.web.search" in cat.by_id["treg.web.search"]["routed_children"]
|
||||
assert cat.adapters["tavily.web.search"].verified
|
||||
direct = {
|
||||
routed = {
|
||||
"tavily.web.search": "treg.web.search",
|
||||
"tavily.web.extract": "treg.web.extract",
|
||||
"tavily.web.map": "treg.web.map",
|
||||
"tavily.web.crawl": "treg.web.crawl",
|
||||
}
|
||||
for child, parent in direct.items():
|
||||
assert child not in cat.adapters
|
||||
assert parent not in cat.by_id or child not in cat.by_id[parent]["routed_children"]
|
||||
assert cat.platform_eligible(cat.by_id[child])
|
||||
for child, parent in routed.items():
|
||||
assert cat.adapters[child].verified
|
||||
assert child in cat.by_id[parent]["routed_children"]
|
||||
assert "tavily.web.crawl" not in cat.adapters
|
||||
assert "treg.web.crawl" not in cat.by_id or (
|
||||
"tavily.web.crawl" not in cat.by_id["treg.web.crawl"]["routed_children"]
|
||||
)
|
||||
assert cat.platform_eligible(cat.by_id["tavily.web.crawl"])
|
||||
|
||||
|
||||
async def test_tavily_routed_empty_search_is_a_paid_miss_then_falls_through(
|
||||
|
||||
Reference in New Issue
Block a user