mirror of
https://github.com/pewdiepie-archdaemon/odysseus.git
synced 2026-08-01 19:18:35 -04:00
5f301a45da
* fix(search): add download budgets to web_fetch with truncation notice and hard ceiling MAX_OUTPUT_CHARS only trims what the agent sees; fetch_webpage_content buffered and cached the entire response body first, so a large or hostile URL could pull arbitrarily many bytes into memory and the content cache. The fetch is now a capped streaming GET (SSRF redirect guard unchanged): a soft default budget (WEB_FETCH_SOFT_MAX_BYTES, 2 MB), a per-call override via full/max_bytes on the web_fetch tool, and a hard ceiling (WEB_FETCH_HARD_MAX_BYTES, 20 MB) that the override can never exceed. When Content-Length already declares a body over the ceiling the fetch is refused before any body bytes are buffered. Truncated results carry truncated/fetched_bytes/total_bytes, the tool output leads with a partial-content notice telling the model how to re-fetch with full=true, and the tool schema documents the flag. A truncated PDF is reported as a budget error since a cut PDF is unparseable. The effective cap is part of the content-cache key so a truncated fetch is never served to a full-budget request. Existing tests that faked httpx.get or the old _get_public_url signature are adapted to the streaming interface; behavior pins are unchanged. Fixes #3812 * fix(search): close compressed-body cap bypass and protect the partial notice Addresses RaresKeY's review on #3955: - Force Accept-Encoding: identity for the capped fetch. With gzip/deflate the wire bytes (and Content-Length) can be a fraction of the decoded body, so a tiny compressed response could pass the hard-cap preflight and then expand past the ceiling in a single decoded chunk before the streamed cap could slice it. Identity makes Content-Length the true body size and keeps each streamed chunk bounded by the network read, so the hard ceiling actually bounds memory. - Lead web_fetch output with the partial-content notice and cap the page title. The notice is the user-facing contract for partial fetches, but the title is untrusted, uncapped page content; placed ahead of the notice a giant title could push it past MAX_OUTPUT_CHARS and drop it. The notice now leads and the title is capped as a second guard. Adds regressions: the fetch advertises identity encoding, and a truncated result with an oversized title still surfaces the partial notice. * fix(search): reject compressed responses that ignore the identity request Requesting Accept-Encoding: identity is not enough on its own: a server can ignore it and still return Content-Encoding: gzip, and httpx.iter_bytes would decode that, so a tiny compressed body could balloon into one decoded chunk far past the hard cap before the streamed loop slices it (and Content-Length, the compressed wire length, makes the preflight and size metadata unreliable). Refuse a non-identity Content-Encoding before reading the body. Adds a regression where the server ignores the identity request and returns gzip; the fetch is refused before any body is decoded.
132 lines
5.9 KiB
Python
132 lines
5.9 KiB
Python
import asyncio
|
|
import json
|
|
from typing import Dict, Any
|
|
|
|
from src.constants import MAX_OUTPUT_CHARS
|
|
|
|
class WebSearchTool:
|
|
async def execute(self, content: str, ctx: dict) -> dict:
|
|
from src.search import comprehensive_web_search
|
|
raw = content.strip()
|
|
query = raw
|
|
time_filter = None
|
|
max_pages = 5
|
|
if raw.startswith("{"):
|
|
try:
|
|
parsed = json.loads(raw)
|
|
if isinstance(parsed, dict) and "query" in parsed:
|
|
query = str(parsed.get("query", "")).strip()
|
|
tf = parsed.get("time_filter") or parsed.get("freshness")
|
|
if isinstance(tf, str) and tf.lower() in ("day", "week", "month", "year"):
|
|
time_filter = tf.lower()
|
|
mp = parsed.get("max_pages")
|
|
if isinstance(mp, int) and 1 <= mp <= 10:
|
|
max_pages = mp
|
|
except json.JSONDecodeError:
|
|
pass
|
|
if not query:
|
|
query = raw.split("\n")[0].strip()
|
|
if time_filter is None:
|
|
q_lc = query.lower()
|
|
if any(kw in q_lc for kw in ("today", "latest", "breaking", "this morning", "right now", "currently")):
|
|
time_filter = "day"
|
|
elif any(kw in q_lc for kw in ("this week", "past week", "recent news", "last few days")):
|
|
time_filter = "week"
|
|
elif any(kw in q_lc for kw in ("this month", "past month")):
|
|
time_filter = "month"
|
|
elif " news" in q_lc or q_lc.startswith("news ") or q_lc.endswith(" news"):
|
|
time_filter = "week"
|
|
loop = asyncio.get_running_loop()
|
|
text, sources = await asyncio.wait_for(
|
|
loop.run_in_executor(
|
|
None,
|
|
lambda: comprehensive_web_search(
|
|
query,
|
|
max_pages=max_pages,
|
|
time_filter=time_filter,
|
|
return_sources=True,
|
|
),
|
|
),
|
|
timeout=30,
|
|
)
|
|
output = text[:MAX_OUTPUT_CHARS] if len(text) > MAX_OUTPUT_CHARS else text
|
|
if sources:
|
|
output += "\n\n<!-- SOURCES:" + json.dumps(sources) + " -->"
|
|
return {"output": output, "exit_code": 0}
|
|
|
|
class WebFetchTool:
|
|
async def execute(self, content: str, ctx: dict) -> dict:
|
|
from src.search.content import fetch_webpage_content
|
|
from src.constants import WEB_FETCH_HARD_MAX_BYTES
|
|
raw = content.strip()
|
|
url = ""
|
|
max_bytes = None
|
|
if raw.startswith("{"):
|
|
try:
|
|
parsed = json.loads(raw)
|
|
if isinstance(parsed, dict):
|
|
url = str(parsed.get("url") or "").strip()
|
|
# Download-budget override (#3812): "full": true raises the
|
|
# budget to the hard cap; an explicit max_bytes is clamped
|
|
# to the hard cap downstream. Default stays the soft cap.
|
|
if parsed.get("full") is True:
|
|
max_bytes = WEB_FETCH_HARD_MAX_BYTES
|
|
mb = parsed.get("max_bytes")
|
|
if isinstance(mb, int) and mb > 0:
|
|
max_bytes = mb
|
|
except json.JSONDecodeError:
|
|
url = ""
|
|
if not url:
|
|
url = raw.split("\n")[0].strip()
|
|
if not url or url.startswith("{") or any(c in url for c in (" ", "\t", "\n")):
|
|
return {"error": "web_fetch: provide a single URL or domain, e.g. example.com", "exit_code": 1}
|
|
low = url.lower()
|
|
if "://" in low and not low.startswith(("http://", "https://")):
|
|
return {"error": f"web_fetch: unsupported URL scheme (only http/https): {url[:80]}", "exit_code": 1}
|
|
if not low.startswith(("http://", "https://")):
|
|
url = "https://" + url
|
|
loop = asyncio.get_running_loop()
|
|
try:
|
|
result = await asyncio.wait_for(
|
|
loop.run_in_executor(None, lambda: fetch_webpage_content(url, timeout=10, max_bytes=max_bytes)),
|
|
timeout=30,
|
|
)
|
|
except asyncio.TimeoutError:
|
|
return {"error": f"web_fetch: timed out fetching {url}", "exit_code": 1}
|
|
except Exception as e:
|
|
return {"error": f"web_fetch: {url}: {e}", "exit_code": 1}
|
|
err = result.get("error")
|
|
text = (result.get("content") or "").strip()
|
|
title = result.get("title") or ""
|
|
|
|
if not text:
|
|
if err:
|
|
return {"error": f"web_fetch: {url}: {err}", "exit_code": 1}
|
|
return {"error": f"web_fetch: {url}: no readable text content (not HTML, or the page needs JS/login)", "exit_code": 1}
|
|
|
|
# Tell the model when the download budget cut the body short and how
|
|
# to get the rest, instead of silently presenting a partial page as
|
|
# the whole thing.
|
|
size_note = ""
|
|
if result.get("truncated"):
|
|
fetched = result.get("fetched_bytes") or 0
|
|
total = result.get("total_bytes")
|
|
total_txt = f" of {total:,} bytes" if total else ""
|
|
size_note = (
|
|
f"[partial content: download stopped at {fetched:,} bytes{total_txt}. "
|
|
f'Re-call with {{"url": "{url}", "full": true}} to fetch up to '
|
|
f"{WEB_FETCH_HARD_MAX_BYTES:,} bytes.]\n\n"
|
|
)
|
|
|
|
# The notice must lead the output so the MAX_OUTPUT_CHARS trim below can
|
|
# never drop it. The title is untrusted, uncapped page content, so a
|
|
# giant title ahead of the notice could push it out of range; keep the
|
|
# notice first and cap the title as a second guard.
|
|
if len(title) > 300:
|
|
title = title[:300] + "..."
|
|
header = (f"# {title}\n" if title else "") + f"Source: {url}\n\n"
|
|
output = size_note + header + text
|
|
if len(output) > MAX_OUTPUT_CHARS:
|
|
output = output[:MAX_OUTPUT_CHARS] + "\n\n[...truncated]"
|
|
return {"output": output, "exit_code": 0}
|