Files
AI-Profile-Router/dev/test_web_search_mcp.py
T

149 lines
6.1 KiB
Python

#!/usr/bin/env python3
"""Regression tests for the compact web MCP facade."""
from __future__ import annotations
import importlib.util
import json
import pathlib
import unittest
from unittest import mock
ROOT = pathlib.Path(__file__).resolve().parents[1]
SPEC = importlib.util.spec_from_file_location(
"web_search_mcp", ROOT / "platform/web-search/web_search_mcp.py"
)
WEB = importlib.util.module_from_spec(SPEC)
assert SPEC.loader
SPEC.loader.exec_module(WEB)
class WebSearchMcpTests(unittest.TestCase):
def setUp(self) -> None:
WEB._search_attempts.clear()
def test_tool_surface_stays_small_and_explicit(self) -> None:
self.assertEqual(
[tool["name"] for tool in WEB.TOOLS],
["web_search", "web_read", "web_youtube", "web_compare", "web_shop", "web_research"],
)
def test_current_queries_do_not_get_wikipedia_noise(self) -> None:
with (
mock.patch.object(WEB, "SEARXNG_URL", "http://searxng:8080"),
mock.patch.object(WEB, "searxng_json", return_value={"results": []}),
mock.patch.object(WEB, "wikipedia_search") as wikipedia,
):
results, _ = WEB.general_discovery("latest video The Proper People", 4)
self.assertEqual(results, [])
wikipedia.assert_not_called()
def test_empty_fresh_search_relaxes_once_and_marks_result(self) -> None:
hit = {"title": "Current page", "url": "https://example.com/current"}
with mock.patch.object(
WEB,
"general_discovery",
side_effect=[([], []), ([hit], [])],
) as discovery:
result = WEB.web_search({
"query": "current test release",
"freshness": "week",
"max_results": 3,
})
self.assertTrue(result["task_complete"])
self.assertFalse(result["freshness_applied"])
self.assertIn("unfiltered", result["backend_warning"])
self.assertEqual(discovery.call_count, 2)
def test_related_search_budget_is_enforced(self) -> None:
with mock.patch.object(WEB, "SEARCH_BUDGET_MAX_RELATED_CALLS", 2):
self.assertTrue(WEB.consume_search_budget("latest Proper People video")[0])
self.assertTrue(WEB.consume_search_budget("Proper People newest video")[0])
self.assertFalse(WEB.consume_search_budget("newest video by Proper People")[0])
def test_youtube_feed_provides_order_and_dates(self) -> None:
feed = b'''<?xml version="1.0" encoding="UTF-8"?>
<feed xmlns:yt="http://www.youtube.com/xml/schemas/2015"
xmlns="http://www.w3.org/2005/Atom">
<entry><yt:videoId>new123</yt:videoId><title>Newest</title>
<published>2026-08-23T12:00:00+00:00</published>
<author><name>The Proper People</name></author></entry>
<entry><yt:videoId>old456</yt:videoId><title>Older</title>
<published>2026-08-10T12:00:00+00:00</published>
<author><name>The Proper People</name></author></entry>
</feed>'''
with mock.patch.object(WEB, "fetch_public_bytes", return_value=feed):
rows = WEB.youtube_feed_records(
"https://www.youtube.com/channel/UCcem9I78ybZLHLRUlkUO3sw", 2
)
self.assertEqual([row["title"] for row in rows], ["Newest", "Older"])
self.assertEqual(rows[0]["published_at"], "2026-08-23T12:00:00+00:00")
def test_latest_youtube_is_one_bounded_specialist_operation(self) -> None:
row = {
"title": "Newest",
"url": "https://www.youtube.com/watch?v=new123",
"source_kind": "youtube_channel_feed",
}
with (
mock.patch.object(WEB, "resolve_youtube_channel", return_value="https://www.youtube.com/channel/UCcem9I78ybZLHLRUlkUO3sw"),
mock.patch.object(WEB, "youtube_feed_records", return_value=[row]),
mock.patch.object(WEB, "run_ytdlp") as ytdlp,
):
result = WEB.web_youtube({"query": "The Proper People", "mode": "latest"})
self.assertTrue(result["task_complete"])
self.assertEqual(result["results"][0]["title"], "Newest")
ytdlp.assert_not_called()
def test_latest_long_youtube_uses_verified_videos_tab(self) -> None:
row = {
"title": "Newest long video",
"url": "https://www.youtube.com/watch?v=long123",
"content_type": "long",
"content_type_verified": True,
"source_kind": "youtube_videos_tab",
}
with (
mock.patch.object(
WEB,
"resolve_youtube_channel",
return_value="https://www.youtube.com/channel/UCcem9I78ybZLHLRUlkUO3sw",
),
mock.patch.object(WEB, "youtube_tab_records", return_value=[row]) as tab,
mock.patch.object(WEB, "youtube_feed_records") as feed,
):
result = WEB.web_youtube({
"query": "The Proper People",
"mode": "latest",
"content_type": "long",
})
tab.assert_called_once_with(
"https://www.youtube.com/channel/UCcem9I78ybZLHLRUlkUO3sw", "long", 5
)
feed.assert_not_called()
self.assertEqual(result["content_type_filter"], "long")
self.assertTrue(result["results"][0]["content_type_verified"])
def test_youtube_rejects_content_filter_outside_latest_mode(self) -> None:
with self.assertRaisesRegex(ValueError, "only supported with mode=latest"):
WEB.web_youtube({
"query": "The Proper People",
"mode": "search",
"content_type": "long",
})
def test_web_read_does_not_consume_search_loop_budget(self) -> None:
page = {"url": "https://example.com/a", "page_evidence": ["Evidence"]}
with mock.patch.object(WEB, "scrape", return_value=[page]):
result = json.loads(WEB.call_tool("web_read", {
"url": "https://example.com/a",
"question": "What does this page say?",
}))
self.assertTrue(result["task_complete"])
self.assertEqual(WEB._search_attempts, [])
if __name__ == "__main__":
unittest.main()