smarter: factual-model routing (beh-10), fetch_url main-content extraction + 8k cap (url-08), get_weather via met.no (spec-016)

This commit is contained in:
Oleksandr Kozachuk
2026-07-17 19:15:18 +02:00
parent f8b9bc75ee
commit e0b97363c9
11 changed files with 404 additions and 6 deletions
+25
View File
@@ -149,6 +149,31 @@ class TestTextExtraction(unittest.TestCase):
self.assertNotIn("evil", text)
self.assertNotIn("x{}", text)
def test_chrome_and_link_boilerplate_dropped(self):
"""URL-08: nav/header/footer skipped; short link-dominated blocks (menus, related lists) removed."""
reader = URLReader(lambda: {}, None)
html = (
"<html><body>"
"<nav><a href='/a'>Home</a> <a href='/b'>Games</a></nav>"
"<header><a href='/login'>Login</a></header>"
"<ul><li><a href='/1'>Related article one</a></li><li><a href='/2'>Related article two</a></li></ul>"
"<article><p>The pop-up event runs from August 4 in Shibuya, with details "
"<a href='/x'>on the official page</a> for anyone attending the exhibition.</p></article>"
"<footer><a href='/imprint'>Imprint</a></footer>"
"</body></html>"
)
text = reader._to_text(html)
self.assertIn("pop-up event", text)
self.assertIn("on the official page", text) # inline link in a real paragraph survives
for chrome in ("Home", "Login", "Related article one", "Imprint"):
self.assertNotIn(chrome, text)
def test_default_cap_is_8000(self):
"""URL-08: the default url-max-chars budget is 8000."""
from fjerkroa_bot.url_reader import DEFAULT_MAX_CHARS
self.assertEqual(DEFAULT_MAX_CHARS, 8000)
class TestBodyReadCollectsAllChunks(unittest.IsolatedAsyncioTestCase):
async def test_get_reads_past_first_chunk(self):