fix: import paths (fetchers→readers), schema field mismatch, RSS detection

Bugs found during fresh pip install testing:
- readers/*.py still referenced agent_eyes.fetchers → fixed to agent_eyes.readers
- reader.py passed 'author'/'metadata' to UnifiedContent which doesn't have those → use 'extra' field
- RSS URL detection missed domains containing 'rss' (e.g. hnrss.org)
This commit is contained in:
Panniantong
2026-02-24 05:13:52 +01:00
parent 8eab038cb9
commit d891af5b7d
5 changed files with 11 additions and 13 deletions
+3 -3
View File
@@ -13,7 +13,7 @@ from loguru import logger
from typing import Dict, Any
from pathlib import Path
from agent_eyes.fetchers.jina import fetch_via_jina
from agent_eyes.readers.jina import fetch_via_jina
async def fetch_xhs(url: str) -> Dict[str, Any]:
@@ -43,7 +43,7 @@ async def fetch_xhs(url: str) -> Dict[str, Any]:
logger.warning(f"[XHS] Jina failed ({e}), falling back to browser")
# Tier 2: Playwright with session
from agent_eyes.fetchers.browser import get_session_path, SESSION_DIR
from agent_eyes.readers.browser import get_session_path, SESSION_DIR
session_path = get_session_path("xhs")
if not Path(session_path).exists():
@@ -56,7 +56,7 @@ async def fetch_xhs(url: str) -> Dict[str, Any]:
try:
logger.info(f"[XHS] Tier 2 — Playwright with session: {url}")
from agent_eyes.fetchers.browser import fetch_via_browser
from agent_eyes.readers.browser import fetch_via_browser
data = await fetch_via_browser(url, storage_state=session_path)
return {