1503 lines
56 KiB
Python
1503 lines
56 KiB
Python
from datetime import UTC, datetime, timedelta
|
|
from types import SimpleNamespace
|
|
|
|
import pytest
|
|
|
|
from app.services.earth_news import (
|
|
NewsFeedEndpoint,
|
|
NewsFeedSource,
|
|
NewsTargetLocation,
|
|
ParsedNewsItem,
|
|
apply_news_classification,
|
|
default_earth_news_sources_payload,
|
|
normalize_earth_news_sources_payload,
|
|
_fetch_source,
|
|
_enrich_items_with_target_locations,
|
|
_extract_target_location_from_text,
|
|
_parse_feed_entries,
|
|
_rank_and_trim_items,
|
|
_serialize_item,
|
|
get_earth_news_payload,
|
|
test_news_source_config as run_news_source_config_test,
|
|
)
|
|
from app.services.earth_news_queue import NewsTargetLocationMessage
|
|
from app.services.earth_news_worker import process_target_location_message
|
|
from app.services.collectors.media_news_archive import MediaNewsArchiveCollector
|
|
|
|
|
|
def test_serialize_item_includes_region_anchor_for_cruise():
|
|
item = ParsedNewsItem(
|
|
id="google-apac:test",
|
|
title="Example APAC story",
|
|
summary="Example summary",
|
|
url="https://example.com/story",
|
|
source="Example Source",
|
|
feed_name="Global Monitor / APAC",
|
|
feed_region="asia-pacific",
|
|
homepage_url="https://example.com",
|
|
published_at=datetime(2026, 4, 23, 2, 30, tzinfo=UTC),
|
|
)
|
|
|
|
payload = _serialize_item(item, active_region="asia-pacific")
|
|
|
|
assert payload["latitude"] == 1.3521
|
|
assert payload["longitude"] == 103.8198
|
|
assert payload["location_label"] == "亚太"
|
|
assert payload["location_source"] == "region_anchor"
|
|
assert payload["verified"] is False
|
|
assert payload["location_meta"]["target"] is None
|
|
assert payload["location_meta"]["anchor"]["region"] == "asia-pacific"
|
|
assert payload["is_focus_match"] is True
|
|
assert payload["published_at"] == "2026-04-23T02:30:00Z"
|
|
|
|
|
|
def test_serialize_item_includes_breaking_fields():
|
|
item = ParsedNewsItem(
|
|
id="breaking:test",
|
|
title="Major market halt",
|
|
summary="Trading halt after flash crash",
|
|
url="https://example.com/breaking",
|
|
source="Example Source",
|
|
feed_name="Example Feed",
|
|
feed_region="global",
|
|
homepage_url="https://example.com",
|
|
published_at=datetime(2026, 5, 15, 2, 0, tzinfo=UTC),
|
|
breaking_level="critical",
|
|
breaking_scope="global",
|
|
breaking_reasons=["重大金融市场异常"],
|
|
breaking_source="rules",
|
|
breaking_confidence=0.72,
|
|
breaking_expires_at=datetime(2026, 5, 16, 2, 0, tzinfo=UTC),
|
|
)
|
|
|
|
payload = _serialize_item(item, active_region="europe")
|
|
|
|
assert payload["breaking_level"] == "critical"
|
|
assert payload["breaking_scope"] == "global"
|
|
assert payload["breaking_reasons"] == ["重大金融市场异常"]
|
|
assert payload["breaking_source"] == "rules"
|
|
assert payload["breaking_confidence"] == 0.72
|
|
assert payload["breaking_expires_at"] == "2026-05-16T02:00:00Z"
|
|
|
|
|
|
def test_rank_and_trim_items_prioritizes_active_breaking():
|
|
older_breaking = ParsedNewsItem(
|
|
id="global:critical",
|
|
title="Nuclear accident reported",
|
|
summary="A nuclear accident has been reported.",
|
|
url="https://example.com/critical",
|
|
source="Global Source",
|
|
feed_name="Global Feed",
|
|
feed_region="global",
|
|
homepage_url="https://example.com",
|
|
published_at=datetime.now(UTC) - timedelta(hours=2),
|
|
breaking_level="critical",
|
|
breaking_scope="global",
|
|
breaking_expires_at=datetime.now(UTC) + timedelta(hours=6),
|
|
)
|
|
newer_regular = ParsedNewsItem(
|
|
id="europe:regular",
|
|
title="Regular Europe story",
|
|
summary="A newer regular story.",
|
|
url="https://example.com/regular",
|
|
source="Europe Source",
|
|
feed_name="Europe Feed",
|
|
feed_region="europe",
|
|
homepage_url="https://example.com",
|
|
published_at=datetime.now(UTC),
|
|
)
|
|
expired_breaking = ParsedNewsItem(
|
|
id="europe:expired",
|
|
title="Expired breaking",
|
|
summary="Expired breaking story.",
|
|
url="https://example.com/expired",
|
|
source="Europe Source",
|
|
feed_name="Europe Feed",
|
|
feed_region="europe",
|
|
homepage_url="https://example.com",
|
|
published_at=datetime.now(UTC) + timedelta(minutes=1),
|
|
breaking_level="critical",
|
|
breaking_scope="regional",
|
|
breaking_expires_at=datetime.now(UTC) - timedelta(minutes=1),
|
|
)
|
|
|
|
ranked = _rank_and_trim_items(
|
|
[newer_regular, expired_breaking, older_breaking],
|
|
active_region="europe",
|
|
limit=3,
|
|
)
|
|
|
|
assert [item.id for item in ranked] == ["global:critical", "europe:expired", "europe:regular"]
|
|
|
|
|
|
def test_serialize_item_falls_back_to_global_anchor():
|
|
item = ParsedNewsItem(
|
|
id="custom:test",
|
|
title="Fallback story",
|
|
summary="Fallback summary",
|
|
url="https://example.com/fallback",
|
|
source="Fallback Source",
|
|
feed_name="Fallback Feed",
|
|
feed_region="unknown-region",
|
|
homepage_url="https://example.com",
|
|
published_at=None,
|
|
)
|
|
|
|
payload = _serialize_item(item, active_region="americas")
|
|
|
|
assert payload["latitude"] == 20.0
|
|
assert payload["longitude"] == 0.0
|
|
assert payload["location_label"] == "全球"
|
|
assert payload["location_source"] == "region_anchor"
|
|
assert payload["verified"] is False
|
|
assert payload["location_meta"]["anchor"]["region"] == "global"
|
|
assert payload["is_focus_match"] is False
|
|
assert payload["published_at"] is None
|
|
|
|
|
|
def test_serialize_item_includes_inferred_target_location():
|
|
item = ParsedNewsItem(
|
|
id="bbc-world:f55310fb667b",
|
|
title="Watch: What happened on day one of Trump's China visit?",
|
|
summary=(
|
|
"China welcomed US President Donald Trump with cheering children "
|
|
"and a troop parade."
|
|
),
|
|
url="https://example.com/china-visit",
|
|
source="BBC World",
|
|
feed_name="BBC World",
|
|
feed_region="global",
|
|
homepage_url="https://example.com",
|
|
published_at=datetime(2026, 5, 14, 7, 13, 24, tzinfo=UTC),
|
|
target_location=NewsTargetLocation(
|
|
latitude=39.9042,
|
|
longitude=116.4074,
|
|
label="Beijing, China",
|
|
source="ai_inferred_target",
|
|
confidence=0.88,
|
|
country="中国",
|
|
city="Beijing",
|
|
),
|
|
)
|
|
|
|
payload = _serialize_item(item, active_region="global")
|
|
|
|
assert payload["latitude"] == 39.9042
|
|
assert payload["longitude"] == 116.4074
|
|
assert payload["location_label"] == "Beijing, China"
|
|
assert payload["location_source"] == "ai_inferred_target"
|
|
assert payload["verified"] is True
|
|
assert payload["location_meta"]["target"]["confidence"] == 0.88
|
|
assert payload["location_meta"]["target"]["country"] == "中国"
|
|
assert payload["location_meta"]["target"]["city"] == "Beijing"
|
|
assert payload["location_meta"]["resolution_stage"] == "unresolved"
|
|
assert payload["location_meta"]["ai_attempted"] is False
|
|
assert payload["location_meta"]["ai_status"] == "not_attempted"
|
|
assert payload["location_meta"]["ai_error"] is None
|
|
|
|
|
|
def test_parse_plain_rss_uses_feed_name_as_source():
|
|
source = NewsFeedSource(
|
|
id="bbc-world",
|
|
name="BBC World",
|
|
region="global",
|
|
feed_url="https://feeds.bbci.co.uk/news/world/rss.xml",
|
|
homepage_url="https://www.bbc.com/news/world",
|
|
)
|
|
xml = """
|
|
<rss>
|
|
<channel>
|
|
<item>
|
|
<title>This may be the last time you hear my voice: Political executions surge in Iran since start of war</title>
|
|
<description>Story summary</description>
|
|
<link>https://www.bbc.com/news/example</link>
|
|
<pubDate>Fri, 15 May 2026 03:00:00 GMT</pubDate>
|
|
</item>
|
|
</channel>
|
|
</rss>
|
|
"""
|
|
|
|
items = _parse_feed_entries(xml, source)
|
|
|
|
assert items[0].title == "This may be the last time you hear my voice: Political executions surge in Iran since start of war"
|
|
assert items[0].source == "BBC World"
|
|
|
|
|
|
def test_parse_aggregated_rss_splits_publisher_from_title():
|
|
source = NewsFeedSource(
|
|
id="global-scan",
|
|
name="Global Monitor / World",
|
|
region="global",
|
|
feed_url="https://news.google.com/rss",
|
|
homepage_url="https://news.google.com/",
|
|
source_type="aggregated",
|
|
)
|
|
xml = """
|
|
<rss>
|
|
<channel>
|
|
<item>
|
|
<title>Example headline - Reuters</title>
|
|
<description>Story summary</description>
|
|
<link>https://news.google.com/example</link>
|
|
</item>
|
|
</channel>
|
|
</rss>
|
|
"""
|
|
|
|
items = _parse_feed_entries(xml, source)
|
|
|
|
assert items[0].title == "Example headline"
|
|
assert items[0].source == "Reuters"
|
|
|
|
|
|
def test_parse_chinese_rss_marks_source_language_and_keeps_zh_localization():
|
|
source = NewsFeedSource(
|
|
id="36kr",
|
|
name="36氪",
|
|
region="asia-pacific",
|
|
feed_url="https://36kr.com/feed",
|
|
homepage_url="https://www.36kr.com/",
|
|
source_tags=("china", "business_news"),
|
|
default_category="business",
|
|
)
|
|
xml = """
|
|
<rss>
|
|
<channel>
|
|
<item>
|
|
<title>中国电商平台发布季度增长数据</title>
|
|
<description>平台表示,跨境电商订单量同比增长。</description>
|
|
<link>https://36kr.com/p/example</link>
|
|
</item>
|
|
</channel>
|
|
</rss>
|
|
"""
|
|
|
|
items = _parse_feed_entries(xml, source)
|
|
payload_zh = _serialize_item(items[0], active_region="global", locale="zh-CN")
|
|
payload_en = _serialize_item(items[0], active_region="global", locale="en-US")
|
|
|
|
assert items[0].content_language == "zh-CN"
|
|
assert items[0].localizations["zh-CN"]["title"] == "中国电商平台发布季度增长数据"
|
|
assert payload_zh["display_title"] == "中国电商平台发布季度增长数据"
|
|
assert payload_en["display_title"] == "中国电商平台发布季度增长数据"
|
|
|
|
|
|
def test_default_news_sources_include_business_and_ecommerce_sources():
|
|
payload = default_earth_news_sources_payload()
|
|
sources_by_id = {source["id"]: source for source in payload["sources"]}
|
|
source_ids = {source["id"] for source in payload["sources"]}
|
|
category_keys = {category["key"] for category in payload["categories"]}
|
|
tag_keys = {tag["key"] for tag in payload["source_tags"]}
|
|
|
|
assert "cnbc-business" in source_ids
|
|
assert "36kr" in source_ids
|
|
assert "techcrunch" in source_ids
|
|
assert "retaildive" in source_ids
|
|
assert "prnewswire-retail" in source_ids
|
|
assert "google-news" in source_ids
|
|
assert "global-scan" not in source_ids
|
|
assert "google-americas" not in source_ids
|
|
assert "google-europe" not in source_ids
|
|
assert "google-mea" not in source_ids
|
|
assert "google-apac" not in source_ids
|
|
assert "businesswire-ecommerce" in source_ids
|
|
assert "us-census-ecommerce" in source_ids
|
|
assert "mofcom-data" in source_ids
|
|
assert "stats-china-online-retail" in source_ids
|
|
assert "ebrun" in source_ids
|
|
assert sources_by_id["36kr"]["source_type"] == "rss"
|
|
assert sources_by_id["36kr"]["homepage_url"] == "https://www.36kr.com/"
|
|
assert sources_by_id["36kr"]["feed_directory_url"] == "https://www.36kr.com/rss-center"
|
|
kr_feeds = {feed["id"]: feed for feed in sources_by_id["36kr"]["feeds"]}
|
|
assert set(kr_feeds) == {"feed", "article", "newsflash", "moment"}
|
|
assert kr_feeds["feed"]["url"] == "https://36kr.com/feed"
|
|
assert kr_feeds["article"]["url"] == "https://36kr.com/feed-article"
|
|
assert kr_feeds["newsflash"]["url"] == "https://36kr.com/feed-newsflash"
|
|
assert kr_feeds["moment"]["url"] == "https://36kr.com/feed-moment"
|
|
assert all(feed["enabled"] is True for feed in kr_feeds.values())
|
|
assert all(feed["default_category"] == "business" for feed in kr_feeds.values())
|
|
assert "https://36kr.com/feed-article" in sources_by_id["36kr"]["feed_urls"]
|
|
assert "https://36kr.com/feed-newsflash" in sources_by_id["36kr"]["feed_urls"]
|
|
assert "https://36kr.com/feed-moment" in sources_by_id["36kr"]["feed_urls"]
|
|
assert sources_by_id["ebrun"]["source_type"] == "rss"
|
|
assert sources_by_id["ebrun"]["homepage_url"] == "https://www.ebrun.com/"
|
|
assert sources_by_id["ebrun"]["feed_directory_url"] == "https://www.ebrun.com/rss/"
|
|
ebrun_feeds = {feed["id"]: feed for feed in sources_by_id["ebrun"]["feeds"]}
|
|
assert {"b2c", "b2b", "retail", "o2o", "service", "data", "policy"}.issubset(ebrun_feeds)
|
|
assert all(feed["enabled"] is True for feed in ebrun_feeds.values())
|
|
assert all(feed["default_category"] == "ecommerce" for feed in ebrun_feeds.values())
|
|
assert "https://www.ebrun.com/rss/news_b2c.xml" in sources_by_id["ebrun"]["feed_urls"]
|
|
assert "https://www.ebrun.com/rss/news_retail.xml" in sources_by_id["ebrun"]["feed_urls"]
|
|
assert sources_by_id["businesswire-ecommerce"]["source_type"] == "reference"
|
|
assert sources_by_id["businesswire-ecommerce"]["enabled"] is False
|
|
assert sources_by_id["google-news"]["source_type"] == "aggregated"
|
|
assert sources_by_id["google-news"]["homepage_url"] == "https://news.google.com/"
|
|
assert sources_by_id["google-news"]["feed_directory_url"] == "https://news.google.com/rss"
|
|
google_feeds = {feed["id"]: feed for feed in sources_by_id["google-news"]["feeds"]}
|
|
assert set(google_feeds) == {"world", "americas", "europe", "middle-east-africa", "asia-pacific"}
|
|
assert all(feed["type"] == "aggregated" for feed in google_feeds.values())
|
|
assert all(feed["enabled"] is True for feed in google_feeds.values())
|
|
assert google_feeds["world"]["region"] == "global"
|
|
assert google_feeds["europe"]["region"] == "europe"
|
|
assert sources_by_id["stats-china-online-retail"]["source_type"] == "rss"
|
|
assert sources_by_id["stats-china-online-retail"]["enabled"] is True
|
|
assert "https://www.stats.gov.cn/sj/zxfb/rss.xml" in sources_by_id["stats-china-online-retail"]["feed_urls"]
|
|
assert {"business", "ecommerce", "finance"}.issubset(category_keys)
|
|
assert {"official_data", "business_news", "ecommerce", "press_release", "finance", "logistics"}.issubset(tag_keys)
|
|
|
|
|
|
def test_default_enabled_fetchable_sources_have_explicit_types_and_urls():
|
|
payload = default_earth_news_sources_payload()
|
|
for source in payload["sources"]:
|
|
source_type = source["source_type"]
|
|
assert source_type in {"rss", "atom", "aggregated", "reference"}
|
|
if source_type == "reference":
|
|
assert source["enabled"] is False
|
|
assert source["feeds"] == []
|
|
continue
|
|
if source["enabled"]:
|
|
assert source["feed_url"]
|
|
assert source["feed_urls"]
|
|
assert source["feeds"]
|
|
assert any(feed["enabled"] for feed in source["feeds"])
|
|
for feed in source["feeds"]:
|
|
assert feed["url"] != source["homepage_url"]
|
|
assert feed["url"] != source.get("feed_directory_url", "")
|
|
|
|
|
|
def test_legacy_news_source_urls_migrate_to_feed_children():
|
|
payload = normalize_earth_news_sources_payload(
|
|
{
|
|
"sources": [
|
|
{
|
|
"id": "legacy-source",
|
|
"name": "Legacy Source",
|
|
"region": "global",
|
|
"source_type": "rss",
|
|
"feed_urls": ["https://example.com/a.xml", "https://example.com/b.xml"],
|
|
"default_category": "business",
|
|
}
|
|
]
|
|
}
|
|
)
|
|
|
|
source = payload["sources"][0]
|
|
|
|
assert source["feed_urls"] == ["https://example.com/a.xml", "https://example.com/b.xml"]
|
|
assert [feed["url"] for feed in source["feeds"]] == ["https://example.com/a.xml", "https://example.com/b.xml"]
|
|
assert [feed["id"] for feed in source["feeds"]] == ["feed-1", "feed-2"]
|
|
assert all(feed["default_category"] == "business" for feed in source["feeds"])
|
|
|
|
|
|
def test_builtin_news_source_legacy_directory_url_is_repaired():
|
|
payload = normalize_earth_news_sources_payload(
|
|
{
|
|
"sources": [
|
|
{
|
|
"id": "36kr",
|
|
"name": "36氪",
|
|
"region": "asia-pacific",
|
|
"source_type": "rss",
|
|
"homepage_url": "https://www.36kr.com/",
|
|
"feed_url": "https://www.36kr.com/rss-center",
|
|
"feed_urls": ["https://www.36kr.com/rss-center"],
|
|
"feeds": [
|
|
{
|
|
"id": "feed-1",
|
|
"name": "36氪",
|
|
"url": "https://www.36kr.com/rss-center",
|
|
"type": "rss",
|
|
"enabled": True,
|
|
"default_category": "business",
|
|
}
|
|
],
|
|
"default_category": "business",
|
|
}
|
|
]
|
|
}
|
|
)
|
|
|
|
source = payload["sources"][0]
|
|
feed_urls = {feed["url"] for feed in source["feeds"]}
|
|
|
|
assert source["homepage_url"] == "https://www.36kr.com/"
|
|
assert source["feed_directory_url"] == "https://www.36kr.com/rss-center"
|
|
assert "https://www.36kr.com/rss-center" not in feed_urls
|
|
assert {
|
|
"https://36kr.com/feed",
|
|
"https://36kr.com/feed-article",
|
|
"https://36kr.com/feed-newsflash",
|
|
"https://36kr.com/feed-moment",
|
|
}.issubset(feed_urls)
|
|
|
|
|
|
def test_builtin_news_source_without_feed_children_gets_explicit_defaults():
|
|
payload = normalize_earth_news_sources_payload(
|
|
{
|
|
"sources": [
|
|
{
|
|
"id": "ebrun",
|
|
"name": "亿邦动力",
|
|
"region": "asia-pacific",
|
|
"source_type": "rss",
|
|
"homepage_url": "https://www.ebrun.com/",
|
|
"feed_url": "https://www.ebrun.com/rss/news_b2c.xml",
|
|
"feed_urls": ["https://www.ebrun.com/rss/news_b2c.xml"],
|
|
"default_category": "ecommerce",
|
|
}
|
|
]
|
|
}
|
|
)
|
|
|
|
source = payload["sources"][0]
|
|
feed_urls = {feed["url"] for feed in source["feeds"]}
|
|
|
|
assert source["feed_directory_url"] == "https://www.ebrun.com/rss/"
|
|
assert "https://www.ebrun.com/rss/" not in feed_urls
|
|
assert {
|
|
"https://www.ebrun.com/rss/news_b2c.xml",
|
|
"https://www.ebrun.com/rss/news_b2b.xml",
|
|
"https://www.ebrun.com/rss/news_retail.xml",
|
|
"https://www.ebrun.com/rss/news_o2o.xml",
|
|
"https://www.ebrun.com/rss/news_service.xml",
|
|
"https://www.ebrun.com/rss/news_data.xml",
|
|
"https://www.ebrun.com/rss/news_policy.xml",
|
|
}.issubset(feed_urls)
|
|
|
|
|
|
def test_builtin_fetchable_source_saved_as_reference_is_repaired():
|
|
payload = normalize_earth_news_sources_payload(
|
|
{
|
|
"sources": [
|
|
{
|
|
"id": "stats-china-online-retail",
|
|
"name": "国家统计局数据发布",
|
|
"region": "asia-pacific",
|
|
"source_type": "reference",
|
|
"enabled": False,
|
|
"homepage_url": "https://www.stats.gov.cn/sj/zxfb/",
|
|
"feed_url": "https://www.stats.gov.cn/sj/zxfb/",
|
|
"default_category": "ecommerce",
|
|
}
|
|
]
|
|
}
|
|
)
|
|
|
|
source = payload["sources"][0]
|
|
|
|
assert source["source_type"] == "rss"
|
|
assert source["enabled"] is True
|
|
assert source["priority"] == 19
|
|
assert source["source_tags"] == ["official_data", "ecommerce", "retail", "china"]
|
|
assert source["default_category"] == "ecommerce"
|
|
assert source["importance_weight"] == 36
|
|
assert source["feed_directory_url"] == ""
|
|
assert source["feeds"] == [
|
|
{
|
|
"id": "release",
|
|
"name": "数据发布",
|
|
"url": "https://www.stats.gov.cn/sj/zxfb/rss.xml",
|
|
"type": "rss",
|
|
"region": "asia-pacific",
|
|
"enabled": True,
|
|
"default_category": "ecommerce",
|
|
"tags": [],
|
|
"priority": 1,
|
|
}
|
|
]
|
|
|
|
|
|
def test_legacy_google_sources_merge_into_google_news_source():
|
|
payload = normalize_earth_news_sources_payload(
|
|
{
|
|
"sources": [
|
|
{
|
|
"id": "global-scan",
|
|
"name": "Global Monitor / World",
|
|
"region": "global",
|
|
"source_type": "aggregated",
|
|
"feed_url": "https://news.google.com/rss/search?q=world",
|
|
"homepage_url": "https://news.google.com/",
|
|
},
|
|
{
|
|
"id": "google-europe",
|
|
"name": "Global Monitor / Europe",
|
|
"region": "europe",
|
|
"source_type": "aggregated",
|
|
"feed_url": "https://news.google.com/rss/search?q=europe",
|
|
"homepage_url": "https://news.google.com/",
|
|
},
|
|
]
|
|
}
|
|
)
|
|
|
|
sources_by_id = {source["id"]: source for source in payload["sources"]}
|
|
|
|
assert "global-scan" not in sources_by_id
|
|
assert "google-europe" not in sources_by_id
|
|
assert "google-news" in sources_by_id
|
|
assert {feed["id"] for feed in sources_by_id["google-news"]["feeds"]} == {
|
|
"world",
|
|
"americas",
|
|
"europe",
|
|
"middle-east-africa",
|
|
"asia-pacific",
|
|
}
|
|
|
|
|
|
def test_feed_child_default_category_overrides_source_default():
|
|
source = NewsFeedSource(
|
|
id="multi-feed",
|
|
name="Multi Feed",
|
|
region="global",
|
|
feed_url="https://example.com/source.xml",
|
|
homepage_url="https://example.com",
|
|
default_category="business",
|
|
)
|
|
feed = NewsFeedEndpoint(
|
|
id="ecommerce-feed",
|
|
name="Ecommerce Feed",
|
|
url="https://example.com/ecommerce.xml",
|
|
default_category="ecommerce",
|
|
)
|
|
xml = """
|
|
<rss>
|
|
<channel>
|
|
<item>
|
|
<title>Quarterly results released</title>
|
|
<description>Company update.</description>
|
|
<link>https://example.com/results</link>
|
|
</item>
|
|
</channel>
|
|
</rss>
|
|
"""
|
|
|
|
items = _parse_feed_entries(xml, source, feed=feed)
|
|
|
|
assert items[0].feed_id == "ecommerce-feed"
|
|
assert items[0].feed_name == "Ecommerce Feed"
|
|
assert items[0].feed_default_category == "ecommerce"
|
|
assert items[0].category == "ecommerce"
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_fetch_source_only_requests_enabled_feed_children(monkeypatch):
|
|
source = NewsFeedSource(
|
|
id="multi-feed",
|
|
name="Multi Feed",
|
|
region="global",
|
|
feed_url="https://example.com/source.xml",
|
|
homepage_url="https://example.com",
|
|
feeds=(
|
|
NewsFeedEndpoint(id="enabled", name="Enabled", url="https://example.com/enabled.xml", enabled=True),
|
|
NewsFeedEndpoint(id="disabled", name="Disabled", url="https://example.com/disabled.xml", enabled=False),
|
|
),
|
|
)
|
|
calls = []
|
|
|
|
async def fake_fetch_single(_client, feed_source, feed, *, config_payload=None):
|
|
calls.append(feed.id)
|
|
item = ParsedNewsItem(
|
|
id=f"{feed_source.id}:{feed.id}:1",
|
|
title="Fetched story",
|
|
summary="Fetched summary",
|
|
url=f"https://example.com/{feed.id}",
|
|
source="Example",
|
|
feed_name=feed.name,
|
|
feed_region="global",
|
|
homepage_url="https://example.com",
|
|
published_at=None,
|
|
feed_id=feed.id,
|
|
)
|
|
return feed_source, [item], None, {"source_id": feed_source.id, "feed_id": feed.id, "ok": True, "status": "ok", "item_count": 1, "count": 1}
|
|
|
|
monkeypatch.setattr("app.services.earth_news._fetch_single_feed_url", fake_fetch_single)
|
|
|
|
source_result, items, error, health = await _fetch_source(object(), source)
|
|
|
|
assert source_result.id == "multi-feed"
|
|
assert calls == ["enabled"]
|
|
assert error is None
|
|
assert [item.feed_id for item in items] == ["enabled"]
|
|
assert health["ok"] is True
|
|
assert [result["feed_id"] for result in health["feed_results"]] == ["enabled"]
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_fetch_source_filters_google_feed_children_by_active_region(monkeypatch):
|
|
source = NewsFeedSource(
|
|
id="google-news",
|
|
name="Google News",
|
|
region="global",
|
|
feed_url="https://news.google.com/rss",
|
|
homepage_url="https://news.google.com/",
|
|
source_type="aggregated",
|
|
feeds=(
|
|
NewsFeedEndpoint(id="world", name="全球", url="https://example.com/world.xml", type="aggregated", region="global"),
|
|
NewsFeedEndpoint(id="europe", name="欧洲", url="https://example.com/europe.xml", type="aggregated", region="europe"),
|
|
NewsFeedEndpoint(id="americas", name="美洲", url="https://example.com/americas.xml", type="aggregated", region="americas"),
|
|
),
|
|
)
|
|
calls = []
|
|
|
|
async def fake_fetch_single(_client, feed_source, feed, *, config_payload=None):
|
|
calls.append(feed.id)
|
|
item = ParsedNewsItem(
|
|
id=f"{feed_source.id}:{feed.id}:1",
|
|
title=f"{feed.name} headline",
|
|
summary="Fetched summary",
|
|
url=f"https://example.com/{feed.id}",
|
|
source="Example",
|
|
feed_name=feed.name,
|
|
feed_region=feed.region,
|
|
homepage_url="https://example.com",
|
|
published_at=None,
|
|
feed_id=feed.id,
|
|
)
|
|
return feed_source, [item], None, {"source_id": feed_source.id, "feed_id": feed.id, "ok": True, "status": "ok", "item_count": 1, "count": 1}
|
|
|
|
monkeypatch.setattr("app.services.earth_news._fetch_single_feed_url", fake_fetch_single)
|
|
|
|
_source_result, items, error, health = await _fetch_source(object(), source, active_region="europe")
|
|
|
|
assert error is None
|
|
assert calls == ["world", "europe"]
|
|
assert [item.feed_region for item in items] == ["global", "europe"]
|
|
assert [result["feed_id"] for result in health["feed_results"]] == ["world", "europe"]
|
|
|
|
|
|
def test_parse_rdf_rss_items_with_namespaces():
|
|
source = NewsFeedSource(
|
|
id="dw-top",
|
|
name="DW Top Stories",
|
|
region="europe",
|
|
feed_url="https://rss.dw.com/rdf/rss-en-top",
|
|
homepage_url="https://www.dw.com/en/top-stories/s-9097",
|
|
)
|
|
xml = """
|
|
<rdf:RDF xmlns:rdf="http://www.w3.org/1999/02/22-rdf-syntax-ns#"
|
|
xmlns="http://purl.org/rss/1.0/">
|
|
<item rdf:about="https://example.com/dw">
|
|
<title>German retail sales rise</title>
|
|
<link>https://example.com/dw</link>
|
|
<description>Retail summary</description>
|
|
</item>
|
|
</rdf:RDF>
|
|
"""
|
|
|
|
items = _parse_feed_entries(xml, source)
|
|
|
|
assert len(items) == 1
|
|
assert items[0].title == "German retail sales rise"
|
|
|
|
|
|
def test_news_classification_marks_ecommerce_and_importance():
|
|
source = NewsFeedSource(
|
|
id="ebrun",
|
|
name="亿邦动力",
|
|
region="asia-pacific",
|
|
feed_url="https://www.ebrun.com/rss/",
|
|
homepage_url="https://www.ebrun.com/",
|
|
source_tags=("business_news", "ecommerce", "china"),
|
|
default_category="ecommerce",
|
|
importance_weight=14,
|
|
)
|
|
item = ParsedNewsItem(
|
|
id="ebrun:test",
|
|
title="跨境电商平台 GMV 同比增长,物流履约效率提升",
|
|
summary="订单量和网上零售额继续增长。",
|
|
url="https://example.com/ecommerce",
|
|
source="亿邦动力",
|
|
feed_name="亿邦动力",
|
|
feed_region="asia-pacific",
|
|
homepage_url="https://www.ebrun.com/",
|
|
published_at=None,
|
|
)
|
|
|
|
apply_news_classification(item, source)
|
|
|
|
assert item.category == "ecommerce"
|
|
assert "cross_border_ecommerce" in item.item_tags
|
|
assert "logistics_fulfillment" in item.item_tags
|
|
assert item.importance_level in {"high", "critical"}
|
|
assert "命中电商数据指标" in item.importance_reasons
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_enrich_items_with_target_locations_uses_ai_and_geocode(monkeypatch):
|
|
item = ParsedNewsItem(
|
|
id="bbc-world:f55310fb667b",
|
|
title="Watch: What happened on day one of Trump's China visit?",
|
|
summary="China welcomed US President Donald Trump before a long meeting with Xi Jinping.",
|
|
url="https://example.com/china-visit",
|
|
source="BBC World",
|
|
feed_name="BBC World",
|
|
feed_region="global",
|
|
homepage_url="https://example.com",
|
|
published_at=datetime(2026, 5, 14, 7, 13, 24, tzinfo=UTC),
|
|
)
|
|
|
|
async def fake_geocode(_query: str):
|
|
return {
|
|
"lat": "39.9042",
|
|
"lon": "116.4074",
|
|
"display_name": "Beijing, China",
|
|
}
|
|
|
|
class FakeProviderClient:
|
|
async def analyze(self, _request):
|
|
class Response:
|
|
content = (
|
|
'{"country":"China","city":"Beijing","matched_location_name":"Beijing, China",'
|
|
'"latitude":null,"longitude":null,"confidence":0.88}'
|
|
)
|
|
|
|
return Response()
|
|
|
|
monkeypatch.setattr("app.services.earth_news._geocode_target_location", fake_geocode)
|
|
|
|
enriched = await _enrich_items_with_target_locations(
|
|
[item],
|
|
provider_client=FakeProviderClient(),
|
|
)
|
|
|
|
assert len(enriched) == 1
|
|
assert enriched[0].target_location is not None
|
|
assert enriched[0].target_location.latitude == 39.9042
|
|
assert enriched[0].target_location.longitude == 116.4074
|
|
assert enriched[0].target_location.label == "Beijing, China"
|
|
assert enriched[0].target_resolution_stage == "ai_inferred_target"
|
|
assert enriched[0].target_ai_attempted is True
|
|
assert enriched[0].target_ai_status == "success"
|
|
assert enriched[0].target_ai_error is None
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_enrich_items_with_target_locations_adds_localizations(monkeypatch):
|
|
item = ParsedNewsItem(
|
|
id="global-scan:localized",
|
|
title="Global leaders meet to discuss energy security",
|
|
summary="Officials said the talks focused on supply chains and grid resilience.",
|
|
url="https://example.com/energy-security",
|
|
source="Example Source",
|
|
feed_name="Global Monitor / World",
|
|
feed_region="global",
|
|
homepage_url="https://example.com",
|
|
published_at=datetime(2026, 5, 14, 7, 13, 24, tzinfo=UTC),
|
|
)
|
|
|
|
async def fake_geocode(_query: str):
|
|
return {
|
|
"lat": "50.1109",
|
|
"lon": "8.6821",
|
|
"display_name": "Frankfurt am Main, Germany",
|
|
}
|
|
|
|
class FakeProviderClient:
|
|
async def analyze(self, _request):
|
|
class Response:
|
|
content = (
|
|
'{"location":{"country":"Germany","city":"Frankfurt",'
|
|
'"matched_location_name":"Frankfurt, Germany",'
|
|
'"latitude":null,"longitude":null,"confidence":0.77},'
|
|
'"localizations":{"zh-CN":{"title":"全球领导人讨论能源安全",'
|
|
'"summary":"官员表示,会谈聚焦供应链和电网韧性。"}}}'
|
|
)
|
|
|
|
return Response()
|
|
|
|
monkeypatch.setattr("app.services.earth_news._geocode_target_location", fake_geocode)
|
|
|
|
enriched = await _enrich_items_with_target_locations(
|
|
[item],
|
|
provider_client=FakeProviderClient(),
|
|
)
|
|
payload = _serialize_item(enriched[0], active_region="global")
|
|
|
|
assert payload["title"] == "Global leaders meet to discuss energy security"
|
|
assert payload["summary"] == "Officials said the talks focused on supply chains and grid resilience."
|
|
assert payload["localizations"]["zh-CN"]["title"] == "全球领导人讨论能源安全"
|
|
assert payload["display_title"] == "全球领导人讨论能源安全"
|
|
assert payload["display_summary"] == "官员表示,会谈聚焦供应链和电网韧性。"
|
|
assert payload["enrichment_status"] == "success"
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_extract_target_location_from_text_uses_country_hint(monkeypatch):
|
|
item = ParsedNewsItem(
|
|
id="bbc-world:country-hint",
|
|
title="Giant new dinosaur identified from fossils in Thailand",
|
|
summary="The nagatitan is the largest dinosaur found in South-East Asia.",
|
|
url="https://example.com/thailand-dinosaur",
|
|
source="BBC World",
|
|
feed_name="BBC World",
|
|
feed_region="global",
|
|
homepage_url="https://example.com",
|
|
published_at=datetime(2026, 5, 14, 17, 28, 56, tzinfo=UTC),
|
|
)
|
|
|
|
target = await _extract_target_location_from_text(item)
|
|
|
|
assert target is not None
|
|
assert target.country == "泰国"
|
|
assert target.latitude == 15.87
|
|
assert target.longitude == 100.9925
|
|
assert target.source == "headline_country_hint"
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_enrich_items_with_target_locations_records_ai_provider_error():
|
|
item = ParsedNewsItem(
|
|
id="global-scan:no-hint",
|
|
title="The New Geopolitics of Power: Whoever Controls Electrons Wins the Decade",
|
|
summary="A broad analysis of industrial policy and energy systems.",
|
|
url="https://example.com/geopolitics-power",
|
|
source="Example Source",
|
|
feed_name="Global Monitor / World",
|
|
feed_region="global",
|
|
homepage_url="https://example.com",
|
|
published_at=datetime(2026, 5, 14, 17, 3, 10, tzinfo=UTC),
|
|
)
|
|
|
|
class FailingProviderClient:
|
|
async def analyze(self, _request):
|
|
raise RuntimeError("upstream ai timeout")
|
|
|
|
enriched = await _enrich_items_with_target_locations(
|
|
[item],
|
|
provider_client=FailingProviderClient(),
|
|
)
|
|
|
|
assert len(enriched) == 1
|
|
assert enriched[0].target_location is None
|
|
assert enriched[0].target_resolution_stage == "unresolved"
|
|
assert enriched[0].target_ai_attempted is True
|
|
assert enriched[0].target_ai_status == "provider_error"
|
|
assert enriched[0].target_ai_error == "upstream ai timeout"
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_earth_news_payload_returns_anchor_items_and_enqueues_location_job(monkeypatch):
|
|
source = NewsFeedSource(
|
|
id="test-feed",
|
|
name="Test Feed",
|
|
region="global",
|
|
homepage_url="https://example.com",
|
|
feed_url="https://example.com/rss.xml",
|
|
)
|
|
item = ParsedNewsItem(
|
|
id="test-feed:timeout",
|
|
title="Example story",
|
|
summary="Example summary",
|
|
url="https://example.com/story",
|
|
source="Test Feed",
|
|
feed_name="Test Feed",
|
|
feed_region="global",
|
|
homepage_url="https://example.com",
|
|
published_at=datetime(2026, 5, 15, 3, 0, tzinfo=UTC),
|
|
)
|
|
|
|
async def fake_fetch_source(_client, feed_source, **_kwargs):
|
|
return feed_source, [item], None, {"source_id": feed_source.id, "ok": True, "status": "ok", "count": 1}
|
|
|
|
async def fake_get_cached_target_location_patch(_item_id):
|
|
return None
|
|
|
|
enqueued_payloads = []
|
|
|
|
async def fake_enqueue_target_location_job(payload, **_kwargs):
|
|
enqueued_payloads.append(payload)
|
|
return True
|
|
|
|
monkeypatch.setattr("app.services.earth_news.get_sources_for_region", lambda _region: [source])
|
|
monkeypatch.setattr("app.services.earth_news._fetch_source", fake_fetch_source)
|
|
monkeypatch.setattr(
|
|
"app.services.earth_news_queue.get_cached_target_location_patch",
|
|
fake_get_cached_target_location_patch,
|
|
)
|
|
monkeypatch.setattr(
|
|
"app.services.earth_news_queue.enqueue_target_location_job",
|
|
fake_enqueue_target_location_job,
|
|
)
|
|
|
|
payload = await get_earth_news_payload(provider_client=None)
|
|
|
|
assert len(payload["items"]) == 1
|
|
assert payload["items"][0]["id"] == "test-feed:timeout"
|
|
assert payload["items"][0]["display_title"] == ""
|
|
assert payload["items"][0]["display_summary"] == ""
|
|
assert payload["items"][0]["latitude"] == 20.0
|
|
assert payload["items"][0]["longitude"] == 0.0
|
|
assert payload["items"][0]["location_source"] == "region_anchor"
|
|
assert payload["items"][0]["verified"] is False
|
|
assert payload["items"][0]["location_meta"]["ai_status"] == "queued"
|
|
assert enqueued_payloads[0]["id"] == "test-feed:timeout"
|
|
assert payload["errors"] == []
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_earth_news_payload_uses_fresh_database_items_without_rss(monkeypatch):
|
|
db = object()
|
|
item = ParsedNewsItem(
|
|
id="db:fresh",
|
|
title="Fresh database story",
|
|
summary="Stored summary",
|
|
url="https://example.com/fresh",
|
|
source="Stored Source",
|
|
feed_name="Stored Feed",
|
|
feed_region="global",
|
|
homepage_url="https://example.com",
|
|
published_at=datetime(2026, 5, 15, 3, 0, tzinfo=UTC),
|
|
)
|
|
item.location_patch = {
|
|
"latitude": 39.9057136,
|
|
"longitude": 116.3912972,
|
|
"location_label": "北京市, 中国",
|
|
"location_source": "headline_location_hint",
|
|
"verified": True,
|
|
"location_meta": {"target": {"city": "Beijing"}, "anchor": {"region": "global"}},
|
|
}
|
|
|
|
async def fake_get_earth_news_freshness(_db, *, active_region):
|
|
return 12, datetime.now(UTC)
|
|
|
|
async def fake_list_earth_news_items(_db, *, active_region, limit, categories=None):
|
|
assert limit == 12
|
|
return [item]
|
|
|
|
async def fail_fetch(_sources):
|
|
raise AssertionError("fresh database items should not fetch RSS")
|
|
|
|
monkeypatch.setattr("app.services.earth_news_store.get_earth_news_freshness", fake_get_earth_news_freshness)
|
|
monkeypatch.setattr("app.services.earth_news_store.list_earth_news_items", fake_list_earth_news_items)
|
|
monkeypatch.setattr("app.services.earth_news._fetch_rss_items_for_sources", fail_fetch)
|
|
|
|
payload = await get_earth_news_payload(db=db)
|
|
|
|
assert payload["items"][0]["id"] == "db:fresh"
|
|
assert payload["items"][0]["verified"] is True
|
|
assert payload["items"][0]["latitude"] == 39.9057136
|
|
assert payload["stale"] is False
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_earth_news_payload_keeps_current_items_and_all_cruise_items(monkeypatch):
|
|
class FakeDb:
|
|
execute = object()
|
|
|
|
current_item = ParsedNewsItem(
|
|
id="db:current",
|
|
title="Current region story",
|
|
summary="Current summary",
|
|
url="https://example.com/current",
|
|
source="Stored Source",
|
|
feed_name="Stored Feed",
|
|
feed_region="americas",
|
|
homepage_url="https://example.com",
|
|
published_at=datetime(2026, 5, 15, 3, 0, tzinfo=UTC),
|
|
)
|
|
cruise_item = ParsedNewsItem(
|
|
id="db:apac",
|
|
title="APAC story",
|
|
summary="APAC summary",
|
|
url="https://example.com/apac",
|
|
source="Stored Source",
|
|
feed_name="Stored Feed",
|
|
feed_region="asia-pacific",
|
|
homepage_url="https://example.com",
|
|
published_at=datetime(2026, 5, 15, 4, 0, tzinfo=UTC),
|
|
)
|
|
|
|
async def fake_get_earth_news_freshness(_db, *, active_region):
|
|
return 12, datetime.now(UTC)
|
|
|
|
async def fake_list_earth_news_items(_db, *, active_region, limit, categories=None):
|
|
return [current_item]
|
|
|
|
async def fake_list_earth_news_cruise_items(_db, *, limit, categories=None):
|
|
return [current_item, cruise_item]
|
|
|
|
async def fake_enqueue_target_location_job(_payload, **_kwargs):
|
|
return True
|
|
|
|
async def fail_fetch(_sources):
|
|
raise AssertionError("fresh database items should not fetch RSS")
|
|
|
|
monkeypatch.setattr("app.services.earth_news_store.get_earth_news_freshness", fake_get_earth_news_freshness)
|
|
monkeypatch.setattr("app.services.earth_news_store.list_earth_news_items", fake_list_earth_news_items)
|
|
monkeypatch.setattr("app.services.earth_news_store.list_earth_news_cruise_items", fake_list_earth_news_cruise_items)
|
|
monkeypatch.setattr("app.services.earth_news_queue.enqueue_target_location_job", fake_enqueue_target_location_job)
|
|
monkeypatch.setattr("app.services.earth_news._fetch_rss_items_for_sources", fail_fetch)
|
|
|
|
payload = await get_earth_news_payload(lat=35.0, lon=-100.0, db=FakeDb())
|
|
|
|
assert [item["id"] for item in payload["items"]] == ["db:current"]
|
|
assert [item["id"] for item in payload["cruise_items"]] == ["db:current", "db:apac"]
|
|
assert payload["cruise_items"][1]["region"] == "asia-pacific"
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_earth_news_payload_passes_region_and_category_filters_to_store(monkeypatch):
|
|
class FakeDb:
|
|
execute = object()
|
|
|
|
captured = {}
|
|
item = ParsedNewsItem(
|
|
id="db:business",
|
|
title="Business story",
|
|
summary="Business summary",
|
|
url="https://example.com/business",
|
|
source="Stored Source",
|
|
feed_name="Stored Feed",
|
|
feed_region="europe",
|
|
homepage_url="https://example.com",
|
|
published_at=datetime(2026, 5, 15, 3, 0, tzinfo=UTC),
|
|
category="business",
|
|
)
|
|
|
|
async def fake_get_earth_news_freshness(_db, *, active_region):
|
|
captured["freshness_region"] = active_region
|
|
return 12, datetime.now(UTC)
|
|
|
|
async def fake_list_earth_news_items(_db, *, active_region, limit, categories=None, source_ids=None):
|
|
captured["items_region"] = active_region
|
|
captured["items_categories"] = categories
|
|
captured["items_source_ids"] = source_ids
|
|
return [item]
|
|
|
|
async def fake_list_earth_news_cruise_items(_db, *, limit, categories=None, source_ids=None):
|
|
captured["cruise_categories"] = categories
|
|
captured["cruise_source_ids"] = source_ids
|
|
return [item]
|
|
|
|
async def fake_enqueue_target_location_job(_payload, **_kwargs):
|
|
return True
|
|
|
|
monkeypatch.setattr("app.services.earth_news_store.get_earth_news_freshness", fake_get_earth_news_freshness)
|
|
monkeypatch.setattr("app.services.earth_news_store.list_earth_news_items", fake_list_earth_news_items)
|
|
monkeypatch.setattr("app.services.earth_news_store.list_earth_news_cruise_items", fake_list_earth_news_cruise_items)
|
|
monkeypatch.setattr("app.services.earth_news_queue.enqueue_target_location_job", fake_enqueue_target_location_job)
|
|
monkeypatch.setattr("app.services.earth_news._fetch_rss_items_for_sources", lambda _sources: (_ for _ in ()).throw(AssertionError("fresh database items should not fetch RSS")))
|
|
|
|
payload = await get_earth_news_payload(
|
|
lat=35.0,
|
|
lon=-100.0,
|
|
region="europe",
|
|
categories={"business", "ecommerce"},
|
|
db=FakeDb(),
|
|
)
|
|
|
|
assert captured["freshness_region"] == "europe"
|
|
assert captured["items_region"] == "europe"
|
|
assert captured["items_categories"] == {"business", "ecommerce"}
|
|
assert captured["items_source_ids"] is None
|
|
assert captured["cruise_categories"] == {"business", "ecommerce"}
|
|
assert captured["cruise_source_ids"] is None
|
|
assert payload["filters"] == {
|
|
"region": "europe",
|
|
"categories": ["business", "ecommerce"],
|
|
"sources": [],
|
|
"limit": 12,
|
|
"locale": "zh-CN",
|
|
"has_breaking": False,
|
|
"highest_breaking_level": "none",
|
|
}
|
|
assert payload["items"][0]["category"] == "business"
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_news_source_test_treats_type_reference_as_non_fetching():
|
|
result = await run_news_source_config_test(
|
|
{
|
|
"id": "reference-only",
|
|
"name": "Reference Only",
|
|
"type": "reference",
|
|
"feed_url": "https://example.com",
|
|
}
|
|
)
|
|
|
|
assert result["ok"] is False
|
|
assert result["health"]["status"] == "reference"
|
|
assert "不参与 RSS/Atom 抓取" in result["error"]
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_earth_news_payload_initializes_empty_database_from_rss(monkeypatch):
|
|
db = object()
|
|
source = NewsFeedSource(
|
|
id="test-feed",
|
|
name="Test Feed",
|
|
region="global",
|
|
homepage_url="https://example.com",
|
|
feed_url="https://example.com/rss.xml",
|
|
)
|
|
item = ParsedNewsItem(
|
|
id="test-feed:init",
|
|
title="Initial RSS story",
|
|
summary="Initial summary",
|
|
url="https://example.com/init",
|
|
source="Test Feed",
|
|
feed_name="Test Feed",
|
|
feed_region="global",
|
|
homepage_url="https://example.com",
|
|
published_at=datetime(2026, 5, 15, 3, 0, tzinfo=UTC),
|
|
)
|
|
item.location_patch = {
|
|
"latitude": 20.0,
|
|
"longitude": 0.0,
|
|
"location_label": "全球",
|
|
"location_source": "region_anchor",
|
|
"verified": False,
|
|
"location_meta": {"target": None, "anchor": {"region": "global"}},
|
|
}
|
|
upserted = []
|
|
enqueued = []
|
|
|
|
async def fake_get_earth_news_freshness(_db, *, active_region):
|
|
return 0, None
|
|
|
|
async def fake_fetch_rss_items_for_sources(_sources, **_kwargs):
|
|
return [item], [], {"test-feed": {"source_id": "test-feed", "ok": True, "status": "ok", "count": 1}}
|
|
|
|
async def fake_upsert_earth_news_items(_db, items):
|
|
upserted.extend(items)
|
|
return len(items)
|
|
|
|
async def fake_list_earth_news_items(_db, *, active_region, limit, categories=None):
|
|
return [item]
|
|
|
|
async def fake_enqueue_target_location_job(payload, **_kwargs):
|
|
enqueued.append(payload)
|
|
return True
|
|
|
|
monkeypatch.setattr("app.services.earth_news.get_sources_for_region", lambda _region: [source])
|
|
monkeypatch.setattr("app.services.earth_news_store.get_earth_news_freshness", fake_get_earth_news_freshness)
|
|
monkeypatch.setattr("app.services.earth_news._fetch_rss_items_for_sources", fake_fetch_rss_items_for_sources)
|
|
monkeypatch.setattr("app.services.earth_news_store.upsert_earth_news_items", fake_upsert_earth_news_items)
|
|
monkeypatch.setattr("app.services.earth_news_store.list_earth_news_items", fake_list_earth_news_items)
|
|
monkeypatch.setattr("app.services.earth_news_queue.enqueue_target_location_job", fake_enqueue_target_location_job)
|
|
|
|
payload = await get_earth_news_payload(db=db)
|
|
|
|
assert upserted[0].id == "test-feed:init"
|
|
assert payload["items"][0]["id"] == "test-feed:init"
|
|
assert payload["items"][0]["verified"] is False
|
|
assert enqueued[0]["id"] == "test-feed:init"
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_earth_news_payload_supplements_stale_database_items(monkeypatch):
|
|
db = object()
|
|
old_item = ParsedNewsItem(
|
|
id="db:old",
|
|
title="Old story",
|
|
summary="Old summary",
|
|
url="https://example.com/old",
|
|
source="Stored Source",
|
|
feed_name="Stored Feed",
|
|
feed_region="global",
|
|
homepage_url="https://example.com",
|
|
published_at=datetime(2026, 5, 14, 3, 0, tzinfo=UTC),
|
|
)
|
|
old_item.location_patch = {
|
|
"latitude": 20.0,
|
|
"longitude": 0.0,
|
|
"location_label": "全球",
|
|
"location_source": "region_anchor",
|
|
"verified": False,
|
|
"location_meta": {"target": None, "anchor": {"region": "global"}},
|
|
}
|
|
fetched = []
|
|
|
|
async def fake_get_earth_news_freshness(_db, *, active_region):
|
|
return 12, datetime(2026, 5, 14, 3, 0, tzinfo=UTC)
|
|
|
|
async def fake_fetch_rss_items_for_sources(_sources, **_kwargs):
|
|
fetched.append(True)
|
|
return [old_item], [], {"stored": {"source_id": "stored", "ok": True, "status": "ok", "count": 1}}
|
|
|
|
async def fake_upsert_earth_news_items(_db, items):
|
|
return len(items)
|
|
|
|
async def fake_list_earth_news_items(_db, *, active_region, limit, categories=None):
|
|
return [old_item]
|
|
|
|
async def fake_enqueue_target_location_job(_payload, **_kwargs):
|
|
return True
|
|
|
|
monkeypatch.setattr("app.services.earth_news_store.get_earth_news_freshness", fake_get_earth_news_freshness)
|
|
monkeypatch.setattr("app.services.earth_news._fetch_rss_items_for_sources", fake_fetch_rss_items_for_sources)
|
|
monkeypatch.setattr("app.services.earth_news_store.upsert_earth_news_items", fake_upsert_earth_news_items)
|
|
monkeypatch.setattr("app.services.earth_news_store.list_earth_news_items", fake_list_earth_news_items)
|
|
monkeypatch.setattr("app.services.earth_news_queue.enqueue_target_location_job", fake_enqueue_target_location_job)
|
|
|
|
payload = await get_earth_news_payload(db=db)
|
|
|
|
assert fetched == [True]
|
|
assert payload["items"][0]["id"] == "db:old"
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_earth_news_payload_merges_cached_location_patch(monkeypatch):
|
|
source = NewsFeedSource(
|
|
id="test-feed",
|
|
name="Test Feed",
|
|
region="global",
|
|
homepage_url="https://example.com",
|
|
feed_url="https://example.com/rss.xml",
|
|
)
|
|
item = ParsedNewsItem(
|
|
id="test-feed:cached",
|
|
title="Cached story",
|
|
summary="Cached summary",
|
|
url="https://example.com/cached",
|
|
source="Test Feed",
|
|
feed_name="Test Feed",
|
|
feed_region="global",
|
|
homepage_url="https://example.com",
|
|
published_at=datetime(2026, 5, 15, 3, 0, tzinfo=UTC),
|
|
)
|
|
cached_patch = {
|
|
"latitude": 39.9057136,
|
|
"longitude": 116.3912972,
|
|
"location_label": "北京市, 中国",
|
|
"location_source": "headline_location_hint",
|
|
"verified": True,
|
|
"location_meta": {
|
|
"resolution_stage": "headline_location_hint",
|
|
"ai_attempted": False,
|
|
"ai_status": "skipped_text_hint",
|
|
"ai_error": None,
|
|
"debug_note": "text hint matched 北京市, 中国",
|
|
"target": {"city": "Beijing"},
|
|
"anchor": {"region": "global"},
|
|
},
|
|
}
|
|
|
|
async def fake_fetch_source(_client, feed_source, **_kwargs):
|
|
return feed_source, [item], None, {"source_id": feed_source.id, "ok": True, "status": "ok", "count": 1}
|
|
|
|
async def fake_get_cached_target_location_patch(_item_id):
|
|
return cached_patch
|
|
|
|
enqueued = []
|
|
|
|
async def fake_enqueue_target_location_job(payload, **_kwargs):
|
|
enqueued.append(payload)
|
|
return True
|
|
|
|
monkeypatch.setattr("app.services.earth_news.get_sources_for_region", lambda _region: [source])
|
|
monkeypatch.setattr("app.services.earth_news._fetch_source", fake_fetch_source)
|
|
monkeypatch.setattr(
|
|
"app.services.earth_news_queue.get_cached_target_location_patch",
|
|
fake_get_cached_target_location_patch,
|
|
)
|
|
monkeypatch.setattr(
|
|
"app.services.earth_news_queue.enqueue_target_location_job",
|
|
fake_enqueue_target_location_job,
|
|
)
|
|
|
|
payload = await get_earth_news_payload(provider_client=None)
|
|
|
|
assert payload["items"][0]["latitude"] == 39.9057136
|
|
assert payload["items"][0]["longitude"] == 116.3912972
|
|
assert payload["items"][0]["verified"] is True
|
|
assert payload["items"][0]["location_source"] == "headline_location_hint"
|
|
assert enqueued[0]["id"] == "test-feed:cached"
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_earth_news_payload_requeues_cached_failed_localization(monkeypatch):
|
|
source = NewsFeedSource(
|
|
id="test-feed",
|
|
name="Test Feed",
|
|
region="global",
|
|
homepage_url="https://example.com",
|
|
feed_url="https://example.com/rss.xml",
|
|
)
|
|
item = ParsedNewsItem(
|
|
id="test-feed:failed-localization",
|
|
title="Failed localization story",
|
|
summary="English source summary.",
|
|
url="https://example.com/failed-localization",
|
|
source="Test Feed",
|
|
feed_name="Test Feed",
|
|
feed_region="global",
|
|
homepage_url="https://example.com",
|
|
published_at=datetime(2026, 5, 15, 3, 0, tzinfo=UTC),
|
|
)
|
|
cached_patch = {
|
|
"latitude": 20.0,
|
|
"longitude": 0.0,
|
|
"location_label": "全球",
|
|
"location_source": "region_anchor",
|
|
"verified": False,
|
|
"location_meta": {"target": None, "anchor": {"region": "global"}},
|
|
"content_language": "en",
|
|
"localizations": {},
|
|
"enrichment_status": "parse_error",
|
|
"enrichment_error": "AI response did not contain a parseable JSON object.",
|
|
"enriched_at": None,
|
|
}
|
|
enqueued = []
|
|
|
|
async def fake_fetch_source(_client, feed_source, **_kwargs):
|
|
return feed_source, [item], None, {"source_id": feed_source.id, "ok": True, "status": "ok", "count": 1}
|
|
|
|
async def fake_get_cached_target_location_patch(_item_id):
|
|
return cached_patch
|
|
|
|
async def fake_enqueue_target_location_job(payload, **_kwargs):
|
|
enqueued.append(payload)
|
|
return True
|
|
|
|
monkeypatch.setattr("app.services.earth_news.get_sources_for_region", lambda _region: [source])
|
|
monkeypatch.setattr("app.services.earth_news._fetch_source", fake_fetch_source)
|
|
monkeypatch.setattr(
|
|
"app.services.earth_news_queue.get_cached_target_location_patch",
|
|
fake_get_cached_target_location_patch,
|
|
)
|
|
monkeypatch.setattr(
|
|
"app.services.earth_news_queue.enqueue_target_location_job",
|
|
fake_enqueue_target_location_job,
|
|
)
|
|
|
|
payload = await get_earth_news_payload(provider_client=None)
|
|
|
|
assert enqueued[0]["id"] == "test-feed:failed-localization"
|
|
assert payload["items"][0]["display_title"] == ""
|
|
assert payload["items"][0]["enrichment_status"] == "queued"
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_worker_processes_target_location_message_and_returns_patch(monkeypatch):
|
|
message = NewsTargetLocationMessage(
|
|
message_id="1-0",
|
|
item_id="bbc-world:worker",
|
|
payload={
|
|
"id": "bbc-world:worker",
|
|
"title": "Ukraine rescuers pull dead from rubble of Kyiv flats",
|
|
"summary": "Massive Russian drone and missile attacks in Ukraine's capital.",
|
|
"url": "https://example.com/kyiv",
|
|
"source": "BBC World",
|
|
"feed_name": "BBC World",
|
|
"feed_region": "global",
|
|
"homepage_url": "https://www.bbc.com/news/world",
|
|
"published_at": "2026-05-14T13:16:32Z",
|
|
},
|
|
)
|
|
|
|
async def fake_geocode(_query: str):
|
|
return {
|
|
"lat": "50.4500336",
|
|
"lon": "30.5241361",
|
|
"display_name": "Київ, Україна",
|
|
}
|
|
|
|
saved = {}
|
|
broadcasted = {}
|
|
|
|
async def fake_save_target_location_patch(item_id, patch):
|
|
saved["item_id"] = item_id
|
|
saved["patch"] = patch
|
|
|
|
async def fake_update_earth_news_item_location(_session, *, item_id, patch):
|
|
saved["db_item_id"] = item_id
|
|
saved["db_patch"] = patch
|
|
return True
|
|
|
|
async def fake_broadcast_custom(channel, data):
|
|
broadcasted["channel"] = channel
|
|
broadcasted["data"] = data
|
|
|
|
class FakeSession:
|
|
async def commit(self):
|
|
saved["committed"] = True
|
|
|
|
class FakeSessionFactory:
|
|
async def __aenter__(self):
|
|
return FakeSession()
|
|
|
|
async def __aexit__(self, exc_type, exc, tb):
|
|
return False
|
|
|
|
monkeypatch.setattr("app.services.earth_news._geocode_target_location", fake_geocode)
|
|
monkeypatch.setattr(
|
|
"app.services.earth_news_worker.save_target_location_patch",
|
|
fake_save_target_location_patch,
|
|
)
|
|
monkeypatch.setattr(
|
|
"app.services.earth_news_worker.update_earth_news_item_location",
|
|
fake_update_earth_news_item_location,
|
|
)
|
|
monkeypatch.setattr(
|
|
"app.services.earth_news_worker.async_session_factory",
|
|
lambda: FakeSessionFactory(),
|
|
)
|
|
monkeypatch.setattr(
|
|
"app.services.earth_news_worker.broadcaster.broadcast_custom",
|
|
fake_broadcast_custom,
|
|
)
|
|
|
|
patch = await process_target_location_message(message, provider_client=None)
|
|
|
|
assert patch["latitude"] == 50.4500336
|
|
assert patch["longitude"] == 30.5241361
|
|
assert patch["location_source"] == "headline_location_hint"
|
|
assert patch["verified"] is True
|
|
assert saved["item_id"] == "bbc-world:worker"
|
|
assert saved["db_item_id"] == "bbc-world:worker"
|
|
assert saved["committed"] is True
|
|
assert broadcasted["channel"] == "earth_news"
|
|
assert broadcasted["data"]["item_id"] == "bbc-world:worker"
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_media_news_archive_collector_maps_news_items(monkeypatch):
|
|
collector = MediaNewsArchiveCollector()
|
|
collector._db_session = object()
|
|
record = SimpleNamespace(
|
|
id="bbc-world:archive",
|
|
title="Archived news",
|
|
summary="Archived summary",
|
|
url="https://example.com/archive",
|
|
source="BBC World",
|
|
feed_name="BBC World",
|
|
region="global",
|
|
homepage_url="https://www.bbc.com/news/world",
|
|
published_at=datetime(2026, 5, 15, 3, 0, tzinfo=UTC),
|
|
latitude=39.9057136,
|
|
longitude=116.3912972,
|
|
location_label="北京市, 中国",
|
|
location_source="headline_location_hint",
|
|
verified=True,
|
|
location_meta={"target": {"country": "中国", "city": "Beijing"}},
|
|
content_language="en",
|
|
localizations={"zh-CN": {"title": "归档新闻", "summary": "归档概要"}},
|
|
enrichment_status="success",
|
|
enrichment_error=None,
|
|
enriched_at=datetime(2026, 5, 15, 3, 6, tzinfo=UTC),
|
|
first_seen_at=datetime(2026, 5, 15, 2, 0, tzinfo=UTC),
|
|
last_seen_at=datetime(2026, 5, 15, 3, 0, tzinfo=UTC),
|
|
resolved_at=datetime(2026, 5, 15, 3, 5, tzinfo=UTC),
|
|
)
|
|
|
|
async def fake_list_all_earth_news_records(_db):
|
|
return [record]
|
|
|
|
monkeypatch.setattr(
|
|
"app.services.collectors.media_news_archive.list_all_earth_news_records",
|
|
fake_list_all_earth_news_records,
|
|
)
|
|
|
|
items = await collector.fetch()
|
|
|
|
assert items[0]["source_id"] == "bbc-world:archive"
|
|
assert collector.data_type == "news_item"
|
|
assert items[0]["country"] == "中国"
|
|
assert items[0]["city"] == "Beijing"
|
|
assert items[0]["latitude"] == 39.9057136
|
|
assert items[0]["metadata"]["verified"] is True
|
|
assert "localizations" not in items[0]["metadata"]
|
|
assert "enrichment_status" not in items[0]["metadata"]
|