Преглед изворни кода

Add SSR/embedded-JSON extraction and restore Forbes

services/ssr.py walks JSON-LD and Next.js __NEXT_DATA__ for article-like
objects (headline, or title+url, or name with article signals), recovering
JS-rendered sites without a browser. prepare_headlines favors embedded JSON
over regex link scraping. Forbes is re-added as a live proof.
master
Jared Bell пре 3 дана
родитељ
комит
ad03cd96ce
5 измењених фајлова са 220 додато и 9 уклоњено
  1. +9
    -3
      README.md
  2. +2
    -1
      resources/sources.txt
  3. +17
    -5
      services/headlines.py
  4. +125
    -0
      services/ssr.py
  5. +67
    -0
      tests/test_ssr.py

+ 9
- 3
README.md Прегледај датотеку

@@ -10,9 +10,10 @@ question: *which stories are multiple independent sources reporting right now?*
## How it works ## How it works


1. **Load** source URLs and stopwords from `resources/`. 1. **Load** source URLs and stopwords from `resources/`.
2. **Fetch** each source. RSS/Atom feeds are parsed as structured XML (titles and
publication timestamps); HTML pages are scraped for `<a>`/`<span>` text,
associating each headline with the nearest `<time>` publication timestamp.
2. **Fetch** each source. RSS/Atom feeds are parsed as structured XML; HTML pages
are first checked for embedded JSON (JSON-LD / Next.js SSR state) before
falling back to scraping `<a>`/`<span>` text, associating each headline with
the nearest `<time>` publication timestamp.
3. **Normalize** each headline (lowercase, strip punctuation, remove stopwords). 3. **Normalize** each headline (lowercase, strip punctuation, remove stopwords).
4. **Cluster** headlines into stories using pairwise cosine similarity, linking 4. **Cluster** headlines into stories using pairwise cosine similarity, linking
matches transitively so a chain of near-duplicates collapses into one story. matches transitively so a chain of near-duplicates collapses into one story.
@@ -98,6 +99,10 @@ News stories covered by at least 2 distinct sources (published on or after 2026-
more reliable than scraping JavaScript-heavy or paywalled pages, and they carry more reliable than scraping JavaScript-heavy or paywalled pages, and they carry
publication timestamps directly. Feed type is auto-detected from the content, so publication timestamps directly. Feed type is auto-detected from the content, so
HTML and feed URLs can live side by side in `sources.txt`. HTML and feed URLs can live side by side in `sources.txt`.
- **JS-rendered pages are recovered without a browser.** For HTML that embeds its
data as JSON — JSON-LD structured data or Next.js `__NEXT_DATA__` SSR state —
Anya parses that directly and never executes JavaScript. This is far lighter than
a headless browser, at the cost of some per-site variation in the JSON shape.
- **Date windowing drops undated headlines.** When `--since` or `--until` is set, - **Date windowing drops undated headlines.** When `--since` or `--until` is set,
any headline whose page carries no parseable timestamp is excluded because its any headline whose page carries no parseable timestamp is excluded because its
recency can't be established (the count is logged). Without a date flag, recency can't be established (the count is logged). Without a date flag,
@@ -126,6 +131,7 @@ anya/
├── services/ ├── services/
│ ├── headlines.py # fetch + parse headlines (and timestamps) │ ├── headlines.py # fetch + parse headlines (and timestamps)
│ ├── feeds.py # RSS/Atom feed detection and parsing │ ├── feeds.py # RSS/Atom feed detection and parsing
│ ├── ssr.py # JSON-LD / Next.js embedded-JSON extraction
│ ├── normalization.py # stopword/phrase loading and headline normalization │ ├── normalization.py # stopword/phrase loading and headline normalization
│ ├── similarity.py # cosine similarity over token lists │ ├── similarity.py # cosine similarity over token lists
│ ├── sources.py # load source URLs │ ├── sources.py # load source URLs


+ 2
- 1
resources/sources.txt Прегледај датотеку

@@ -37,4 +37,5 @@ https://thehill.com/feed/
https://www.vox.com/rss/index.xml https://www.vox.com/rss/index.xml


# Business # Business
https://www.cnbc.com/id/100003114/device/rss/rss.html
https://www.cnbc.com/id/100003114/device/rss/rss.html
https://www.forbes.com/business/

+ 17
- 5
services/headlines.py Прегледај датотеку

@@ -6,6 +6,7 @@ import requests
from services.dates import parse_datetime from services.dates import parse_datetime
from services.feeds import looks_like_feed, parse_feed from services.feeds import looks_like_feed, parse_feed
from services.normalization import is_excluded, normalize_headline from services.normalization import is_excluded, normalize_headline
from services.ssr import extract_headlines
from structs.headline import Headline from structs.headline import Headline


logger = logging.getLogger(__name__) logger = logging.getLogger(__name__)
@@ -87,10 +88,10 @@ def _extract_candidates(source_content):
return candidates return candidates




def _collect_feed_headlines(source_content, source_url, stopwords, excluded_phrases):
"""Turn parsed feed entries into headline objects, applying normal filters."""
def _build_headlines(items, source_url, stopwords, excluded_phrases):
"""Turn ``(title, published_at)`` tuples into headline objects, applying filters."""
collected = [] collected = []
for title, published_at in parse_feed(source_content):
for title, published_at in items:
cleaned_text = title.strip() cleaned_text = title.strip()
if not cleaned_text or not is_headline(cleaned_text, stopwords, excluded_phrases): if not cleaned_text or not is_headline(cleaned_text, stopwords, excluded_phrases):
continue continue
@@ -98,7 +99,7 @@ def _collect_feed_headlines(source_content, source_url, stopwords, excluded_phra
normalized = normalize_headline(cleaned_text, stopwords) normalized = normalize_headline(cleaned_text, stopwords)
collected.append(Headline(cleaned_text, normalized, source_url, published_at)) collected.append(Headline(cleaned_text, normalized, source_url, published_at))
except Exception as e: except Exception as e:
logger.error("Failed to normalize feed title %r from '%s': %s", cleaned_text, source_url, e, exc_info=True)
logger.error("Failed to normalize title %r from '%s': %s", cleaned_text, source_url, e, exc_info=True)
return collected return collected




@@ -161,7 +162,7 @@ def prepare_headlines(sources, stopwords, timeout=DEFAULT_TIMEOUT, headers=None,
if looks_like_feed(source_content): if looks_like_feed(source_content):
logger.info("Detected RSS/Atom feed for source '%s'", source_url) logger.info("Detected RSS/Atom feed for source '%s'", source_url)
try: try:
feed_headlines = _collect_feed_headlines(source_content, source_url, stopwords, excluded_phrases)
feed_headlines = _build_headlines(parse_feed(source_content), source_url, stopwords, excluded_phrases)
except Exception as e: except Exception as e:
logger.error("Failed to parse feed from '%s': %s", source_url, e, exc_info=True) logger.error("Failed to parse feed from '%s': %s", source_url, e, exc_info=True)
continue continue
@@ -169,6 +170,17 @@ def prepare_headlines(sources, stopwords, timeout=DEFAULT_TIMEOUT, headers=None,
logger.info("Successfully extracted %d headlines from source '%s'", len(feed_headlines), source_url) logger.info("Successfully extracted %d headlines from source '%s'", len(feed_headlines), source_url)
continue continue


try:
ssr_items = extract_headlines(source_content)
except Exception as e:
logger.error("Failed to extract embedded JSON from '%s': %s", source_url, e, exc_info=True)
ssr_items = []
if ssr_items:
ssr_headlines = _build_headlines(ssr_items, source_url, stopwords, excluded_phrases)
headlines.extend(ssr_headlines)
logger.info("Successfully extracted %d headlines via embedded JSON from source '%s'", len(ssr_headlines), source_url)
continue

logger.debug("Parsing HTML content from '%s' for headline candidates", source_url) logger.debug("Parsing HTML content from '%s' for headline candidates", source_url)
try: try:
candidates = _extract_candidates(source_content) candidates = _extract_candidates(source_content)


+ 125
- 0
services/ssr.py Прегледај датотеку

@@ -0,0 +1,125 @@
import html
import json
import logging
import re

from services.dates import parse_datetime

logger = logging.getLogger(__name__)

# JSON-LD structured data blocks (schema.org) that most news sites emit for SEO —
# and Next.js's serialized page state. In both cases the headlines (and dates)
# are already present in the HTML, so no JavaScript execution is required.
_JSON_LD_RE = re.compile(
r"""<script\b[^>]*?\btype=(["'])application/ld\+json\1[^>]*>(.*?)</script>""",
re.IGNORECASE | re.DOTALL,
)
_NEXT_DATA_RE = re.compile(
r"""<script\b[^>]*?\bid=(["'])__NEXT_DATA__\1[^>]*>(.*?)</script>""",
re.IGNORECASE | re.DOTALL,
)

_ARTICLE_TYPES = {
'newsarticle', 'article', 'report', 'analysisnewsarticle',
'opinionnewsarticle', 'reviewnewsarticle', 'blogposting',
'liveblogposting', 'backgroundnewsarticle', 'reportagenewsarticle',
}

_URL_KEYS = ('url', 'href', 'link', 'canonicalUrl', 'slug', 'uri')


def _has_url(obj):
for key in _URL_KEYS:
value = obj.get(key)
if isinstance(value, str) and value.strip():
return True
return False


def _schema_type(obj):
"""Last path segment of an object's ``@type``, lowercased (handles URLs)."""
if not isinstance(obj, dict):
return ''
t = obj.get('@type')
if isinstance(t, list):
t = t[0] if t else None
if not isinstance(t, str):
return ''
return t.strip().rstrip('/').rsplit('/', 1)[-1].lower()


def _iter_json_scripts(content):
"""Yield unescaped JSON bodies of JSON-LD and Next.js SSR ``<script>`` blocks."""
if not content or not isinstance(content, str):
return
for pattern in (_JSON_LD_RE, _NEXT_DATA_RE):
for match in pattern.finditer(content):
body = match.group(2)
if body and body.strip():
yield html.unescape(body.strip())


def _collect(obj, out):
"""Recursively append ``(headline, published_at)`` for article-like objects.

Headlines are recognized by three signals:
* ``headline`` — article-specific in schema.org, trusted directly.
* ``title`` + a url-ish sibling — the common Next.js SSR shape, where
``title`` alone is too ambiguous (section labels use it too).
* ``name`` — trusted only with an article-type/date/author signal.
"""
if isinstance(obj, dict):
stype = _schema_type(obj)
headline = None

raw = obj.get('headline')
if isinstance(raw, str) and raw.strip():
headline = raw.strip()
else:
raw_title = obj.get('title')
if isinstance(raw_title, str) and raw_title.strip() and _has_url(obj):
headline = raw_title.strip()
else:
raw_name = obj.get('name')
if isinstance(raw_name, str) and raw_name.strip() and (
stype in _ARTICLE_TYPES or obj.get('datePublished') or obj.get('author')
):
headline = raw_name.strip()

if headline:
published = obj.get('datePublished') or obj.get('dateModified') or obj.get('date')
published_at = parse_datetime(published) if isinstance(published, str) else None
out.append((headline, published_at))

for value in obj.values():
_collect(value, out)
elif isinstance(obj, list):
for value in obj:
_collect(value, out)


def extract_headlines(content):
"""Best-effort extraction of ``(headline, published_at)`` from embedded JSON.

Walks JSON-LD and Next.js SSR data for article-like objects and returns a
de-duplicated list of tuples, where ``published_at`` is a ``datetime`` or
``None``.
"""
results = []
seen = set()
for body in _iter_json_scripts(content):
try:
data = json.loads(body)
except (json.JSONDecodeError, ValueError, TypeError):
continue
collected = []
_collect(data, collected)
for headline, published in collected:
key = headline.casefold()
if key in seen:
continue
seen.add(key)
results.append((headline, published))

logger.info("Extracted %d headline(s) from embedded JSON", len(results))
return results

+ 67
- 0
tests/test_ssr.py Прегледај датотеку

@@ -0,0 +1,67 @@
import unittest
from unittest.mock import patch, MagicMock

from services.ssr import extract_headlines
from services.headlines import prepare_headlines


HTML = """<html><head>
<script type="application/ld+json">
{
"@context": "https://schema.org",
"@graph": [
{"@type": "NewsArticle", "headline": "Court rules on landmark case", "datePublished": "2026-09-14T12:00:00Z"},
{"@type": "Organization", "name": "Example News"}
]
}
</script>
</head><body>
<script id="__NEXT_DATA__" type="application/json">
{"props": {"pageProps": {"channel": {"items": [
{"title": "Senate passes infrastructure bill", "url": "https://example.com/a/1"},
{"title": "Markets", "url": "https://example.com/markets"}
]}}}}
</script>
</body></html>"""


class TestExtractHeadlines(unittest.TestCase):
def test_json_ld_and_next_data(self):
items = extract_headlines(HTML)
headlines = [h for h, _ in items]

# schema.org headline and Next.js title+url are both recovered.
self.assertIn("Court rules on landmark case", headlines)
self.assertIn("Senate passes infrastructure bill", headlines)

# Organization `name` without an article signal is not treated as a headline.
self.assertNotIn("Example News", headlines)

# Date is parsed from JSON-LD datePublished.
by_title = {h: d for h, d in items}
self.assertIsNotNone(by_title["Court rules on landmark case"])


class TestPrepareHeadlinesSSR(unittest.TestCase):
def setUp(self):
self.stopwords = {"the", "a", "an", "in", "on", "and", "of", "to"}

@patch("services.headlines.requests.get")
def test_ssr_source_is_parsed(self, mock_get):
mock_response = MagicMock()
mock_response.status_code = 200
mock_response.content = HTML.encode("utf-8")
mock_response.text = HTML
mock_get.return_value = mock_response

headlines = prepare_headlines(["https://www.example.com/business/"], self.stopwords)

titles = [h.display_text for h in headlines]
# "Markets" is filtered out by the minimum-word headline check downstream.
self.assertIn("Court rules on landmark case", titles)
self.assertIn("Senate passes infrastructure bill", titles)
self.assertNotIn("Markets", titles)


if __name__ == "__main__":
unittest.main()

Loading…
Откажи
Сачувај