55 lines
1.8 KiB
Python
55 lines
1.8 KiB
Python
"""Unit tests for candidate parsing without regex covering scenarios PAR-001 to PAR-010."""
|
|
|
|
from src.runtime.candidate.parser import (
|
|
parse_metadata_candidates,
|
|
parse_raw_text_into_candidates,
|
|
resolve_canonical_source_url,
|
|
)
|
|
|
|
|
|
def test_parse_raw_text_into_candidates():
|
|
markdown_text = """# Main Header
|
|
|
|
This is the first paragraph of the article.
|
|
|
|
## Subheader
|
|
|
|
Here is a second paragraph.
|
|
|
|
* Bullet one
|
|
* Bullet two
|
|
|
|
> A notable quote from an expert.
|
|
"""
|
|
candidates = parse_raw_text_into_candidates(markdown_text, extractor="trafilatura")
|
|
types = [c.type for c in candidates]
|
|
assert "heading" in types
|
|
assert "paragraph" in types
|
|
assert "list_item" in types
|
|
assert "quote" in types
|
|
|
|
|
|
def test_resolve_canonical_source_url_priority():
|
|
article_full = {
|
|
"crawled_url": "https://example.com/crawled",
|
|
"input_meta": {"url": "https://example.com/meta"},
|
|
"trafilatura": {"canonical_url": "https://example.com/canonical"},
|
|
}
|
|
# trafilatura canonical_url has top priority
|
|
assert resolve_canonical_source_url(article_full) == "https://example.com/canonical"
|
|
|
|
# fallback to input_meta.url
|
|
article_no_traf = {
|
|
"crawled_url": "https://example.com/crawled",
|
|
"input_meta": {"url": "https://example.com/meta"},
|
|
}
|
|
assert resolve_canonical_source_url(article_no_traf) == "https://example.com/meta"
|
|
|
|
|
|
def test_author_parsing_forbids_delimiter_splitting():
|
|
article = {"trafilatura": {"author": "Dr. Silva, Ph.D. / Reporter Especial"}}
|
|
cand = parse_metadata_candidates(article)
|
|
# The full string must be preserved as a single author candidate, not split by commas or slashes
|
|
assert len(cand["author_candidates"]) == 1
|
|
assert cand["author_candidates"][0]["text"] == "Dr. Silva, Ph.D. / Reporter Especial"
|