Files
TextNLPClassifierApp/tests/runtime/unit/test_candidate_parser.py
T

55 lines
1.8 KiB
Python

"""Unit tests for candidate parsing without regex covering scenarios PAR-001 to PAR-010."""
from src.runtime.candidate.parser import (
parse_metadata_candidates,
parse_raw_text_into_candidates,
resolve_canonical_source_url,
)
def test_parse_raw_text_into_candidates():
markdown_text = """# Main Header
This is the first paragraph of the article.
## Subheader
Here is a second paragraph.
* Bullet one
* Bullet two
> A notable quote from an expert.
"""
candidates = parse_raw_text_into_candidates(markdown_text, extractor="trafilatura")
types = [c.type for c in candidates]
assert "heading" in types
assert "paragraph" in types
assert "list_item" in types
assert "quote" in types
def test_resolve_canonical_source_url_priority():
article_full = {
"crawled_url": "https://example.com/crawled",
"input_meta": {"url": "https://example.com/meta"},
"trafilatura": {"canonical_url": "https://example.com/canonical"},
}
# trafilatura canonical_url has top priority
assert resolve_canonical_source_url(article_full) == "https://example.com/canonical"
# fallback to input_meta.url
article_no_traf = {
"crawled_url": "https://example.com/crawled",
"input_meta": {"url": "https://example.com/meta"},
}
assert resolve_canonical_source_url(article_no_traf) == "https://example.com/meta"
def test_author_parsing_forbids_delimiter_splitting():
article = {"trafilatura": {"author": "Dr. Silva, Ph.D. / Reporter Especial"}}
cand = parse_metadata_candidates(article)
# The full string must be preserved as a single author candidate, not split by commas or slashes
assert len(cand["author_candidates"]) == 1
assert cand["author_candidates"][0]["text"] == "Dr. Silva, Ph.D. / Reporter Especial"