"""Unit tests for candidate parsing without regex covering scenarios PAR-001 to PAR-010.""" from src.runtime.candidate.parser import ( parse_metadata_candidates, parse_raw_text_into_candidates, resolve_canonical_source_url, ) def test_parse_raw_text_into_candidates(): markdown_text = """# Main Header This is the first paragraph of the article. ## Subheader Here is a second paragraph. * Bullet one * Bullet two > A notable quote from an expert. """ candidates = parse_raw_text_into_candidates(markdown_text, extractor="trafilatura") types = [c.type for c in candidates] assert "heading" in types assert "paragraph" in types assert "list_item" in types assert "quote" in types def test_resolve_canonical_source_url_priority(): article_full = { "crawled_url": "https://example.com/crawled", "input_meta": {"url": "https://example.com/meta"}, "trafilatura": {"canonical_url": "https://example.com/canonical"}, } # trafilatura canonical_url has top priority assert resolve_canonical_source_url(article_full) == "https://example.com/canonical" # fallback to input_meta.url article_no_traf = { "crawled_url": "https://example.com/crawled", "input_meta": {"url": "https://example.com/meta"}, } assert resolve_canonical_source_url(article_no_traf) == "https://example.com/meta" def test_author_parsing_forbids_delimiter_splitting(): article = {"trafilatura": {"author": "Dr. Silva, Ph.D. / Reporter Especial"}} cand = parse_metadata_candidates(article) # The full string must be preserved as a single author candidate, not split by commas or slashes assert len(cand["author_candidates"]) == 1 assert cand["author_candidates"][0]["text"] == "Dr. Silva, Ph.D. / Reporter Especial"