feat(extractor): implement multi-engine article content extractor
- Added scripts/extract_article_contents.py for batch scraping with stealth Foxcape and triple extraction (Trafilatura, Newspaper4k, Readability) - Created unit, integration, and E2E test suite in tests/test_extract_article_contents.py (90/90 passing) - Updated specs/003-article-content-extractor and README.md with usage documentation and CLI contracts - Passed ruff linting/formatting and mypy type checking cleanly
This commit is contained in:
@@ -42,9 +42,7 @@ class SearchQuery:
|
||||
raise ValueError("A palavra-chave não pode ser vazia.")
|
||||
|
||||
if not self.language or len(self.language.strip()) < 2:
|
||||
raise ValueError(
|
||||
"O idioma deve conter pelo menos 2 caracteres (ex: 'pt', 'en', 'es')."
|
||||
)
|
||||
raise ValueError("O idioma deve conter pelo menos 2 caracteres (ex: 'pt', 'en', 'es').")
|
||||
|
||||
if self.max_pages < 1 or self.max_pages > 10:
|
||||
raise ValueError("O número máximo de páginas deve estar entre 1 e 10.")
|
||||
@@ -434,9 +432,7 @@ def main(argv: list[str] | None = None) -> int:
|
||||
verbose = not getattr(args, "silent", False)
|
||||
|
||||
try:
|
||||
result = extract_google_news(
|
||||
query, resolve_urls=args.resolve_urls, verbose=verbose
|
||||
)
|
||||
result = extract_google_news(query, resolve_urls=args.resolve_urls, verbose=verbose)
|
||||
json_output = json.dumps(
|
||||
result.to_dict(),
|
||||
ensure_ascii=False,
|
||||
|
||||
Reference in New Issue
Block a user