feat(extractor): implement multi-engine article content extractor
- Added scripts/extract_article_contents.py for batch scraping with stealth Foxcape and triple extraction (Trafilatura, Newspaper4k, Readability) - Created unit, integration, and E2E test suite in tests/test_extract_article_contents.py (90/90 passing) - Updated specs/003-article-content-extractor and README.md with usage documentation and CLI contracts - Passed ruff linting/formatting and mypy type checking cleanly
This commit is contained in:
@@ -63,7 +63,9 @@ class ExtractNewsInputDTO(BaseModel):
|
||||
|
||||
keyword: str = Field(..., description="Palavra ou expressão de busca")
|
||||
language: str = Field(..., description="Código do idioma (ex: 'es', 'pt', 'en')")
|
||||
max_pages: int = Field(default=3, ge=1, le=10, description="Quantidade de páginas para extrair (1 a 10)")
|
||||
max_pages: int = Field(
|
||||
default=3, ge=1, le=10, description="Quantidade de páginas para extrair (1 a 10)"
|
||||
)
|
||||
```
|
||||
|
||||
| Campo | Tipo | Obrigatório | Descrição |
|
||||
@@ -90,7 +92,9 @@ class SearchQuery:
|
||||
raise InvalidSearchQueryError("A palavra-chave não pode ser vazia.")
|
||||
|
||||
if not self.language or len(self.language.strip()) < 2:
|
||||
raise InvalidSearchQueryError("O idioma deve conter pelo menos 2 caracteres (ex: 'es', 'pt', 'en').")
|
||||
raise InvalidSearchQueryError(
|
||||
"O idioma deve conter pelo menos 2 caracteres (ex: 'es', 'pt', 'en')."
|
||||
)
|
||||
|
||||
if self.max_pages < 1 or self.max_pages > 10:
|
||||
raise InvalidSearchQueryError("O número máximo de páginas deve estar entre 1 e 10.")
|
||||
@@ -123,6 +127,7 @@ class ExtractNewsUseCase:
|
||||
from googlenews_etl.infrastructure.adapters.google_news_extractor_adapter import (
|
||||
GoogleNewsExtractorAdapter,
|
||||
)
|
||||
|
||||
self.extractor = GoogleNewsExtractorAdapter()
|
||||
else:
|
||||
self.extractor = extractor
|
||||
@@ -199,7 +204,9 @@ class GoogleNewsExtractorAdapter(NewsExtractorPort):
|
||||
resolve_final_urls: bool = True,
|
||||
) -> None:
|
||||
self.impersonate = impersonate
|
||||
self.rate_limiter = rate_limiter or RateLimiterService(min_delay_seconds=0.5, max_delay_seconds=1.0)
|
||||
self.rate_limiter = rate_limiter or RateLimiterService(
|
||||
min_delay_seconds=0.5, max_delay_seconds=1.0
|
||||
)
|
||||
self.url_resolver = url_resolver or PlaywrightUrlResolverAdapter()
|
||||
self.resolve_final_urls = resolve_final_urls
|
||||
self.session = requests.Session(impersonate=self.impersonate)
|
||||
@@ -351,8 +358,12 @@ class PlaywrightUrlResolverAdapter(UrlResolverPort):
|
||||
if not any(
|
||||
x in u
|
||||
for x in [
|
||||
"google.", "gstatic.", "googleapis.", "googletagmanager.",
|
||||
"w3.org", "schema.org",
|
||||
"google.",
|
||||
"gstatic.",
|
||||
"googleapis.",
|
||||
"googletagmanager.",
|
||||
"w3.org",
|
||||
"schema.org",
|
||||
]
|
||||
):
|
||||
if not target_url or target_url == url:
|
||||
@@ -554,8 +565,8 @@ from googlenews_etl.application.use_cases.extract_news_use_case import ExtractNe
|
||||
dto_in = ExtractNewsInputDTO(keyword="inteligencia artificial", language="pt", max_pages=1)
|
||||
resultado = ExtractNewsUseCase().execute(dto_in)
|
||||
|
||||
print(resultado.total_itens) # ex: 10
|
||||
print(resultado.items[0].titulo) # título da primeira manchete
|
||||
print(resultado.items[0].url) # URL final resolvida
|
||||
print(resultado.total_itens) # ex: 10
|
||||
print(resultado.items[0].titulo) # título da primeira manchete
|
||||
print(resultado.items[0].url) # URL final resolvida
|
||||
print(resultado.items[0].quando_publicado) # data crua do RSS
|
||||
```
|
||||
Reference in New Issue
Block a user