feat(extractor): add Google News headlines extractor with Foxcape headless and URL resolution

- Add standalone CLI script scripts/extract_google_news.py for Google News RSS scraping
- Integrate foxcape in headless mode as primary stealth anti-bot engine
- Implement parallel article URL resolution using googlenewsdecoder and ThreadPoolExecutor
- Support language and regional locale mapping (-l, --lang, --locale)
- Implement real-time progress logging in stderr and --silent flag
- Add unit, integration, and live E2E tests in tests/test_extract_google_news.py
- Add full SpecKit documentation (specs/002-google-news-extractor/)
- Create comprehensive README.md covering both NLP Classifier and Google News Extractor
This commit is contained in:
2026-08-20 11:50:16 -03:00
parent 67cc40f91a
commit 6e3d57619b
59 changed files with 16118 additions and 2160 deletions
+29 -5
View File
@@ -39,15 +39,15 @@
"37": "Plan Setup",
"38": "Task Setup",
"39": "Graphify Workflows",
"40": "Feature Specification: Multilingual NLP Entity Inherence Classifier (POC)",
"40": "main",
"41": "1. Technical Decisions & Tradeoffs",
"42": "1. Input Schemas",
"43": "2. Basic CLI Usage Examples",
"44": "2. Standard Streams & Exit Codes",
"45": "ECPSnapshot",
"46": "test_models.py",
"45": "ClassificationResult",
"46": "InherenceClassifier",
"47": "detect_language",
"48": "main",
"48": "test_models.py",
"49": "content_northvolt_de.md",
"50": "content_presal_pt.md",
"51": "content_tangential_es.md",
@@ -78,5 +78,29 @@
"76": "pt/not_related.md",
"77": "pt/tangential.md",
"78": "tests/__init__.py",
"79": "text-nlp-classifier"
"79": "text-nlp-classifier",
"80": "get_hl_gl_ceid",
"81": "Extrator de Notícias do Google News — Guia Completo de Funcionamento",
"82": "extract_google_news.py",
"83": "ExtractionResult",
"84": "test_extract_google_news.py",
"85": "Implementation Tasks: Google News Headlines Extractor",
"86": "Feature Specification: Google News Headlines Extractor",
"87": "2. Cenários Práticos de Uso",
"88": "Implementation Plan: Google News Headlines Extractor",
"89": "scripts/__init__.py",
"90": "SearchQuery",
"91": "1. Technical Decisions & Tradeoffs",
"92": "General Readiness Checklist: Google News Headlines Extractor",
"93": "1. Entidades de Domínio & DTOs",
"94": "Specification Quality Checklist: Google News Headlines Extractor",
"95": "CLI Contract: Google News Headlines Extractor",
"96": "readiness.md",
"97": "🧠 TextNLPClassifierApp",
"98": "build_parser",
"99": "sample_rss_xml",
"100": "classifier.py",
"101": "ECPSnapshot",
"102": "Feature Specification: Multilingual NLP Entity Inherence Classifier (POC)",
"103": "main"
}