feat(extractor): add Google News headlines extractor with Foxcape headless and URL resolution
- Add standalone CLI script scripts/extract_google_news.py for Google News RSS scraping - Integrate foxcape in headless mode as primary stealth anti-bot engine - Implement parallel article URL resolution using googlenewsdecoder and ThreadPoolExecutor - Support language and regional locale mapping (-l, --lang, --locale) - Implement real-time progress logging in stderr and --silent flag - Add unit, integration, and live E2E tests in tests/test_extract_google_news.py - Add full SpecKit documentation (specs/002-google-news-extractor/) - Create comprehensive README.md covering both NLP Classifier and Google News Extractor
This commit is contained in:
@@ -4,7 +4,7 @@
|
||||
"2": "SpecKit Utilities",
|
||||
"3": "Graphify Commands",
|
||||
"4": "speckit-analyze/SKILL.md",
|
||||
"5": "Tasks: Multilingual NLP Entity Inherence Classifier (POC)",
|
||||
"5": "POC Readiness & Requirements Quality Checklist: Multilingual NLP Entity Inherence Classifier",
|
||||
"6": "Feature Specification Template",
|
||||
"7": "Graphify Rules",
|
||||
"8": "Implementation Planning",
|
||||
@@ -39,15 +39,15 @@
|
||||
"37": "Plan Setup",
|
||||
"38": "Task Setup",
|
||||
"39": "Graphify Workflows",
|
||||
"40": "Feature Specification: Multilingual NLP Entity Inherence Classifier (POC)",
|
||||
"40": "main",
|
||||
"41": "1. Technical Decisions & Tradeoffs",
|
||||
"42": "1. Input Schemas",
|
||||
"43": "2. Basic CLI Usage Examples",
|
||||
"44": "2. Standard Streams & Exit Codes",
|
||||
"45": "ECPSnapshot",
|
||||
"46": "classifier.py",
|
||||
"46": "Tasks: Multilingual NLP Entity Inherence Classifier (POC)",
|
||||
"47": "detect_language",
|
||||
"48": "main",
|
||||
"48": "test_models.py",
|
||||
"49": "content_northvolt_de.md",
|
||||
"50": "content_presal_pt.md",
|
||||
"51": "content_tangential_es.md",
|
||||
@@ -78,5 +78,24 @@
|
||||
"76": "pt/not_related.md",
|
||||
"77": "pt/tangential.md",
|
||||
"78": "tests/__init__.py",
|
||||
"79": "text-nlp-classifier"
|
||||
"79": "text-nlp-classifier",
|
||||
"80": "get_hl_gl_ceid",
|
||||
"81": "Extrator de Notícias do Google News — Guia Completo de Funcionamento",
|
||||
"82": "extract_google_news.py",
|
||||
"83": "ExtractionResult",
|
||||
"84": "test_extract_google_news.py",
|
||||
"85": "Implementation Tasks: Google News Headlines Extractor",
|
||||
"86": "Feature Specification: Google News Headlines Extractor",
|
||||
"87": "2. Cenários Práticos de Uso",
|
||||
"88": "Implementation Plan: Google News Headlines Extractor",
|
||||
"89": "scripts/__init__.py",
|
||||
"90": "SearchQuery",
|
||||
"91": "1. Technical Decisions & Tradeoffs",
|
||||
"92": "General Readiness Checklist: Google News Headlines Extractor",
|
||||
"93": "1. Entidades de Domínio & DTOs",
|
||||
"94": "Specification Quality Checklist: Google News Headlines Extractor",
|
||||
"95": "CLI Contract: Google News Headlines Extractor",
|
||||
"96": "readiness.md",
|
||||
"98": "build_parser",
|
||||
"99": "sample_rss_xml"
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user