Files
TextNLPClassifierApp/graphify-out/2026-08-20/.graphify_labels.json
T
andreferraro 6e3d57619b feat(extractor): add Google News headlines extractor with Foxcape headless and URL resolution
- Add standalone CLI script scripts/extract_google_news.py for Google News RSS scraping
- Integrate foxcape in headless mode as primary stealth anti-bot engine
- Implement parallel article URL resolution using googlenewsdecoder and ThreadPoolExecutor
- Support language and regional locale mapping (-l, --lang, --locale)
- Implement real-time progress logging in stderr and --silent flag
- Add unit, integration, and live E2E tests in tests/test_extract_google_news.py
- Add full SpecKit documentation (specs/002-google-news-extractor/)
- Create comprehensive README.md covering both NLP Classifier and Google News Extractor
2026-08-20 11:50:16 -03:00

102 lines
3.3 KiB
JSON

{
"0": "Task Planning",
"1": "Convergence Workflow",
"2": "SpecKit Utilities",
"3": "Graphify Commands",
"4": "speckit-analyze/SKILL.md",
"5": "POC Readiness & Requirements Quality Checklist: Multilingual NLP Entity Inherence Classifier",
"6": "Feature Specification Template",
"7": "Graphify Rules",
"8": "Implementation Planning",
"9": "Feature Specification",
"10": "Task Generation",
"11": "Project Constitution",
"12": "Constitution Template",
"13": "Graphify Exports",
"14": "Ponytail Configuration",
"15": "Implementation Planning Template",
"16": "Ponytail Help",
"17": "Checklist Generation",
"18": "Clarification Workflow",
"19": "Implementation Workflow",
"20": "Graph Query",
"21": "Constitution Workflow",
"22": "Feature Branch Creation",
"23": "Ponytail Audit",
"24": "Ponytail Metrics",
"25": "Ponytail Review",
"26": "Task Issue Conversion",
"27": "Checklist Template",
"28": "Graphify Watch Mode",
"29": "Graphify Hooks",
"30": "Graphify Updates",
"31": "Ponytail Debt",
"32": "Repository Merge",
"33": "Media Transcription",
"34": "Extraction Specification",
"35": "Prerequisite Checks",
"36": "Template Resolution",
"37": "Plan Setup",
"38": "Task Setup",
"39": "Graphify Workflows",
"40": "main",
"41": "1. Technical Decisions & Tradeoffs",
"42": "1. Input Schemas",
"43": "2. Basic CLI Usage Examples",
"44": "2. Standard Streams & Exit Codes",
"45": "ECPSnapshot",
"46": "Tasks: Multilingual NLP Entity Inherence Classifier (POC)",
"47": "detect_language",
"48": "test_models.py",
"49": "content_northvolt_de.md",
"50": "content_presal_pt.md",
"51": "content_tangential_es.md",
"52": "adapters/__init__.py",
"53": "src/__init__.py",
"54": "de/contextual.md",
"55": "de/direct.md",
"56": "de/not_related.md",
"57": "de/tangential.md",
"58": "en/contextual.md",
"59": "en/direct.md",
"60": "en/not_related.md",
"61": "en/tangential.md",
"62": "es/contextual.md",
"63": "es/direct.md",
"64": "es/not_related.md",
"65": "es/tangential.md",
"66": "fr/contextual.md",
"67": "fr/direct.md",
"68": "fr/not_related.md",
"69": "fr/tangential.md",
"70": "it/contextual.md",
"71": "it/direct.md",
"72": "it/not_related.md",
"73": "it/tangential.md",
"74": "pt/contextual.md",
"75": "pt/direct.md",
"76": "pt/not_related.md",
"77": "pt/tangential.md",
"78": "tests/__init__.py",
"79": "text-nlp-classifier",
"80": "get_hl_gl_ceid",
"81": "Extrator de Notícias do Google News — Guia Completo de Funcionamento",
"82": "extract_google_news.py",
"83": "ExtractionResult",
"84": "test_extract_google_news.py",
"85": "Implementation Tasks: Google News Headlines Extractor",
"86": "Feature Specification: Google News Headlines Extractor",
"87": "2. Cenários Práticos de Uso",
"88": "Implementation Plan: Google News Headlines Extractor",
"89": "scripts/__init__.py",
"90": "SearchQuery",
"91": "1. Technical Decisions & Tradeoffs",
"92": "General Readiness Checklist: Google News Headlines Extractor",
"93": "1. Entidades de Domínio & DTOs",
"94": "Specification Quality Checklist: Google News Headlines Extractor",
"95": "CLI Contract: Google News Headlines Extractor",
"96": "readiness.md",
"98": "build_parser",
"99": "sample_rss_xml"
}