Files
TextNLPClassifierApp/graphify-out/cache/ast/v0.9.47-s2/71e796c83a1c293fa380b4f01a2eed810275e4628548e0f8e8b9ecaa90d3ac13.json
T
andreferraro 6e3d57619b feat(extractor): add Google News headlines extractor with Foxcape headless and URL resolution
- Add standalone CLI script scripts/extract_google_news.py for Google News RSS scraping
- Integrate foxcape in headless mode as primary stealth anti-bot engine
- Implement parallel article URL resolution using googlenewsdecoder and ThreadPoolExecutor
- Support language and regional locale mapping (-l, --lang, --locale)
- Implement real-time progress logging in stderr and --silent flag
- Add unit, integration, and live E2E tests in tests/test_extract_google_news.py
- Add full SpecKit documentation (specs/002-google-news-extractor/)
- Create comprehensive README.md covering both NLP Classifier and Google News Extractor
2026-08-20 11:50:16 -03:00

1 line
5.2 KiB
JSON

{"nodes": [{"id": "$graphify-root$_specs_002_google_news_extractor_research_md", "label": "research.md", "file_type": "document", "node_kind": "page", "source_file": "specs/002-google-news-extractor/research.md", "source_location": "L1"}, {"id": "$graphify-root$_specs_002_google_news_extractor_research_research_google_news_headlines_extractor", "label": "Research: Google News Headlines Extractor", "file_type": "document", "node_kind": "heading", "source_file": "specs/002-google-news-extractor/research.md", "source_location": "L1"}, {"id": "$graphify-root$_specs_002_google_news_extractor_research_1_technical_decisions_tradeoffs", "label": "1. Technical Decisions & Tradeoffs", "file_type": "document", "node_kind": "heading", "source_file": "specs/002-google-news-extractor/research.md", "source_location": "L3"}, {"id": "$graphify-root$_specs_002_google_news_extractor_research_decision_1_motor_de_requisi\u00e7\u00e3o_e_scraping_com_foxcape_em_modo_headless", "label": "Decision 1: Motor de Requisi\u00e7\u00e3o e Scraping com `foxcape` em Modo Headless", "file_type": "document", "node_kind": "heading", "source_file": "specs/002-google-news-extractor/research.md", "source_location": "L5"}, {"id": "$graphify-root$_specs_002_google_news_extractor_research_decision_2_endpoint_rss_do_google_news_vs_scraping_de_dom", "label": "Decision 2: Endpoint RSS do Google News vs. Scraping de DOM", "file_type": "document", "node_kind": "heading", "source_file": "specs/002-google-news-extractor/research.md", "source_location": "L15"}, {"id": "$graphify-root$_specs_002_google_news_extractor_research_decision_3_mapeamento_de_idioma_e_locale_hl_gl_ceid", "label": "Decision 3: Mapeamento de Idioma e Locale (`hl`, `gl`, `ceid`)", "file_type": "document", "node_kind": "heading", "source_file": "specs/002-google-news-extractor/research.md", "source_location": "L23"}, {"id": "$graphify-root$_specs_002_google_news_extractor_research_decision_4_resolu\u00e7\u00e3o_de_urls_do_google_news_via_googlenewsdecoder", "label": "Decision 4: Resolu\u00e7\u00e3o de URLs do Google News via `googlenewsdecoder`", "file_type": "document", "node_kind": "heading", "source_file": "specs/002-google-news-extractor/research.md", "source_location": "L36"}, {"id": "$graphify-root$_specs_002_google_news_extractor_research_decision_5_logging_em_tempo_real_no_stderr_e_segrega\u00e7\u00e3o_de_streams", "label": "Decision 5: Logging em Tempo Real no `stderr` e Segrega\u00e7\u00e3o de Streams", "file_type": "document", "node_kind": "heading", "source_file": "specs/002-google-news-extractor/research.md", "source_location": "L45"}], "edges": [{"source": "$graphify-root$_specs_002_google_news_extractor_research_md", "target": "$graphify-root$_specs_002_google_news_extractor_research_research_google_news_headlines_extractor", "relation": "contains", "confidence": "EXTRACTED", "source_file": "specs/002-google-news-extractor/research.md", "source_location": "L1", "weight": 1.0}, {"source": "$graphify-root$_specs_002_google_news_extractor_research_research_google_news_headlines_extractor", "target": "$graphify-root$_specs_002_google_news_extractor_research_1_technical_decisions_tradeoffs", "relation": "contains", "confidence": "EXTRACTED", "source_file": "specs/002-google-news-extractor/research.md", "source_location": "L3", "weight": 1.0}, {"source": "$graphify-root$_specs_002_google_news_extractor_research_1_technical_decisions_tradeoffs", "target": "$graphify-root$_specs_002_google_news_extractor_research_decision_1_motor_de_requisi\u00e7\u00e3o_e_scraping_com_foxcape_em_modo_headless", "relation": "contains", "confidence": "EXTRACTED", "source_file": "specs/002-google-news-extractor/research.md", "source_location": "L5", "weight": 1.0}, {"source": "$graphify-root$_specs_002_google_news_extractor_research_1_technical_decisions_tradeoffs", "target": "$graphify-root$_specs_002_google_news_extractor_research_decision_2_endpoint_rss_do_google_news_vs_scraping_de_dom", "relation": "contains", "confidence": "EXTRACTED", "source_file": "specs/002-google-news-extractor/research.md", "source_location": "L15", "weight": 1.0}, {"source": "$graphify-root$_specs_002_google_news_extractor_research_1_technical_decisions_tradeoffs", "target": "$graphify-root$_specs_002_google_news_extractor_research_decision_3_mapeamento_de_idioma_e_locale_hl_gl_ceid", "relation": "contains", "confidence": "EXTRACTED", "source_file": "specs/002-google-news-extractor/research.md", "source_location": "L23", "weight": 1.0}, {"source": "$graphify-root$_specs_002_google_news_extractor_research_1_technical_decisions_tradeoffs", "target": "$graphify-root$_specs_002_google_news_extractor_research_decision_4_resolu\u00e7\u00e3o_de_urls_do_google_news_via_googlenewsdecoder", "relation": "contains", "confidence": "EXTRACTED", "source_file": "specs/002-google-news-extractor/research.md", "source_location": "L36", "weight": 1.0}, {"source": "$graphify-root$_specs_002_google_news_extractor_research_1_technical_decisions_tradeoffs", "target": "$graphify-root$_specs_002_google_news_extractor_research_decision_5_logging_em_tempo_real_no_stderr_e_segrega\u00e7\u00e3o_de_streams", "relation": "contains", "confidence": "EXTRACTED", "source_file": "specs/002-google-news-extractor/research.md", "source_location": "L45", "weight": 1.0}], "input_tokens": 0, "output_tokens": 0}