Files
TextNLPClassifierApp/graphify-out/cache/ast/v0.9.47-s2/5daaacf1ae38b390fecd514996fb7150aea886c3c763e400ac5af36d2d9e0f6a.json
T
andreferraro 6e3d57619b feat(extractor): add Google News headlines extractor with Foxcape headless and URL resolution
- Add standalone CLI script scripts/extract_google_news.py for Google News RSS scraping
- Integrate foxcape in headless mode as primary stealth anti-bot engine
- Implement parallel article URL resolution using googlenewsdecoder and ThreadPoolExecutor
- Support language and regional locale mapping (-l, --lang, --locale)
- Implement real-time progress logging in stderr and --silent flag
- Add unit, integration, and live E2E tests in tests/test_extract_google_news.py
- Add full SpecKit documentation (specs/002-google-news-extractor/)
- Create comprehensive README.md covering both NLP Classifier and Google News Extractor
2026-08-20 11:50:16 -03:00

1 line
5.1 KiB
JSON

{"nodes": [{"id": "$graphify-root$_specs_002_google_news_extractor_research_md", "label": "research.md", "file_type": "document", "node_kind": "page", "source_file": "specs/002-google-news-extractor/research.md", "source_location": "L1"}, {"id": "$graphify-root$_specs_002_google_news_extractor_research_research_google_news_headlines_extractor", "label": "Research: Google News Headlines Extractor", "file_type": "document", "node_kind": "heading", "source_file": "specs/002-google-news-extractor/research.md", "source_location": "L1"}, {"id": "$graphify-root$_specs_002_google_news_extractor_research_1_technical_decisions_tradeoffs", "label": "1. Technical Decisions & Tradeoffs", "file_type": "document", "node_kind": "heading", "source_file": "specs/002-google-news-extractor/research.md", "source_location": "L3"}, {"id": "$graphify-root$_specs_002_google_news_extractor_research_decision_1_motor_de_requisi\u00e7\u00e3o_e_scraping_com_foxcape", "label": "Decision 1: Motor de Requisi\u00e7\u00e3o e Scraping com `foxcape`", "file_type": "document", "node_kind": "heading", "source_file": "specs/002-google-news-extractor/research.md", "source_location": "L5"}, {"id": "$graphify-root$_specs_002_google_news_extractor_research_decision_2_endpoint_rss_do_google_news_vs_scraping_de_dom", "label": "Decision 2: Endpoint RSS do Google News vs. Scraping de DOM", "file_type": "document", "node_kind": "heading", "source_file": "specs/002-google-news-extractor/research.md", "source_location": "L13"}, {"id": "$graphify-root$_specs_002_google_news_extractor_research_decision_3_mapeamento_de_idioma_e_locale_hl_gl_ceid", "label": "Decision 3: Mapeamento de Idioma e Locale (`hl`, `gl`, `ceid`)", "file_type": "document", "node_kind": "heading", "source_file": "specs/002-google-news-extractor/research.md", "source_location": "L19"}, {"id": "$graphify-root$_specs_002_google_news_extractor_research_decision_4_interface_cli_e_sa\u00edda_json_para_stdout", "label": "Decision 4: Interface CLI e Sa\u00edda JSON para Stdout", "file_type": "document", "node_kind": "heading", "source_file": "specs/002-google-news-extractor/research.md", "source_location": "L30"}, {"id": "$graphify-root$_specs_002_google_news_extractor_research_decision_5_limpeza_de_tags_html_e_deduplica\u00e7\u00e3o", "label": "Decision 5: Limpeza de Tags HTML e Deduplica\u00e7\u00e3o", "file_type": "document", "node_kind": "heading", "source_file": "specs/002-google-news-extractor/research.md", "source_location": "L36"}], "edges": [{"source": "$graphify-root$_specs_002_google_news_extractor_research_md", "target": "$graphify-root$_specs_002_google_news_extractor_research_research_google_news_headlines_extractor", "relation": "contains", "confidence": "EXTRACTED", "source_file": "specs/002-google-news-extractor/research.md", "source_location": "L1", "weight": 1.0}, {"source": "$graphify-root$_specs_002_google_news_extractor_research_research_google_news_headlines_extractor", "target": "$graphify-root$_specs_002_google_news_extractor_research_1_technical_decisions_tradeoffs", "relation": "contains", "confidence": "EXTRACTED", "source_file": "specs/002-google-news-extractor/research.md", "source_location": "L3", "weight": 1.0}, {"source": "$graphify-root$_specs_002_google_news_extractor_research_1_technical_decisions_tradeoffs", "target": "$graphify-root$_specs_002_google_news_extractor_research_decision_1_motor_de_requisi\u00e7\u00e3o_e_scraping_com_foxcape", "relation": "contains", "confidence": "EXTRACTED", "source_file": "specs/002-google-news-extractor/research.md", "source_location": "L5", "weight": 1.0}, {"source": "$graphify-root$_specs_002_google_news_extractor_research_1_technical_decisions_tradeoffs", "target": "$graphify-root$_specs_002_google_news_extractor_research_decision_2_endpoint_rss_do_google_news_vs_scraping_de_dom", "relation": "contains", "confidence": "EXTRACTED", "source_file": "specs/002-google-news-extractor/research.md", "source_location": "L13", "weight": 1.0}, {"source": "$graphify-root$_specs_002_google_news_extractor_research_1_technical_decisions_tradeoffs", "target": "$graphify-root$_specs_002_google_news_extractor_research_decision_3_mapeamento_de_idioma_e_locale_hl_gl_ceid", "relation": "contains", "confidence": "EXTRACTED", "source_file": "specs/002-google-news-extractor/research.md", "source_location": "L19", "weight": 1.0}, {"source": "$graphify-root$_specs_002_google_news_extractor_research_1_technical_decisions_tradeoffs", "target": "$graphify-root$_specs_002_google_news_extractor_research_decision_4_interface_cli_e_sa\u00edda_json_para_stdout", "relation": "contains", "confidence": "EXTRACTED", "source_file": "specs/002-google-news-extractor/research.md", "source_location": "L30", "weight": 1.0}, {"source": "$graphify-root$_specs_002_google_news_extractor_research_1_technical_decisions_tradeoffs", "target": "$graphify-root$_specs_002_google_news_extractor_research_decision_5_limpeza_de_tags_html_e_deduplica\u00e7\u00e3o", "relation": "contains", "confidence": "EXTRACTED", "source_file": "specs/002-google-news-extractor/research.md", "source_location": "L36", "weight": 1.0}], "input_tokens": 0, "output_tokens": 0}