Files
TextNLPClassifierApp/graphify-out/cache/ast/v0.9.47-s2/75d8e6444528450b5d11eee5b51965c98525070bb65eb0edabc996701a4c5075.json
T
andreferraro 6a45368cb0 feat(extractor): implement multi-engine article content extractor
- Added scripts/extract_article_contents.py for batch scraping with stealth Foxcape and triple extraction (Trafilatura, Newspaper4k, Readability)
- Created unit, integration, and E2E test suite in tests/test_extract_article_contents.py (90/90 passing)
- Updated specs/003-article-content-extractor and README.md with usage documentation and CLI contracts
- Passed ruff linting/formatting and mypy type checking cleanly
2026-08-20 19:22:20 -03:00

1 line
6.4 KiB
JSON

{"nodes": [{"id": "$graphify-root$_src_language_py", "label": "language.py", "file_type": "code", "source_file": "src/language.py", "source_location": "L1"}, {"id": "$graphify-root$_src_language_normalize_text", "label": "normalize_text()", "file_type": "code", "source_file": "src/language.py", "source_location": "L462", "_callable": true}, {"id": "$graphify-root$_src_language_extract_words", "label": "extract_words()", "file_type": "code", "source_file": "src/language.py", "source_location": "L472", "_callable": true}, {"id": "$graphify-root$_src_language_detect_language", "label": "detect_language()", "file_type": "code", "source_file": "src/language.py", "source_location": "L477", "_callable": true}, {"id": "$graphify-root$_src_language_rationale_1", "label": "Lightweight multilingual language detection and text normalization.", "file_type": "rationale", "source_file": "src/language.py", "source_location": "L1"}, {"id": "$graphify-root$_src_language_rationale_463", "label": "Normalize text by converting to lowercase and stripping combining diacritical\u2026", "file_type": "rationale", "source_file": "src/language.py", "source_location": "L463"}, {"id": "$graphify-root$_src_language_rationale_473", "label": "Tokenize text into lowercase alphanumeric words.", "file_type": "rationale", "source_file": "src/language.py", "source_location": "L473"}, {"id": "$graphify-root$_src_language_rationale_478", "label": "Detect the ISO-639-1 language code of text among supported languages (pt, en,\u2026", "file_type": "rationale", "source_file": "src/language.py", "source_location": "L478"}], "edges": [{"source": "$graphify-root$_src_language_py", "target": "re", "relation": "imports", "context": "import", "confidence": "EXTRACTED", "source_file": "src/language.py", "source_location": "L5", "weight": 1.0}, {"source": "$graphify-root$_src_language_py", "target": "unicodedata", "relation": "imports", "context": "import", "confidence": "EXTRACTED", "source_file": "src/language.py", "source_location": "L6", "weight": 1.0}, {"source": "$graphify-root$_src_language_py", "target": "$graphify-root$_src_language_normalize_text", "relation": "contains", "confidence": "EXTRACTED", "source_file": "src/language.py", "source_location": "L462", "weight": 1.0}, {"source": "$graphify-root$_src_language_py", "target": "$graphify-root$_src_language_extract_words", "relation": "contains", "confidence": "EXTRACTED", "source_file": "src/language.py", "source_location": "L472", "weight": 1.0}, {"source": "$graphify-root$_src_language_py", "target": "$graphify-root$_src_language_detect_language", "relation": "contains", "confidence": "EXTRACTED", "source_file": "src/language.py", "source_location": "L477", "weight": 1.0}, {"source": "$graphify-root$_src_language_detect_language", "target": "$graphify-root$_src_language_extract_words", "relation": "calls", "context": "call", "confidence": "EXTRACTED", "source_file": "src/language.py", "source_location": "L485", "weight": 1.0}, {"source": "$graphify-root$_src_language_detect_language", "target": "$graphify-root$_src_language_normalize_text", "relation": "calls", "context": "call", "confidence": "EXTRACTED", "source_file": "src/language.py", "source_location": "L556", "weight": 1.0}, {"source": "$graphify-root$_src_language_rationale_1", "target": "$graphify-root$_src_language_py", "relation": "rationale_for", "confidence": "EXTRACTED", "source_file": "src/language.py", "source_location": "L1", "weight": 1.0}, {"source": "$graphify-root$_src_language_rationale_463", "target": "$graphify-root$_src_language_normalize_text", "relation": "rationale_for", "confidence": "EXTRACTED", "source_file": "src/language.py", "source_location": "L463", "weight": 1.0}, {"source": "$graphify-root$_src_language_rationale_473", "target": "$graphify-root$_src_language_extract_words", "relation": "rationale_for", "confidence": "EXTRACTED", "source_file": "src/language.py", "source_location": "L473", "weight": 1.0}, {"source": "$graphify-root$_src_language_rationale_478", "target": "$graphify-root$_src_language_detect_language", "relation": "rationale_for", "confidence": "EXTRACTED", "source_file": "src/language.py", "source_location": "L478", "weight": 1.0}], "raw_calls": [{"caller_nid": "$graphify-root$_src_language_normalize_text", "callee": "normalize", "is_member_call": true, "source_file": "src/language.py", "source_location": "L467", "receiver": "unicodedata"}, {"caller_nid": "$graphify-root$_src_language_normalize_text", "callee": "lower", "is_member_call": true, "source_file": "src/language.py", "source_location": "L467", "receiver": "text"}, {"caller_nid": "$graphify-root$_src_language_normalize_text", "callee": "join", "is_member_call": true, "source_file": "src/language.py", "source_location": "L469", "receiver": null}, {"caller_nid": "$graphify-root$_src_language_normalize_text", "callee": "category", "is_member_call": true, "source_file": "src/language.py", "source_location": "L469", "receiver": "unicodedata"}, {"caller_nid": "$graphify-root$_src_language_extract_words", "callee": "findall", "is_member_call": true, "source_file": "src/language.py", "source_location": "L474", "receiver": "re"}, {"caller_nid": "$graphify-root$_src_language_extract_words", "callee": "lower", "is_member_call": true, "source_file": "src/language.py", "source_location": "L474", "receiver": "text"}, {"caller_nid": "$graphify-root$_src_language_detect_language", "callee": "strip", "is_member_call": true, "source_file": "src/language.py", "source_location": "L482", "receiver": "text"}, {"caller_nid": "$graphify-root$_src_language_detect_language", "callee": "items", "is_member_call": true, "source_file": "src/language.py", "source_location": "L493", "receiver": "LANGUAGE_STOPWORDS"}, {"caller_nid": "$graphify-root$_src_language_detect_language", "callee": "intersection", "is_member_call": true, "source_file": "src/language.py", "source_location": "L494", "receiver": "word_set"}, {"caller_nid": "$graphify-root$_src_language_detect_language", "callee": "items", "is_member_call": true, "source_file": "src/language.py", "source_location": "L498", "receiver": "scores"}, {"caller_nid": "$graphify-root$_src_language_detect_language", "callee": "intersection", "is_member_call": true, "source_file": "src/language.py", "source_location": "L560", "receiver": "norm_word_set"}, {"caller_nid": "$graphify-root$_src_language_detect_language", "callee": "intersection", "is_member_call": true, "source_file": "src/language.py", "source_location": "L561", "receiver": "norm_word_set"}]}