docs(tests): create tests/README.md with full suite inventory, coverage matrix, and execution guide
This commit is contained in:
@@ -46,7 +46,7 @@
|
||||
"44": "2. Standard Streams & Exit Codes",
|
||||
"45": "ClassificationResult",
|
||||
"46": "test_adversarial.py",
|
||||
"47": "LLMFallbackAdapter",
|
||||
"47": "InherenceClassifier",
|
||||
"48": "test_convert_article_to_markdown.py",
|
||||
"49": "content_northvolt_de.md",
|
||||
"50": "content_presal_pt.md",
|
||||
@@ -99,8 +99,8 @@
|
||||
"97": "🧠 TextNLPClassifierApp",
|
||||
"98": "Extraction Pipeline Checklist: Article Content Multi-Engine Extractor",
|
||||
"99": "parametrize",
|
||||
"100": "main",
|
||||
"101": "classifier.py",
|
||||
"100": "Path",
|
||||
"101": "main",
|
||||
"102": "Feature Specification: Multilingual NLP Entity Inherence Classifier (POC)",
|
||||
"103": "4. Requisitos Funcionais (FR)",
|
||||
"104": "Tasks: Article Content Multi-Engine Extractor",
|
||||
@@ -120,7 +120,7 @@
|
||||
"118": "Tasks: Deterministic Article Content Selection",
|
||||
"119": "select_article_extractor",
|
||||
"120": "process_batch",
|
||||
"121": "test_models.py",
|
||||
"121": "detect_language",
|
||||
"122": "test_select_article_extractor.py",
|
||||
"123": "Feature Specification: Deterministic Content Selection",
|
||||
"124": "2. Entity Descriptions & Fields",
|
||||
@@ -164,9 +164,13 @@
|
||||
"162": "test_normalize_date_iso_8601_variants",
|
||||
"163": "test_metadata_priority_original_url_all_fallbacks",
|
||||
"164": "test_normalize_scalar_non_string_types",
|
||||
"165": "InherenceClassifier",
|
||||
"165": "test_e2e_text_analysis_pipeline.py",
|
||||
"166": "remove_duplicate_initial_h1",
|
||||
"167": "test_normalize_scalar_whitespace_collapsing",
|
||||
"168": ".disambiguate",
|
||||
"169": "test_funnel_cli_subprocess_end_to_end"
|
||||
"168": "LLMFallbackAdapter",
|
||||
"169": "test_funnel_cli_subprocess_end_to_end",
|
||||
"170": "test_models.py",
|
||||
"171": "extract_evidence_snippets",
|
||||
"172": ".classify",
|
||||
"173": "🧪 Documentação da Suíte de Testes Automatizados"
|
||||
}
|
||||
|
||||
@@ -1 +1 @@
|
||||
{"0": "36bdb6f09c457f7c", "1": "8c5bf6244cf710c6", "2": "efbcc9c62a3ee78b", "3": "8599153989b07faa", "4": "b5952a1f7fee9f20", "5": "5b8462a3f82d188c", "6": "80f79e9e2011a3e3", "7": "4654167fd211d027", "8": "50acfa00fe353440", "9": "c6d2f770737823f1", "10": "44f2ca451aea24be", "11": "feaac5ab67a8c17a", "12": "b71bd92e5edbf2e0", "13": "219d65ba6d2689e4", "14": "8e30bb8112fd02d1", "15": "03906ab80b99db85", "16": "5d51c60ba1bc2be0", "17": "a1da914f522dcd21", "18": "fbad840891b90569", "19": "0686ff2d6fe29fb3", "20": "060baa9e1924b465", "21": "a5c8f2c3080b8243", "22": "0d76852f1d29eeb1", "23": "6ff68619f2d72924", "24": "3da11675eee7ec46", "25": "a6696589e9556f97", "26": "6c752999e8a4d4b6", "27": "2d4e13ea2111d750", "28": "4b60cb0ee1ac186a", "29": "f56fbca9bb8235ec", "30": "c7beed940704509f", "31": "38be2d254fb31ae8", "32": "ee5596fcf7e7c0b3", "33": "e4d4e0a440bc599f", "34": "c897e49c001acdae", "35": "3aad272a2cf5d495", "36": "0a197439d306b956", "37": "f43acf5c8b1329af", "38": "6775efafc9b33338", "39": "8176a164778526f9", "40": "66b69189c0acc3ff", "41": "0322ff824966a4d8", "42": "784c9e3d336a7f53", "43": "4b8bb6c3f7b64856", "44": "18c0ff3e6225bcb2", "45": "943c894b7e96a921", "46": "b30963ae66d7e3c9", "47": "85bec2d6e2b742cf", "48": "0237e1e02ee47a27", "49": "0d0f9f015921feef", "50": "8d0c81e5ca23e9a6", "51": "f79963571b9c15ee", "52": "5935824c825606cb", "53": "9685f9cbe158e50b", "54": "3d5ab759f350bc79", "55": "d549f24931a990e9", "56": "3cc031dcb648797c", "57": "a0ab88e6c629251d", "58": "76bd6412e2a22ecd", "59": "54827845564490c9", "60": "0a9736c416c0c6b9", "61": "77358620ac528153", "62": "3b0c585df09df48a", "63": "7e78cd3b28828c20", "64": "1c0c958231735f61", "65": "60b0f81225f62f69", "66": "920754c65cc94b88", "67": "df911472140a9b94", "68": "8e17bc11bcea91b9", "69": "7e905b75e4f28b95", "70": "a28424eca5d36c55", "71": "2cdb53d5b6051ab6", "72": "e42fbd3dc744e730", "73": "7fe2cac980de160c", "74": "2b1343a6a9db1487", "75": "54a1bb232f1d4ceb", "76": "442ba11d31ec0e0a", "77": "852a25b8b95bf8d1", "78": "1810ab370b9cd608", "79": "0fc5dca02a3f02f6", "80": "6ff8a97e63c9a2f3", "81": "a38f84ae3d895236", "82": "08e48bd11f9714df", "83": "5095122914e83cf5", "84": "1aef305bd7d7d63f", "85": "f8bfd0cfe9e8b478", "86": "410d15a346bd5894", "87": "6b41d288cfd834ab", "88": "5aa6db96312a8811", "89": "80225792bb62ba04", "90": "fd291228c3311f40", "91": "d4579c5b7aa2742a", "92": "7b9ba7c3bff11361", "93": "71cd9c1fa4a857f0", "94": "34cd980be3c32d21", "95": "970093453f3b7d90", "96": "9e96780a2b7c4bd6", "97": "b7c10b0e09caac0b", "98": "089ea6a55861c693", "99": "cb6165a7dc822d29", "100": "8cb2e59dcf557313", "101": "75f2ab420202693e", "102": "6aa00d5a83295f11", "103": "f58668f5b10ccdeb", "104": "4ec787414cc6f50b", "105": "1cf3077fd45d874a", "106": "edcd5d9bb3c4b00f", "107": "37f2f47110fe3eaa", "108": "b7ad5abb1da8cf8d", "109": "cb48a9c4f54efa38", "110": "f6dd36fd7f3edbe5", "111": "2925b620f0b1fd17", "112": "d8b3099917c3b711", "113": "3bb61caa0302c804", "114": "0d4f1d08dd056bb9", "115": "4ac2dcddeec2ff11", "116": "07da9aae9668f573", "117": "196f63e0c4536d30", "118": "ade84262e3cfac12", "119": "ebe4e5e0c42c613f", "120": "27256931b19a2867", "121": "3d7cd9541766116e", "122": "aa8a1de55696b666", "123": "96618c9a362af46c", "124": "83f104cbb62fd03e", "125": "6db738fb27190349", "126": "6a087a22cbcef972", "127": "85fd71a0cad8d3a5", "128": "22dd4feed96c4229", "129": "c4d2f60f532e6f16", "130": "f6b0aa8a1568926b", "131": "142d0db70bad18fe", "132": "67ea4284cbc02c54", "133": "d899cfc86c7a4a27", "134": "ba9464410a9b4168", "135": "0d496a12149eca27", "136": "521f5c7b9d566b4d", "137": "9e37828bdd2ba8c5", "138": "ec03c97194c56f91", "139": "f4e6d5dfa30034c5", "140": "d9b47fa423cf0748", "141": "1e0330b8757f333e", "142": "f35d75e1194c008d", "143": "4d2ae7190b514a34", "144": "4a98716cabf43f86", "145": "56b2431193739c38", "146": "edc785fd71bb0675", "147": "4c7347f8f86e1fbd", "148": "8ca77cc4fd6fd437", "149": "8968e9e7d55afcbe", "150": "f4f4ce1a1180ddb1", "151": "e426746f6e9ee15f", "152": "73cf7c102ed797ed", "153": "a3593e6f45bafb20", "154": "56747bad6345d66b", "155": "a8e7498fa7e257df", "156": "56e7b2355898077f", "157": "f2fc88f7d8214711", "158": "c966f6f8570c8c29", "159": "5f6094aa385f3bfe", "160": "2834e7d59672e756", "161": "cc6e436d94fd0033", "162": "64f33a2fc8969cd2", "163": "26ac1c0a00eabce1", "164": "cdcea44a6805ae55", "165": "f3f95b2d8f95c75f", "166": "85c97dab928b9b1b", "167": "7ab5695391e32126", "168": "8c55178d04b00f23", "169": "a68dbc0869da4c5e"}
|
||||
{"0": "36bdb6f09c457f7c", "1": "8c5bf6244cf710c6", "2": "efbcc9c62a3ee78b", "3": "8599153989b07faa", "4": "b5952a1f7fee9f20", "5": "5b8462a3f82d188c", "6": "80f79e9e2011a3e3", "7": "4654167fd211d027", "8": "50acfa00fe353440", "9": "c6d2f770737823f1", "10": "44f2ca451aea24be", "11": "feaac5ab67a8c17a", "12": "b71bd92e5edbf2e0", "13": "219d65ba6d2689e4", "14": "8e30bb8112fd02d1", "15": "03906ab80b99db85", "16": "5d51c60ba1bc2be0", "17": "a1da914f522dcd21", "18": "fbad840891b90569", "19": "0686ff2d6fe29fb3", "20": "060baa9e1924b465", "21": "a5c8f2c3080b8243", "22": "0d76852f1d29eeb1", "23": "6ff68619f2d72924", "24": "3da11675eee7ec46", "25": "a6696589e9556f97", "26": "6c752999e8a4d4b6", "27": "2d4e13ea2111d750", "28": "4b60cb0ee1ac186a", "29": "f56fbca9bb8235ec", "30": "c7beed940704509f", "31": "38be2d254fb31ae8", "32": "ee5596fcf7e7c0b3", "33": "e4d4e0a440bc599f", "34": "c897e49c001acdae", "35": "3aad272a2cf5d495", "36": "0a197439d306b956", "37": "f43acf5c8b1329af", "38": "6775efafc9b33338", "39": "8176a164778526f9", "40": "66b69189c0acc3ff", "41": "0322ff824966a4d8", "42": "784c9e3d336a7f53", "43": "4b8bb6c3f7b64856", "44": "18c0ff3e6225bcb2", "45": "722991c46c10afed", "46": "137ad23716b24a73", "47": "69c7215a1a600ded", "48": "0237e1e02ee47a27", "49": "0d0f9f015921feef", "50": "8d0c81e5ca23e9a6", "51": "f79963571b9c15ee", "52": "5935824c825606cb", "53": "9685f9cbe158e50b", "54": "3d5ab759f350bc79", "55": "d549f24931a990e9", "56": "3cc031dcb648797c", "57": "a0ab88e6c629251d", "58": "76bd6412e2a22ecd", "59": "54827845564490c9", "60": "0a9736c416c0c6b9", "61": "77358620ac528153", "62": "3b0c585df09df48a", "63": "7e78cd3b28828c20", "64": "1c0c958231735f61", "65": "60b0f81225f62f69", "66": "920754c65cc94b88", "67": "df911472140a9b94", "68": "8e17bc11bcea91b9", "69": "7e905b75e4f28b95", "70": "a28424eca5d36c55", "71": "2cdb53d5b6051ab6", "72": "e42fbd3dc744e730", "73": "7fe2cac980de160c", "74": "2b1343a6a9db1487", "75": "54a1bb232f1d4ceb", "76": "442ba11d31ec0e0a", "77": "852a25b8b95bf8d1", "78": "1810ab370b9cd608", "79": "0fc5dca02a3f02f6", "80": "6ff8a97e63c9a2f3", "81": "a38f84ae3d895236", "82": "08e48bd11f9714df", "83": "5095122914e83cf5", "84": "1aef305bd7d7d63f", "85": "f8bfd0cfe9e8b478", "86": "410d15a346bd5894", "87": "6b41d288cfd834ab", "88": "5aa6db96312a8811", "89": "80225792bb62ba04", "90": "fd291228c3311f40", "91": "d4579c5b7aa2742a", "92": "7b9ba7c3bff11361", "93": "71cd9c1fa4a857f0", "94": "34cd980be3c32d21", "95": "970093453f3b7d90", "96": "9e96780a2b7c4bd6", "97": "b7c10b0e09caac0b", "98": "089ea6a55861c693", "99": "cb6165a7dc822d29", "100": "94ef9e0e5a443fe6", "101": "f1ad5eee3650a7ca", "102": "6aa00d5a83295f11", "103": "f58668f5b10ccdeb", "104": "4ec787414cc6f50b", "105": "1cf3077fd45d874a", "106": "edcd5d9bb3c4b00f", "107": "37f2f47110fe3eaa", "108": "b7ad5abb1da8cf8d", "109": "cb48a9c4f54efa38", "110": "f6dd36fd7f3edbe5", "111": "2925b620f0b1fd17", "112": "d8b3099917c3b711", "113": "3bb61caa0302c804", "114": "0d4f1d08dd056bb9", "115": "4ac2dcddeec2ff11", "116": "07da9aae9668f573", "117": "196f63e0c4536d30", "118": "ade84262e3cfac12", "119": "ebe4e5e0c42c613f", "120": "27256931b19a2867", "121": "65c28abd5c9a3103", "122": "aa8a1de55696b666", "123": "96618c9a362af46c", "124": "83f104cbb62fd03e", "125": "6db738fb27190349", "126": "6a087a22cbcef972", "127": "85fd71a0cad8d3a5", "128": "22dd4feed96c4229", "129": "c4d2f60f532e6f16", "130": "f6b0aa8a1568926b", "131": "142d0db70bad18fe", "132": "67ea4284cbc02c54", "133": "d899cfc86c7a4a27", "134": "ba9464410a9b4168", "135": "0d496a12149eca27", "136": "521f5c7b9d566b4d", "137": "9e37828bdd2ba8c5", "138": "ec03c97194c56f91", "139": "f4e6d5dfa30034c5", "140": "d9b47fa423cf0748", "141": "1e0330b8757f333e", "142": "f35d75e1194c008d", "143": "4d2ae7190b514a34", "144": "4a98716cabf43f86", "145": "56b2431193739c38", "146": "edc785fd71bb0675", "147": "4c7347f8f86e1fbd", "148": "8ca77cc4fd6fd437", "149": "8968e9e7d55afcbe", "150": "f4f4ce1a1180ddb1", "151": "e426746f6e9ee15f", "152": "290e66456e1dbc21", "153": "a3593e6f45bafb20", "154": "56747bad6345d66b", "155": "a8e7498fa7e257df", "156": "56e7b2355898077f", "157": "f2fc88f7d8214711", "158": "c966f6f8570c8c29", "159": "5f6094aa385f3bfe", "160": "2834e7d59672e756", "161": "cc6e436d94fd0033", "162": "64f33a2fc8969cd2", "163": "26ac1c0a00eabce1", "164": "cdcea44a6805ae55", "165": "b93c2f21e3c60614", "166": "85c97dab928b9b1b", "167": "7ab5695391e32126", "168": "4ac0ca3e4db3991c", "169": "a68dbc0869da4c5e", "170": "0fff6a9dc7ef908d", "171": "c93e305384cf368f", "172": "b9b5ff435e14f765", "173": "cbb280810ad58329"}
|
||||
@@ -148,7 +148,7 @@
|
||||
"146": "Specification Quality Checklist: Convert Article JSON to Markdown",
|
||||
"147": "CLI Contract: `convert_article_to_markdown.py`",
|
||||
"148": "9. Interface CLI",
|
||||
"149": "get_hl_gl_ceid",
|
||||
"149": "sample_rss_xml",
|
||||
"150": "13. Estratégia de testes",
|
||||
"151": "6. Contrato de entrada",
|
||||
"152": "ECPSnapshot",
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
# Graph Report - TextNLPClassifierApp (2026-08-21)
|
||||
|
||||
## Corpus Check
|
||||
- 203 files · ~114,894 words
|
||||
- 203 files · ~114,931 words
|
||||
- Verdict: corpus is large enough that graph structure adds value.
|
||||
|
||||
## Summary
|
||||
@@ -10,7 +10,7 @@
|
||||
- Token cost: 0 input · 0 output
|
||||
|
||||
## Graph Freshness
|
||||
- Built from commit: `a874b98d`
|
||||
- Built from commit: `2cdd3547`
|
||||
- Run `git rev-parse HEAD` and compare to check if the graph is stale.
|
||||
- Run `graphify update .` after code changes (no API cost).
|
||||
|
||||
@@ -157,7 +157,7 @@
|
||||
- Specification Quality Checklist: Convert Article JSON to Markdown
|
||||
- CLI Contract: `convert_article_to_markdown.py`
|
||||
- 9. Interface CLI
|
||||
- get_hl_gl_ceid
|
||||
- sample_rss_xml
|
||||
- 13. Estratégia de testes
|
||||
- 6. Contrato de entrada
|
||||
- ECPSnapshot
|
||||
@@ -196,7 +196,7 @@
|
||||
classify.py → src/models.py
|
||||
- `main()` --uses--> `ErrorCode` [INFERRED]
|
||||
classify.py → src/models.py
|
||||
- `test_e2e_extract_google_news_live_pipeline()` --uses--> `ExtractionResult` [INFERRED]
|
||||
- `test_extract_google_news_orchestration_mocked()` --uses--> `ExtractionResult` [INFERRED]
|
||||
tests/test_extract_google_news.py → scripts/extract_google_news.py
|
||||
- `test_llm_adapter_interface()` --calls--> `LLMFallbackAdapter` [EXTRACTED]
|
||||
tests/test_adapters.py → src/adapters/llm.py
|
||||
@@ -373,16 +373,16 @@ Cohesion: 0.08
|
||||
Nodes (24): 1. Visão geral (arquitetura), 2.1 DTO de entrada (`googlenews_etl/application/dtos/extract_news_dto.py`), 2.2 Value Object de validação (`googlenews_etl/domain/entities/search_query.py`), 2. Entrada, 3.1 O caso de uso (`googlenews_etl/application/use_cases/extract_news_use_case.py`), 3.2 A porta (`googlenews_etl/domain/ports/news_extractor_port.py`), 3.3.1 Inicialização: sessão HTTP com impersonação de browser, 3.3.2 Mapeamento idioma → parâmetros `hl`/`gl` (`_get_hl_gl`) (+16 more)
|
||||
|
||||
### Community 82 - "extract_google_news.py"
|
||||
Cohesion: 0.20
|
||||
Nodes (14): extract_google_news(), _fetch_rss_content(), NewsArticle, _normalize_text_for_comparison(), parse_google_news_rss(), Remove pontuação e espaços extras para comparação de redundância., Parseia o XML do RSS do Google News e extrai os itens estruturados., Resolve em paralelo as URLs intermediárias do Google News para os links finais… (+6 more)
|
||||
Cohesion: 0.15
|
||||
Nodes (18): extract_google_news(), _fetch_rss_content(), get_hl_gl_ceid(), NewsArticle, _normalize_text_for_comparison(), parse_google_news_rss(), Mapeia idioma e locale para os parâmetros hl, gl e ceid do Google News., Remove pontuação e espaços extras para comparação de redundância. (+10 more)
|
||||
|
||||
### Community 83 - "ExtractionResult"
|
||||
Cohesion: 0.29
|
||||
Nodes (5): ExtractionResult, Any, Resultado consolidado da extração., Valida a consolidação do ExtractionResult a partir da busca mockada com URLs…, test_extract_google_news_orchestration_mocked()
|
||||
Nodes (5): ExtractionResult, Any, Resultado consolidado da extração., Valida E2E o fluxo completo de busca, parsing e resolução de URLs reais ao vivo., test_e2e_extract_google_news_live_pipeline()
|
||||
|
||||
### Community 84 - "test_extract_google_news.py"
|
||||
Cohesion: 0.16
|
||||
Nodes (14): Resolve a URL intermediária do Google News para a URL real do veículo., resolve_article_url(), fixture, Testes unitários e de integração para o Extrator de Manchetes do Google News.…, Valida o parsing do feed RSS, higienização de tags HTML e deduplicação., Valida fallback gracioso de URL quando não é link do Google News ou em erro., Valida resolução bem-sucedida de URL do Google News para o portal destino., Valida E2E que o decodificador resolve uma URL real do Google News para o… (+6 more)
|
||||
Cohesion: 0.15
|
||||
Nodes (15): Resolve a URL intermediária do Google News para a URL real do veículo., resolve_article_url(), Testes unitários e de integração para o Extrator de Manchetes do Google News.…, Valida fallback gracioso de URL quando não é link do Google News ou em erro., Valida resolução bem-sucedida de URL do Google News para o portal destino., Valida E2E que o decodificador resolve uma URL real do Google News para o…, Valida o mapeamento padrão de idiomas para pares (hl, gl, ceid)., Valida a sobrescrita geográfica quando o argumento locale é especificado. (+7 more)
|
||||
|
||||
### Community 85 - "Implementation Tasks: Google News Headlines Extractor"
|
||||
Cohesion: 0.14
|
||||
@@ -402,7 +402,7 @@ Nodes (7): Architecture & Pipeline, Documentation (this feature), Implementation
|
||||
|
||||
### Community 90 - "SearchQuery"
|
||||
Cohesion: 0.20
|
||||
Nodes (6): Value Object com parâmetros de busca validados., SearchQuery, Valida E2E o fluxo completo de busca, parsing e resolução de URLs reais ao vivo., Valida as regras de negócio e limites de SearchQuery., test_e2e_extract_google_news_live_pipeline(), test_search_query_validation()
|
||||
Nodes (6): Value Object com parâmetros de busca validados., SearchQuery, Valida a consolidação do ExtractionResult a partir da busca mockada com URLs…, Valida as regras de negócio e limites de SearchQuery., test_extract_google_news_orchestration_mocked(), test_search_query_validation()
|
||||
|
||||
### Community 91 - "1. Technical Decisions & Tradeoffs"
|
||||
Cohesion: 0.25
|
||||
@@ -624,9 +624,9 @@ Nodes (5): 1. Script Signature, 2. Command-Line Arguments, 3. Exit Codes, 4. Sta
|
||||
Cohesion: 0.40
|
||||
Nodes (5): 9.1 Script, 9.2 Argumentos, 9.3 Exemplos, 9.4 Saída do processo, 9. Interface CLI
|
||||
|
||||
### Community 149 - "get_hl_gl_ceid"
|
||||
Cohesion: 0.25
|
||||
Nodes (8): get_hl_gl_ceid(), Mapeia idioma e locale para os parâmetros hl, gl e ceid do Google News., Valida o mapeamento padrão de idiomas para pares (hl, gl, ceid)., Valida a sobrescrita geográfica quando o argumento locale é especificado., Valida fallback dinâmico para idiomas regionais não listados explicitamente., test_get_hl_gl_ceid_default_mappings(), test_get_hl_gl_ceid_dynamic_fallback(), test_get_hl_gl_ceid_with_custom_locale()
|
||||
### Community 149 - "sample_rss_xml"
|
||||
Cohesion: 0.67
|
||||
Nodes (3): fixture, Fixture que fornece o conteúdo do XML de exemplo para testes offline., sample_rss_xml()
|
||||
|
||||
### Community 150 - "13. Estratégia de testes"
|
||||
Cohesion: 0.50
|
||||
|
||||
+143
-143
@@ -1689,52 +1689,16 @@
|
||||
"source_location": "L755"
|
||||
},
|
||||
{
|
||||
"id": "scripts_extract_google_news_get_hl_gl_ceid",
|
||||
"label": "get_hl_gl_ceid()",
|
||||
"id": "tests_test_extract_google_news_sample_rss_xml",
|
||||
"label": "sample_rss_xml()",
|
||||
"_callable": true,
|
||||
"_origin": "ast",
|
||||
"community": 149,
|
||||
"community_name": "get_hl_gl_ceid",
|
||||
"community_name": "sample_rss_xml",
|
||||
"file_type": "code",
|
||||
"norm_label": "get_hl_gl_ceid()",
|
||||
"source_file": "scripts/extract_google_news.py",
|
||||
"source_location": "L107"
|
||||
},
|
||||
{
|
||||
"id": "tests_test_extract_google_news_test_get_hl_gl_ceid_default_mappings",
|
||||
"label": "test_get_hl_gl_ceid_default_mappings()",
|
||||
"_callable": true,
|
||||
"_origin": "ast",
|
||||
"community": 149,
|
||||
"community_name": "get_hl_gl_ceid",
|
||||
"file_type": "code",
|
||||
"norm_label": "test_get_hl_gl_ceid_default_mappings()",
|
||||
"norm_label": "sample_rss_xml()",
|
||||
"source_file": "tests/test_extract_google_news.py",
|
||||
"source_location": "L37"
|
||||
},
|
||||
{
|
||||
"id": "tests_test_extract_google_news_test_get_hl_gl_ceid_dynamic_fallback",
|
||||
"label": "test_get_hl_gl_ceid_dynamic_fallback()",
|
||||
"_callable": true,
|
||||
"_origin": "ast",
|
||||
"community": 149,
|
||||
"community_name": "get_hl_gl_ceid",
|
||||
"file_type": "code",
|
||||
"norm_label": "test_get_hl_gl_ceid_dynamic_fallback()",
|
||||
"source_file": "tests/test_extract_google_news.py",
|
||||
"source_location": "L78"
|
||||
},
|
||||
{
|
||||
"id": "tests_test_extract_google_news_test_get_hl_gl_ceid_with_custom_locale",
|
||||
"label": "test_get_hl_gl_ceid_with_custom_locale()",
|
||||
"_callable": true,
|
||||
"_origin": "ast",
|
||||
"community": 149,
|
||||
"community_name": "get_hl_gl_ceid",
|
||||
"file_type": "code",
|
||||
"norm_label": "test_get_hl_gl_ceid_with_custom_locale()",
|
||||
"source_file": "tests/test_extract_google_news.py",
|
||||
"source_location": "L60"
|
||||
"source_location": "L32"
|
||||
},
|
||||
{
|
||||
"id": "src_models_ecpsnapshot_from_json_str",
|
||||
@@ -2310,7 +2274,7 @@
|
||||
"file_type": "code",
|
||||
"norm_label": "._parse_llm_response()",
|
||||
"source_file": "src/adapters/llm.py",
|
||||
"source_location": "L214"
|
||||
"source_location": "L225"
|
||||
},
|
||||
{
|
||||
"id": "tests_test_e2e_text_analysis_pipeline_test_funnel_cli_subprocess_end_to_end",
|
||||
@@ -3548,6 +3512,18 @@
|
||||
"source_file": "scripts/extract_google_news.py",
|
||||
"source_location": "L253"
|
||||
},
|
||||
{
|
||||
"id": "scripts_extract_google_news_get_hl_gl_ceid",
|
||||
"label": "get_hl_gl_ceid()",
|
||||
"_callable": true,
|
||||
"_origin": "ast",
|
||||
"community": 82,
|
||||
"community_name": "extract_google_news.py",
|
||||
"file_type": "code",
|
||||
"norm_label": "get_hl_gl_ceid()",
|
||||
"source_file": "scripts/extract_google_news.py",
|
||||
"source_location": "L107"
|
||||
},
|
||||
{
|
||||
"id": "scripts_extract_google_news_normalize_text_for_comparison",
|
||||
"label": "_normalize_text_for_comparison()",
|
||||
@@ -3584,6 +3560,18 @@
|
||||
"source_file": "scripts/extract_google_news.py",
|
||||
"source_location": "L220"
|
||||
},
|
||||
{
|
||||
"id": "tests_test_extract_google_news_test_parse_google_news_rss_with_fixture",
|
||||
"label": "test_parse_google_news_rss_with_fixture()",
|
||||
"_callable": true,
|
||||
"_origin": "ast",
|
||||
"community": 82,
|
||||
"community_name": "extract_google_news.py",
|
||||
"file_type": "code",
|
||||
"norm_label": "test_parse_google_news_rss_with_fixture()",
|
||||
"source_file": "tests/test_extract_google_news.py",
|
||||
"source_location": "L111"
|
||||
},
|
||||
{
|
||||
"id": "tests_test_extract_google_news_test_resolve_articles_urls_batch",
|
||||
"label": "test_resolve_articles_urls_batch()",
|
||||
@@ -3621,16 +3609,16 @@
|
||||
"source_location": "L73"
|
||||
},
|
||||
{
|
||||
"id": "tests_test_extract_google_news_test_extract_google_news_orchestration_mocked",
|
||||
"label": "test_extract_google_news_orchestration_mocked()",
|
||||
"id": "tests_test_extract_google_news_test_e2e_extract_google_news_live_pipeline",
|
||||
"label": "test_e2e_extract_google_news_live_pipeline()",
|
||||
"_callable": true,
|
||||
"_origin": "ast",
|
||||
"community": 83,
|
||||
"community_name": "ExtractionResult",
|
||||
"file_type": "code",
|
||||
"norm_label": "test_extract_google_news_orchestration_mocked()",
|
||||
"norm_label": "test_e2e_extract_google_news_live_pipeline()",
|
||||
"source_file": "tests/test_extract_google_news.py",
|
||||
"source_location": "L186"
|
||||
"source_location": "L302"
|
||||
},
|
||||
{
|
||||
"id": "scripts_extract_google_news_resolve_article_url",
|
||||
@@ -3644,18 +3632,6 @@
|
||||
"source_file": "scripts/extract_google_news.py",
|
||||
"source_location": "L207"
|
||||
},
|
||||
{
|
||||
"id": "tests_test_extract_google_news_sample_rss_xml",
|
||||
"label": "sample_rss_xml()",
|
||||
"_callable": true,
|
||||
"_origin": "ast",
|
||||
"community": 84,
|
||||
"community_name": "test_extract_google_news.py",
|
||||
"file_type": "code",
|
||||
"norm_label": "sample_rss_xml()",
|
||||
"source_file": "tests/test_extract_google_news.py",
|
||||
"source_location": "L32"
|
||||
},
|
||||
{
|
||||
"id": "tests_test_extract_google_news_test_e2e_resolve_real_google_news_url",
|
||||
"label": "test_e2e_resolve_real_google_news_url()",
|
||||
@@ -3669,16 +3645,40 @@
|
||||
"source_location": "L281"
|
||||
},
|
||||
{
|
||||
"id": "tests_test_extract_google_news_test_parse_google_news_rss_with_fixture",
|
||||
"label": "test_parse_google_news_rss_with_fixture()",
|
||||
"id": "tests_test_extract_google_news_test_get_hl_gl_ceid_default_mappings",
|
||||
"label": "test_get_hl_gl_ceid_default_mappings()",
|
||||
"_callable": true,
|
||||
"_origin": "ast",
|
||||
"community": 84,
|
||||
"community_name": "test_extract_google_news.py",
|
||||
"file_type": "code",
|
||||
"norm_label": "test_parse_google_news_rss_with_fixture()",
|
||||
"norm_label": "test_get_hl_gl_ceid_default_mappings()",
|
||||
"source_file": "tests/test_extract_google_news.py",
|
||||
"source_location": "L111"
|
||||
"source_location": "L37"
|
||||
},
|
||||
{
|
||||
"id": "tests_test_extract_google_news_test_get_hl_gl_ceid_dynamic_fallback",
|
||||
"label": "test_get_hl_gl_ceid_dynamic_fallback()",
|
||||
"_callable": true,
|
||||
"_origin": "ast",
|
||||
"community": 84,
|
||||
"community_name": "test_extract_google_news.py",
|
||||
"file_type": "code",
|
||||
"norm_label": "test_get_hl_gl_ceid_dynamic_fallback()",
|
||||
"source_file": "tests/test_extract_google_news.py",
|
||||
"source_location": "L78"
|
||||
},
|
||||
{
|
||||
"id": "tests_test_extract_google_news_test_get_hl_gl_ceid_with_custom_locale",
|
||||
"label": "test_get_hl_gl_ceid_with_custom_locale()",
|
||||
"_callable": true,
|
||||
"_origin": "ast",
|
||||
"community": 84,
|
||||
"community_name": "test_extract_google_news.py",
|
||||
"file_type": "code",
|
||||
"norm_label": "test_get_hl_gl_ceid_with_custom_locale()",
|
||||
"source_file": "tests/test_extract_google_news.py",
|
||||
"source_location": "L60"
|
||||
},
|
||||
{
|
||||
"id": "tests_test_extract_google_news_test_resolve_article_url_fallback",
|
||||
@@ -3753,16 +3753,16 @@
|
||||
"source_location": "L40"
|
||||
},
|
||||
{
|
||||
"id": "tests_test_extract_google_news_test_e2e_extract_google_news_live_pipeline",
|
||||
"label": "test_e2e_extract_google_news_live_pipeline()",
|
||||
"id": "tests_test_extract_google_news_test_extract_google_news_orchestration_mocked",
|
||||
"label": "test_extract_google_news_orchestration_mocked()",
|
||||
"_callable": true,
|
||||
"_origin": "ast",
|
||||
"community": 90,
|
||||
"community_name": "SearchQuery",
|
||||
"file_type": "code",
|
||||
"norm_label": "test_e2e_extract_google_news_live_pipeline()",
|
||||
"norm_label": "test_extract_google_news_orchestration_mocked()",
|
||||
"source_file": "tests/test_extract_google_news.py",
|
||||
"source_location": "L302"
|
||||
"source_location": "L186"
|
||||
},
|
||||
{
|
||||
"id": "tests_test_extract_google_news_test_search_query_validation",
|
||||
@@ -11112,48 +11112,26 @@
|
||||
"source_location": "L290"
|
||||
},
|
||||
{
|
||||
"id": "scripts_extract_google_news_rationale_108",
|
||||
"label": "Mapeia idioma e locale para os par\u00e2metros hl, gl e ceid do Google News.",
|
||||
"id": "tests_test_extract_google_news_py_fixture",
|
||||
"label": "fixture",
|
||||
"_origin": "ast",
|
||||
"community": 149,
|
||||
"community_name": "get_hl_gl_ceid",
|
||||
"file_type": "rationale",
|
||||
"norm_label": "mapeia idioma e locale para os parametros hl, gl e ceid do google news.",
|
||||
"source_file": "scripts/extract_google_news.py",
|
||||
"source_location": "L108"
|
||||
"community_name": "sample_rss_xml",
|
||||
"file_type": "code",
|
||||
"norm_label": "fixture",
|
||||
"source_file": "",
|
||||
"source_location": ""
|
||||
},
|
||||
{
|
||||
"id": "tests_test_extract_google_news_rationale_38",
|
||||
"label": "Valida o mapeamento padr\u00e3o de idiomas para pares (hl, gl, ceid).",
|
||||
"id": "tests_test_extract_google_news_rationale_33",
|
||||
"label": "Fixture que fornece o conte\u00fado do XML de exemplo para testes offline.",
|
||||
"_origin": "ast",
|
||||
"community": 149,
|
||||
"community_name": "get_hl_gl_ceid",
|
||||
"community_name": "sample_rss_xml",
|
||||
"file_type": "rationale",
|
||||
"norm_label": "valida o mapeamento padrao de idiomas para pares (hl, gl, ceid).",
|
||||
"norm_label": "fixture que fornece o conteudo do xml de exemplo para testes offline.",
|
||||
"source_file": "tests/test_extract_google_news.py",
|
||||
"source_location": "L38"
|
||||
},
|
||||
{
|
||||
"id": "tests_test_extract_google_news_rationale_61",
|
||||
"label": "Valida a sobrescrita geogr\u00e1fica quando o argumento locale \u00e9 especificado.",
|
||||
"_origin": "ast",
|
||||
"community": 149,
|
||||
"community_name": "get_hl_gl_ceid",
|
||||
"file_type": "rationale",
|
||||
"norm_label": "valida a sobrescrita geografica quando o argumento locale e especificado.",
|
||||
"source_file": "tests/test_extract_google_news.py",
|
||||
"source_location": "L61"
|
||||
},
|
||||
{
|
||||
"id": "tests_test_extract_google_news_rationale_79",
|
||||
"label": "Valida fallback din\u00e2mico para idiomas regionais n\u00e3o listados explicitamente.",
|
||||
"_origin": "ast",
|
||||
"community": 149,
|
||||
"community_name": "get_hl_gl_ceid",
|
||||
"file_type": "rationale",
|
||||
"norm_label": "valida fallback dinamico para idiomas regionais nao listados explicitamente.",
|
||||
"source_file": "tests/test_extract_google_news.py",
|
||||
"source_location": "L79"
|
||||
"source_location": "L33"
|
||||
},
|
||||
{
|
||||
"id": "specify_templates_plan_template",
|
||||
@@ -12145,7 +12123,7 @@
|
||||
"source_location": "L155"
|
||||
},
|
||||
{
|
||||
"id": "src_adapters_llm_rationale_219",
|
||||
"id": "src_adapters_llm_rationale_230",
|
||||
"label": "Parses and validates structured JSON response from LLM.",
|
||||
"_origin": "ast",
|
||||
"community": 168,
|
||||
@@ -12153,7 +12131,7 @@
|
||||
"file_type": "rationale",
|
||||
"norm_label": "parses and validates structured json response from llm.",
|
||||
"source_file": "src/adapters/llm.py",
|
||||
"source_location": "L219"
|
||||
"source_location": "L230"
|
||||
},
|
||||
{
|
||||
"id": "src_adapters_llm_rationale_57",
|
||||
@@ -17171,6 +17149,17 @@
|
||||
"source_file": "scripts/extract_google_news.py",
|
||||
"source_location": "L1"
|
||||
},
|
||||
{
|
||||
"id": "scripts_extract_google_news_rationale_108",
|
||||
"label": "Mapeia idioma e locale para os par\u00e2metros hl, gl e ceid do Google News.",
|
||||
"_origin": "ast",
|
||||
"community": 82,
|
||||
"community_name": "extract_google_news.py",
|
||||
"file_type": "rationale",
|
||||
"norm_label": "mapeia idioma e locale para os parametros hl, gl e ceid do google news.",
|
||||
"source_file": "scripts/extract_google_news.py",
|
||||
"source_location": "L108"
|
||||
},
|
||||
{
|
||||
"id": "scripts_extract_google_news_rationale_152",
|
||||
"label": "Remove pontua\u00e7\u00e3o e espa\u00e7os extras para compara\u00e7\u00e3o de redund\u00e2ncia.",
|
||||
@@ -17237,6 +17226,17 @@
|
||||
"source_file": "scripts/extract_google_news.py",
|
||||
"source_location": "L65"
|
||||
},
|
||||
{
|
||||
"id": "tests_test_extract_google_news_rationale_112",
|
||||
"label": "Valida o parsing do feed RSS, higieniza\u00e7\u00e3o de tags HTML e deduplica\u00e7\u00e3o.",
|
||||
"_origin": "ast",
|
||||
"community": 82,
|
||||
"community_name": "extract_google_news.py",
|
||||
"file_type": "rationale",
|
||||
"norm_label": "valida o parsing do feed rss, higienizacao de tags html e deduplicacao.",
|
||||
"source_file": "tests/test_extract_google_news.py",
|
||||
"source_location": "L112"
|
||||
},
|
||||
{
|
||||
"id": "tests_test_extract_google_news_rationale_162",
|
||||
"label": "Valida a resolu\u00e7\u00e3o concorrente em lote de uma lista de NewsArticle.",
|
||||
@@ -17271,15 +17271,15 @@
|
||||
"source_location": "L85"
|
||||
},
|
||||
{
|
||||
"id": "tests_test_extract_google_news_rationale_187",
|
||||
"label": "Valida a consolida\u00e7\u00e3o do ExtractionResult a partir da busca mockada com URLs\u2026",
|
||||
"id": "tests_test_extract_google_news_rationale_303",
|
||||
"label": "Valida E2E o fluxo completo de busca, parsing e resolu\u00e7\u00e3o de URLs reais ao vivo.",
|
||||
"_origin": "ast",
|
||||
"community": 83,
|
||||
"community_name": "ExtractionResult",
|
||||
"file_type": "rationale",
|
||||
"norm_label": "valida a consolidacao do extractionresult a partir da busca mockada com urls...",
|
||||
"norm_label": "valida e2e o fluxo completo de busca, parsing e resolucao de urls reais ao vivo.",
|
||||
"source_file": "tests/test_extract_google_news.py",
|
||||
"source_location": "L187"
|
||||
"source_location": "L303"
|
||||
},
|
||||
{
|
||||
"id": "tests_test_extract_google_news",
|
||||
@@ -17292,17 +17292,6 @@
|
||||
"source_file": "tests/test_extract_google_news.py",
|
||||
"source_location": "L1"
|
||||
},
|
||||
{
|
||||
"id": "tests_test_extract_google_news_py_fixture",
|
||||
"label": "fixture",
|
||||
"_origin": "ast",
|
||||
"community": 84,
|
||||
"community_name": "test_extract_google_news.py",
|
||||
"file_type": "code",
|
||||
"norm_label": "fixture",
|
||||
"source_file": "",
|
||||
"source_location": ""
|
||||
},
|
||||
{
|
||||
"id": "scripts_extract_google_news_rationale_208",
|
||||
"label": "Resolve a URL intermedi\u00e1ria do Google News para a URL real do ve\u00edculo.",
|
||||
@@ -17325,17 +17314,6 @@
|
||||
"source_file": "tests/test_extract_google_news.py",
|
||||
"source_location": "L1"
|
||||
},
|
||||
{
|
||||
"id": "tests_test_extract_google_news_rationale_112",
|
||||
"label": "Valida o parsing do feed RSS, higieniza\u00e7\u00e3o de tags HTML e deduplica\u00e7\u00e3o.",
|
||||
"_origin": "ast",
|
||||
"community": 84,
|
||||
"community_name": "test_extract_google_news.py",
|
||||
"file_type": "rationale",
|
||||
"norm_label": "valida o parsing do feed rss, higienizacao de tags html e deduplicacao.",
|
||||
"source_file": "tests/test_extract_google_news.py",
|
||||
"source_location": "L112"
|
||||
},
|
||||
{
|
||||
"id": "tests_test_extract_google_news_rationale_139",
|
||||
"label": "Valida fallback gracioso de URL quando n\u00e3o \u00e9 link do Google News ou em erro.",
|
||||
@@ -17370,15 +17348,37 @@
|
||||
"source_location": "L282"
|
||||
},
|
||||
{
|
||||
"id": "tests_test_extract_google_news_rationale_33",
|
||||
"label": "Fixture que fornece o conte\u00fado do XML de exemplo para testes offline.",
|
||||
"id": "tests_test_extract_google_news_rationale_38",
|
||||
"label": "Valida o mapeamento padr\u00e3o de idiomas para pares (hl, gl, ceid).",
|
||||
"_origin": "ast",
|
||||
"community": 84,
|
||||
"community_name": "test_extract_google_news.py",
|
||||
"file_type": "rationale",
|
||||
"norm_label": "fixture que fornece o conteudo do xml de exemplo para testes offline.",
|
||||
"norm_label": "valida o mapeamento padrao de idiomas para pares (hl, gl, ceid).",
|
||||
"source_file": "tests/test_extract_google_news.py",
|
||||
"source_location": "L33"
|
||||
"source_location": "L38"
|
||||
},
|
||||
{
|
||||
"id": "tests_test_extract_google_news_rationale_61",
|
||||
"label": "Valida a sobrescrita geogr\u00e1fica quando o argumento locale \u00e9 especificado.",
|
||||
"_origin": "ast",
|
||||
"community": 84,
|
||||
"community_name": "test_extract_google_news.py",
|
||||
"file_type": "rationale",
|
||||
"norm_label": "valida a sobrescrita geografica quando o argumento locale e especificado.",
|
||||
"source_file": "tests/test_extract_google_news.py",
|
||||
"source_location": "L61"
|
||||
},
|
||||
{
|
||||
"id": "tests_test_extract_google_news_rationale_79",
|
||||
"label": "Valida fallback din\u00e2mico para idiomas regionais n\u00e3o listados explicitamente.",
|
||||
"_origin": "ast",
|
||||
"community": 84,
|
||||
"community_name": "test_extract_google_news.py",
|
||||
"file_type": "rationale",
|
||||
"norm_label": "valida fallback dinamico para idiomas regionais nao listados explicitamente.",
|
||||
"source_file": "tests/test_extract_google_news.py",
|
||||
"source_location": "L79"
|
||||
},
|
||||
{
|
||||
"id": "specs_002_google_news_extractor_tasks_implementation_for_user_story_1",
|
||||
@@ -18055,15 +18055,15 @@
|
||||
"source_location": "L33"
|
||||
},
|
||||
{
|
||||
"id": "tests_test_extract_google_news_rationale_303",
|
||||
"label": "Valida E2E o fluxo completo de busca, parsing e resolu\u00e7\u00e3o de URLs reais ao vivo.",
|
||||
"id": "tests_test_extract_google_news_rationale_187",
|
||||
"label": "Valida a consolida\u00e7\u00e3o do ExtractionResult a partir da busca mockada com URLs\u2026",
|
||||
"_origin": "ast",
|
||||
"community": 90,
|
||||
"community_name": "SearchQuery",
|
||||
"file_type": "rationale",
|
||||
"norm_label": "valida e2e o fluxo completo de busca, parsing e resolucao de urls reais ao vivo.",
|
||||
"norm_label": "valida a consolidacao do extractionresult a partir da busca mockada com urls...",
|
||||
"source_file": "tests/test_extract_google_news.py",
|
||||
"source_location": "L303"
|
||||
"source_location": "L187"
|
||||
},
|
||||
{
|
||||
"id": "tests_test_extract_google_news_rationale_87",
|
||||
@@ -20530,7 +20530,7 @@
|
||||
"confidence_score": 1.0,
|
||||
"context": "call",
|
||||
"source_file": "src/adapters/llm.py",
|
||||
"source_location": "L239",
|
||||
"source_location": "L250",
|
||||
"weight": 1.0
|
||||
},
|
||||
{
|
||||
@@ -20542,7 +20542,7 @@
|
||||
"confidence_score": 1.0,
|
||||
"context": "call",
|
||||
"source_file": "src/adapters/llm.py",
|
||||
"source_location": "L248",
|
||||
"source_location": "L259",
|
||||
"weight": 1.0
|
||||
},
|
||||
{
|
||||
@@ -39911,7 +39911,7 @@
|
||||
"confidence": "EXTRACTED",
|
||||
"confidence_score": 1.0,
|
||||
"source_file": "src/adapters/llm.py",
|
||||
"source_location": "L214",
|
||||
"source_location": "L225",
|
||||
"weight": 1.0
|
||||
},
|
||||
{
|
||||
@@ -40828,14 +40828,14 @@
|
||||
"weight": 1.0
|
||||
},
|
||||
{
|
||||
"source": "src_adapters_llm_rationale_219",
|
||||
"source": "src_adapters_llm_rationale_230",
|
||||
"target": "src_adapters_llm_llmfallbackadapter_parse_llm_response",
|
||||
"relation": "rationale_for",
|
||||
"_origin": "ast",
|
||||
"confidence": "EXTRACTED",
|
||||
"confidence_score": 1.0,
|
||||
"source_file": "src/adapters/llm.py",
|
||||
"source_location": "L219",
|
||||
"source_location": "L230",
|
||||
"weight": 1.0
|
||||
},
|
||||
{
|
||||
@@ -43674,7 +43674,7 @@
|
||||
"confidence": "INFERRED",
|
||||
"confidence_score": 0.95,
|
||||
"source_file": "src/adapters/llm.py",
|
||||
"source_location": "L239",
|
||||
"source_location": "L250",
|
||||
"weight": 0.8
|
||||
},
|
||||
{
|
||||
@@ -44625,5 +44625,5 @@
|
||||
}
|
||||
],
|
||||
"hyperedges": [],
|
||||
"built_at_commit": "a874b98dac4bd7e25da62c6cb4b0acb92a60f57a"
|
||||
"built_at_commit": "2cdd3547b22ed202328985e763812c3704892d7b"
|
||||
}
|
||||
@@ -330,9 +330,9 @@
|
||||
"semantic_hash": ""
|
||||
},
|
||||
"src/adapters/llm.py": {
|
||||
"mtime": 1787321086.752706,
|
||||
"seen": 1787321205.5144775,
|
||||
"ast_hash": "21ac74a13ac5dfad7db498b17165f8b3",
|
||||
"mtime": 1787321869.4176295,
|
||||
"seen": 1787322035.8922434,
|
||||
"ast_hash": "357b505df0d92f76b1d0cd606741d1a5",
|
||||
"semantic_hash": ""
|
||||
},
|
||||
"src/classifier.py": {
|
||||
|
||||
@@ -1,16 +1,16 @@
|
||||
# Graph Report - TextNLPClassifierApp (2026-08-21)
|
||||
|
||||
## Corpus Check
|
||||
- 203 files · ~114,931 words
|
||||
- 204 files · ~115,644 words
|
||||
- Verdict: corpus is large enough that graph structure adds value.
|
||||
|
||||
## Summary
|
||||
- 1651 nodes · 2220 edges · 170 communities (122 shown, 48 thin omitted)
|
||||
- 1658 nodes · 2226 edges · 174 communities (126 shown, 48 thin omitted)
|
||||
- Extraction: 94% EXTRACTED · 6% INFERRED · 0% AMBIGUOUS · INFERRED: 127 edges (avg confidence: 0.95)
|
||||
- Token cost: 0 input · 0 output
|
||||
|
||||
## Graph Freshness
|
||||
- Built from commit: `2cdd3547`
|
||||
- Built from commit: `040edb61`
|
||||
- Run `git rev-parse HEAD` and compare to check if the graph is stale.
|
||||
- Run `graphify update .` after code changes (no API cost).
|
||||
|
||||
@@ -58,7 +58,7 @@
|
||||
- 2. Standard Streams & Exit Codes
|
||||
- ClassificationResult
|
||||
- test_adversarial.py
|
||||
- LLMFallbackAdapter
|
||||
- InherenceClassifier
|
||||
- test_convert_article_to_markdown.py
|
||||
- content_northvolt_de.md
|
||||
- content_presal_pt.md
|
||||
@@ -110,8 +110,8 @@
|
||||
- 🧠 TextNLPClassifierApp
|
||||
- Extraction Pipeline Checklist: Article Content Multi-Engine Extractor
|
||||
- parametrize
|
||||
- Path
|
||||
- main
|
||||
- classifier.py
|
||||
- Feature Specification: Multilingual NLP Entity Inherence Classifier (POC)
|
||||
- 4. Requisitos Funcionais (FR)
|
||||
- Tasks: Article Content Multi-Engine Extractor
|
||||
@@ -130,7 +130,7 @@
|
||||
- Tasks: Deterministic Article Content Selection
|
||||
- select_article_extractor
|
||||
- process_batch
|
||||
- test_models.py
|
||||
- detect_language
|
||||
- test_select_article_extractor.py
|
||||
- Feature Specification: Deterministic Content Selection
|
||||
- 2. Entity Descriptions & Fields
|
||||
@@ -173,11 +173,15 @@
|
||||
- test_normalize_date_iso_8601_variants
|
||||
- test_metadata_priority_original_url_all_fallbacks
|
||||
- test_normalize_scalar_non_string_types
|
||||
- InherenceClassifier
|
||||
- test_e2e_text_analysis_pipeline.py
|
||||
- remove_duplicate_initial_h1
|
||||
- test_normalize_scalar_whitespace_collapsing
|
||||
- .disambiguate
|
||||
- LLMFallbackAdapter
|
||||
- test_funnel_cli_subprocess_end_to_end
|
||||
- test_models.py
|
||||
- extract_evidence_snippets
|
||||
- .classify
|
||||
- 🧪 Documentação da Suíte de Testes Automatizados
|
||||
|
||||
## God Nodes (most connected - your core abstractions)
|
||||
1. `ECPSnapshot` - 78 edges
|
||||
@@ -194,19 +198,19 @@
|
||||
## Surprising Connections (you probably didn't know these)
|
||||
- `main()` --uses--> `ECPSnapshot` [INFERRED]
|
||||
classify.py → src/models.py
|
||||
- `main()` --uses--> `ErrorCode` [INFERRED]
|
||||
classify.py → src/models.py
|
||||
- `test_extract_google_news_orchestration_mocked()` --uses--> `ExtractionResult` [INFERRED]
|
||||
tests/test_extract_google_news.py → scripts/extract_google_news.py
|
||||
- `test_llm_adapter_interface()` --calls--> `LLMFallbackAdapter` [EXTRACTED]
|
||||
tests/test_adapters.py → src/adapters/llm.py
|
||||
- `test_embeddings_adapter_interface()` --calls--> `LocalEmbeddingsAdapter` [EXTRACTED]
|
||||
tests/test_adapters.py → src/adapters/embeddings.py
|
||||
- `classifier()` --uses--> `InherenceClassifier` [INFERRED]
|
||||
tests/test_benchmark_24.py → src/classifier.py
|
||||
- `test_funnel_multilingual_language_detection()` --uses--> `InherenceClassifier` [INFERRED]
|
||||
tests/test_e2e_text_analysis_pipeline.py → src/classifier.py
|
||||
|
||||
## Import Cycles
|
||||
- None detected.
|
||||
|
||||
## Communities (170 total, 48 thin omitted)
|
||||
## Communities (174 total, 48 thin omitted)
|
||||
|
||||
### Community 0 - "Task Planning"
|
||||
Cohesion: 0.07
|
||||
@@ -349,16 +353,16 @@ Cohesion: 0.29
|
||||
Nodes (6): 1.1 Arguments & Options, 1. Command Line Interface, 2.1 Exit Codes, 2.2 Standard Output (`stdout`) / Standard Error (`stderr`), 2. Standard Streams & Exit Codes, CLI Contract & Interface Specification (POC)
|
||||
|
||||
### Community 45 - "ClassificationResult"
|
||||
Cohesion: 0.10
|
||||
Nodes (19): ABC, BaseNLPAdapter, Base abstract adapter interface for optional Tier 2 / Tier 3 NLP enhancers., Abstract interface for pluggable NLP classification adapters., Return True if the underlying provider or model is installed and configured., Compute semantic similarity score between text and a set of candidate terms., Optionally refine an ambiguous classification result., LocalEmbeddingsAdapter (+11 more)
|
||||
Cohesion: 0.09
|
||||
Nodes (22): ABC, BaseNLPAdapter, Base abstract adapter interface for optional Tier 2 / Tier 3 NLP enhancers., Abstract interface for pluggable NLP classification adapters., Return True if the underlying provider or model is installed and configured., Compute semantic similarity score between text and a set of candidate terms., Optionally refine an ambiguous classification result., LocalEmbeddingsAdapter (+14 more)
|
||||
|
||||
### Community 46 - "test_adversarial.py"
|
||||
Cohesion: 0.10
|
||||
Nodes (21): RelatedEntity, Adversarial and robustness test suite for Multilingual NLP Entity Inherence…, Run CLI via subprocess without --output and verify stdout is pure parseable…, Run CLI via subprocess with empty content and verify error code and exit code., Content about city/state governance of São Paulo against ECP for São Paulo FC., Run CLI via subprocess with missing target_name and verify error payload., Run CLI via subprocess with corrupted JSON and verify error payload., High-weight related entity mentioned in passing without required domain anchors. (+13 more)
|
||||
Cohesion: 0.12
|
||||
Nodes (16): RelatedEntity, Adversarial and robustness test suite for Multilingual NLP Entity Inherence…, Run CLI via subprocess without --output and verify stdout is pure parseable…, Run CLI via subprocess with empty content and verify error code and exit code., Run CLI via subprocess with missing target_name and verify error payload., Run CLI via subprocess with corrupted JSON and verify error payload., Content about apple fruit/culinary recipe against Apple Inc. tech entity., High-weight related entity mentioned in passing without required domain anchors. (+8 more)
|
||||
|
||||
### Community 47 - "LLMFallbackAdapter"
|
||||
### Community 47 - "InherenceClassifier"
|
||||
Cohesion: 0.06
|
||||
Nodes (43): LLMFallbackAdapter, Optional adapter for LLM fallback boundary disambiguation., Any, parametrize, Suíte de Testes Exaustiva para o Classificador de Inerência (classify.py e…, Cenário 4.1: Caso ambíguo elevado para DIRECT_INHERENT pelo LLM., Cenário 4.2: Caso ambíguo elevado para CONTEXTUAL_INHERENT pelo LLM., Cenário 4.3: LLM confirma categoricamente que a menção é periférica /… (+35 more)
|
||||
Nodes (63): InherenceClassifier, Tier 1 Deterministic NLP Entity Inherence Classifier with optional Tier 2 /…, DecisionCategory, Content about city/state governance of São Paulo against ECP for São Paulo FC., test_adversarial_sao_paulo_city_vs_fc(), Unit tests for deterministic classification decision logic., test_contextual_inherent(), test_direct_inherent() (+55 more)
|
||||
|
||||
### Community 48 - "test_convert_article_to_markdown.py"
|
||||
Cohesion: 0.08
|
||||
@@ -436,13 +440,13 @@ Nodes (34): 1. Requirement Completeness, 2. Requirement Clarity & Non-Ambiguity,
|
||||
Cohesion: 0.22
|
||||
Nodes (9): parametrize, Garante aceitação de URLs absolutas com esquema HTTP e HTTPS válidos., Garante rejeição de esquemas não permitidos, URLs relativas e strings vazias., Garante que a ausência de corpo no extrator selecionado NUNCA faça fallback…, Garante que todos os placeholders documentados no PRD sejam descartados…, test_normalize_scalar_placeholders_discarded(), test_resolve_article_body_strict_isolation_all_extractors(), test_validate_url_invalid_schemes() (+1 more)
|
||||
|
||||
### Community 100 - "main"
|
||||
Cohesion: 0.14
|
||||
Nodes (20): main(), Path, Cenário 6.1: Caminho de ECP inexistente -> Exit Code 1, error_code:…, Cenário 6.2: Arquivo ECP com sintaxe JSON corrompida., Cenário 6.3: Valida erro para falta de cada um dos campos obrigatórios do ECP., Cenário 6.4: Caminho de arquivo Markdown inexistente., Cenário 6.5: Arquivo Markdown vazio ou contendo apenas espaços em branco., Cenário 6.6: A flag -o / --output cria diretórios aninhados automaticamente. (+12 more)
|
||||
### Community 100 - "Path"
|
||||
Cohesion: 0.15
|
||||
Nodes (13): Path, Cenário 6.1: Caminho de ECP inexistente -> Exit Code 1, error_code:…, Cenário 6.2: Arquivo ECP com sintaxe JSON corrompida., Cenário 6.3: Valida erro para falta de cada um dos campos obrigatórios do ECP., Cenário 6.4: Caminho de arquivo Markdown inexistente., Cenário 6.5: Arquivo Markdown vazio ou contendo apenas espaços em branco., Cenário 6.6: A flag -o / --output cria diretórios aninhados automaticamente., test_cli_error_content_file_does_not_exist() (+5 more)
|
||||
|
||||
### Community 101 - "classifier.py"
|
||||
Cohesion: 0.17
|
||||
Nodes (12): emit_error(), parse_args(), Namespace, Core deterministic classification engine (Tier 1 core)., ErrorCode, MatchedGraphEntity, Enum, str (+4 more)
|
||||
### Community 101 - "main"
|
||||
Cohesion: 0.25
|
||||
Nodes (12): emit_error(), main(), parse_args(), Namespace, ErrorCode, str, CLI execution tests covering flags, arguments, stdout, and error handling., test_cli_empty_content_file() (+4 more)
|
||||
|
||||
### Community 102 - "Feature Specification: Multilingual NLP Entity Inherence Classifier (POC)"
|
||||
Cohesion: 0.14
|
||||
@@ -516,9 +520,9 @@ Nodes (35): CandidateStatus, extract_candidate_data(), ExtractorName, Any, Enum,
|
||||
Cohesion: 0.11
|
||||
Nodes (24): atomic_save_json(), process_batch(), Path, Salva dados em JSON de forma atômica utilizando arquivo temporário e rename., Lê o JSON de entrada, valida a estrutura, processa todos os artigos e grava o…, Path, CT-012: A entrada já contém selected_extractor -> Recalcular e substituir…, CT-013: articles está vazio -> Gerar saída válida com articles vazio. (+16 more)
|
||||
|
||||
### Community 121 - "test_models.py"
|
||||
Cohesion: 0.07
|
||||
Nodes (37): count_phrase_occurrences(), match_phrase_in_text(), Check if a normalized phrase appears in normalized text with word boundary…, Count occurrences of a phrase in text., Classify inherence of content against an ECP snapshot., detect_language(), extract_words(), normalize_text() (+29 more)
|
||||
### Community 121 - "detect_language"
|
||||
Cohesion: 0.22
|
||||
Nodes (13): detect_language(), extract_words(), Lightweight multilingual language detection and text normalization., Tokenize text into lowercase alphanumeric words., Detect the ISO-639-1 language code of text among supported languages (pt, en,…, Unit tests for language detection and text normalization., test_detect_english(), test_detect_french() (+5 more)
|
||||
|
||||
### Community 122 - "test_select_article_extractor.py"
|
||||
Cohesion: 0.18
|
||||
@@ -637,8 +641,8 @@ Cohesion: 0.50
|
||||
Nodes (4): 6.1 Formato, 6.2 Valores aceitos para `selected_extractor`, 6.3 Campos obrigatórios após a resolução, 6. Contrato de entrada
|
||||
|
||||
### Community 152 - "ECPSnapshot"
|
||||
Cohesion: 0.13
|
||||
Nodes (20): ECPSnapshot, parametrize, test_benchmark_case(), Suíte de Testes para o Adaptador de Fallback para LLM (Tier 3) do Classificador…, Valida extração de JSON quando a resposta do LLM vem formatada em bloco…, Valida que respostas corrompidas ou JSONs sem campos obrigatórios retornem None…, Garante que o classificador dispare o Tier 3 LLM para casos ambíguos…, Garante que casos claros (alta confiança e alta densidade de âncoras) NÃO… (+12 more)
|
||||
Cohesion: 0.12
|
||||
Nodes (21): ECPSnapshot, test_classifier_with_adapter_flags(), ecp_tech_corp(), fixture, Suíte de Testes para o Adaptador de Fallback para LLM (Tier 3) do Classificador…, Valida extração de JSON quando a resposta do LLM vem formatada em bloco…, Valida que respostas corrompidas ou JSONs sem campos obrigatórios retornem None…, Garante que o classificador dispare o Tier 3 LLM para casos ambíguos… (+13 more)
|
||||
|
||||
### Community 153 - "convert_html_to_markdown"
|
||||
Cohesion: 0.33
|
||||
@@ -652,35 +656,51 @@ Nodes (3): 1. Input JSON Schema, 2. Output JSON Schema, JSON Schema Contract: De
|
||||
Cohesion: 0.67
|
||||
Nodes (3): 5.1 Incluído, 5.2 Fora do escopo, 5. Escopo
|
||||
|
||||
### Community 165 - "InherenceClassifier"
|
||||
Cohesion: 0.07
|
||||
Nodes (45): InherenceClassifier, Tier 1 Deterministic NLP Entity Inherence Classifier with optional Tier 2 /…, DecisionCategory, Content about apple fruit/culinary recipe against Apple Inc. tech entity., test_adversarial_apple_fruit_recipe(), Unit tests for deterministic classification decision logic., test_contextual_inherent(), test_direct_inherent() (+37 more)
|
||||
### Community 165 - "test_e2e_text_analysis_pipeline.py"
|
||||
Cohesion: 0.10
|
||||
Nodes (19): ecp_river_plate(), fixture, parametrize, Suíte de Testes E2E e de Integração Completa para Análise de Texto e…, Cenário 1: Artigo com alta densidade de âncoras do River Plate. Oráculo:…, Cenário 2: Artigo sobre a Bacia do Rio da Prata ou clube homônimo do Uruguai.…, Cenário 3: Menção isolada do clube ('River') em contexto com poucas âncoras…, Cenário 4: Menção metafórica ou turística a um local próximo. Tier 1… (+11 more)
|
||||
|
||||
### Community 166 - "remove_duplicate_initial_h1"
|
||||
Cohesion: 0.50
|
||||
Nodes (4): Remove o primeiro título H1 do corpo somente quando ele for igual ao título…, remove_duplicate_initial_h1(), Testa remoção de H1 inicial coincidente com título com variações de espaços e…, test_remove_duplicate_initial_h1_exact_and_variations()
|
||||
|
||||
### Community 168 - ".disambiguate"
|
||||
Cohesion: 0.25
|
||||
Nodes (4): Executes LLM fallback for ambiguous boundary cases. Returns a refined…, Parses and validates structured JSON response from LLM., Returns True if an API key or custom provider function is configured., Constructs an expert-engineered prompt for multilingual entity inherence…
|
||||
### Community 168 - "LLMFallbackAdapter"
|
||||
Cohesion: 0.10
|
||||
Nodes (16): LLMFallbackAdapter, Executes LLM fallback for ambiguous boundary cases. Returns a refined…, Parses and validates structured JSON response from LLM., Optional adapter for LLM fallback boundary disambiguation., Returns True if an API key or custom provider function is configured., Constructs an expert-engineered prompt for multilingual entity inherence…, Any, Unit tests for optional adapter interfaces (Tier 2 / Tier 3). (+8 more)
|
||||
|
||||
### Community 169 - "test_funnel_cli_subprocess_end_to_end"
|
||||
Cohesion: 0.67
|
||||
Nodes (3): Path, Valida o contrato CLI completo classify.py com saída em arquivo JSON e flags…, test_funnel_cli_subprocess_end_to_end()
|
||||
|
||||
### Community 170 - "test_models.py"
|
||||
Cohesion: 0.16
|
||||
Nodes (9): ClassificationError, Any, parametrize, test_benchmark_case(), Unit tests for ECP models, schema validation, and structured error handling., test_classification_error_serialization(), test_ecp_snapshot_defaults(), test_ecp_snapshot_missing_required() (+1 more)
|
||||
|
||||
### Community 171 - "extract_evidence_snippets"
|
||||
Cohesion: 0.24
|
||||
Nodes (9): extract_evidence_snippets(), extract_sentences(), Markdown content parser and excerpt extraction utilities., Split text into individual sentences., Extract relevant sentence excerpts from Markdown text that contain any of the…, Remove markdown syntax markers (headers, bold, italics, links, code blocks) to…, strip_markdown(), test_extract_evidence_snippets() (+1 more)
|
||||
|
||||
### Community 172 - ".classify"
|
||||
Cohesion: 0.28
|
||||
Nodes (8): count_phrase_occurrences(), match_phrase_in_text(), Check if a normalized phrase appears in normalized text with word boundary…, Count occurrences of a phrase in text., Classify inherence of content against an ECP snapshot., normalize_text(), Normalize text by converting to lowercase and stripping combining diacritical…, test_normalize_text()
|
||||
|
||||
### Community 173 - "🧪 Documentação da Suíte de Testes Automatizados"
|
||||
Cohesion: 0.29
|
||||
Nodes (6): 1. Executar Toda a Suíte do Projeto (247 testes), 2. Executar por Módulo Específico, 🚀 Como Executar os Testes, 🔑 Configuração para Testes com LLM ao Vivo, 🧪 Documentação da Suíte de Testes Automatizados, 📊 Inventário e Mapa de Cobertura das Suítes
|
||||
|
||||
## Knowledge Gaps
|
||||
- **696 isolated node(s):** `text-nlp-classifier`, `MatchedGraphEntity`, `graphify`, `Usage`, `What graphify is for` (+691 more)
|
||||
- **700 isolated node(s):** `text-nlp-classifier`, `MatchedGraphEntity`, `graphify`, `Usage`, `What graphify is for` (+695 more)
|
||||
These have ≤1 connection - possible missing edges or undocumented components.
|
||||
- **48 thin communities (<3 nodes) omitted from report** — run `graphify query` to explore isolated nodes.
|
||||
|
||||
## Suggested Questions
|
||||
_Questions this graph is uniquely positioned to answer:_
|
||||
|
||||
- **Why does `ECPSnapshot` connect `ECPSnapshot` to `main`, `classifier.py`, `InherenceClassifier`, `.disambiguate`, `ClassificationResult`, `test_adversarial.py`, `LLMFallbackAdapter`, `test_models.py`?**
|
||||
_High betweenness centrality (0.009) - this node is a cross-community bridge._
|
||||
- **Why does `ECPSnapshot` connect `ECPSnapshot` to `main`, `test_e2e_text_analysis_pipeline.py`, `LLMFallbackAdapter`, `test_models.py`, `.classify`, `ClassificationResult`, `test_adversarial.py`, `InherenceClassifier`?**
|
||||
_High betweenness centrality (0.008) - this node is a cross-community bridge._
|
||||
- **Why does `PRD — Conversão de artigo JSON para Markdown` connect `PRD — Conversão de artigo JSON para Markdown` to `8. Regras funcionais`, `12. Critérios de aceite`, `11. Requisitos não funcionais`, `9. Interface CLI`, `13. Estratégia de testes`, `6. Contrato de entrada`, `5. Escopo`?**
|
||||
_High betweenness centrality (0.004) - this node is a cross-community bridge._
|
||||
- **Why does `InherenceClassifier` connect `InherenceClassifier` to `main`, `classifier.py`, `ClassificationResult`, `test_adversarial.py`, `LLMFallbackAdapter`, `ECPSnapshot`, `test_models.py`?**
|
||||
- **Why does `InherenceClassifier` connect `InherenceClassifier` to `main`, `test_e2e_text_analysis_pipeline.py`, `LLMFallbackAdapter`, `.classify`, `ClassificationResult`, `test_adversarial.py`, `ECPSnapshot`?**
|
||||
_High betweenness centrality (0.004) - this node is a cross-community bridge._
|
||||
- **Are the 42 inferred relationships involving `ECPSnapshot` (e.g. with `main()` and `BaseNLPAdapter`) actually correct?**
|
||||
_`ECPSnapshot` has 42 INFERRED edges - model-reasoned connections that need verification._
|
||||
|
||||
+1
@@ -0,0 +1 @@
|
||||
{"nodes": [{"id": "$graphify-root$_tests_readme_md", "label": "README.md", "file_type": "document", "node_kind": "page", "source_file": "tests/README.md", "source_location": "L1"}, {"id": "$graphify-root$_tests_readme_documenta\u00e7\u00e3o_da_su\u00edte_de_testes_automatizados", "label": "\ud83e\uddea Documenta\u00e7\u00e3o da Su\u00edte de Testes Automatizados", "file_type": "document", "node_kind": "heading", "source_file": "tests/README.md", "source_location": "L1"}, {"id": "$graphify-root$_tests_readme_invent\u00e1rio_e_mapa_de_cobertura_das_su\u00edtes", "label": "\ud83d\udcca Invent\u00e1rio e Mapa de Cobertura das Su\u00edtes", "file_type": "document", "node_kind": "heading", "source_file": "tests/README.md", "source_location": "L7"}, {"id": "$graphify-root$_tests_readme_como_executar_os_testes", "label": "\ud83d\ude80 Como Executar os Testes", "file_type": "document", "node_kind": "heading", "source_file": "tests/README.md", "source_location": "L29"}, {"id": "$graphify-root$_tests_readme_1_executar_toda_a_su\u00edte_do_projeto_247_testes", "label": "1. Executar Toda a Su\u00edte do Projeto (247 testes)", "file_type": "document", "node_kind": "heading", "source_file": "tests/README.md", "source_location": "L31"}, {"id": "$graphify-root$_tests_readme_2_executar_por_m\u00f3dulo_espec\u00edfico", "label": "2. Executar por M\u00f3dulo Espec\u00edfico", "file_type": "document", "node_kind": "heading", "source_file": "tests/README.md", "source_location": "L36"}, {"id": "$graphify-root$_tests_readme_configura\u00e7\u00e3o_para_testes_com_llm_ao_vivo", "label": "\ud83d\udd11 Configura\u00e7\u00e3o para Testes com LLM ao Vivo", "file_type": "document", "node_kind": "heading", "source_file": "tests/README.md", "source_location": "L63"}], "edges": [{"source": "$graphify-root$_tests_readme_md", "target": "$graphify-root$_tests_readme_documenta\u00e7\u00e3o_da_su\u00edte_de_testes_automatizados", "relation": "contains", "confidence": "EXTRACTED", "source_file": "tests/README.md", "source_location": "L1", "weight": 1.0}, {"source": "$graphify-root$_tests_readme_documenta\u00e7\u00e3o_da_su\u00edte_de_testes_automatizados", "target": "$graphify-root$_tests_readme_invent\u00e1rio_e_mapa_de_cobertura_das_su\u00edtes", "relation": "contains", "confidence": "EXTRACTED", "source_file": "tests/README.md", "source_location": "L7", "weight": 1.0}, {"source": "$graphify-root$_tests_readme_documenta\u00e7\u00e3o_da_su\u00edte_de_testes_automatizados", "target": "$graphify-root$_tests_readme_como_executar_os_testes", "relation": "contains", "confidence": "EXTRACTED", "source_file": "tests/README.md", "source_location": "L29", "weight": 1.0}, {"source": "$graphify-root$_tests_readme_como_executar_os_testes", "target": "$graphify-root$_tests_readme_1_executar_toda_a_su\u00edte_do_projeto_247_testes", "relation": "contains", "confidence": "EXTRACTED", "source_file": "tests/README.md", "source_location": "L31", "weight": 1.0}, {"source": "$graphify-root$_tests_readme_como_executar_os_testes", "target": "$graphify-root$_tests_readme_2_executar_por_m\u00f3dulo_espec\u00edfico", "relation": "contains", "confidence": "EXTRACTED", "source_file": "tests/README.md", "source_location": "L36", "weight": 1.0}, {"source": "$graphify-root$_tests_readme_documenta\u00e7\u00e3o_da_su\u00edte_de_testes_automatizados", "target": "$graphify-root$_tests_readme_configura\u00e7\u00e3o_para_testes_com_llm_ao_vivo", "relation": "contains", "confidence": "EXTRACTED", "source_file": "tests/README.md", "source_location": "L63", "weight": 1.0}], "input_tokens": 0, "output_tokens": 0}
|
||||
Vendored
+1
-1
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
+1370
-1220
File diff suppressed because it is too large
Load Diff
@@ -928,5 +928,11 @@
|
||||
"seen": 1787321481.15137,
|
||||
"ast_hash": "08e3c680d8669b2849a19d73b27b869a",
|
||||
"semantic_hash": ""
|
||||
},
|
||||
"tests/README.md": {
|
||||
"mtime": 1787322185.4865713,
|
||||
"seen": 1787322203.9119415,
|
||||
"ast_hash": "ea97839a9079df402573fbe4d0b33faf",
|
||||
"semantic_hash": ""
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,74 @@
|
||||
# 🧪 Documentação da Suíte de Testes Automatizados
|
||||
|
||||
Este diretório contém a suíte completa de **247 testes automatizados** com **100% de aprovação**, cobrindo testes unitários, testes de integração de pipeline, testes de regressão, contratos de CLI, provas de sensibilidade por mutação e testes End-to-End (E2E) com chamadas reais ao vivo para modelos de linguagem (LLM).
|
||||
|
||||
---
|
||||
|
||||
## 📊 Inventário e Mapa de Cobertura das Suítes
|
||||
|
||||
| Arquivo de Teste | Qtd. Testes | Escopo / O que valida? |
|
||||
|---|:---:|---|
|
||||
| [`test_classify_exhaustive_suite.py`](test_classify_exhaustive_suite.py) | **38** | **Suíte Exaustiva (QA Sênior)**: Todos os caminhos felizes (canônico, aliases, repetição, grafo), infelizes (homônimos, domínio sem alvo, empate de negativas), limiares/ambiguidade (`TANGENTIAL`), fallback LLM (upgrades, confirmações, rejeições, erros 500, timeouts, parsing resiliente, clipping de confidence), contratos de CLI e matriz multilíngue de 6 idiomas. |
|
||||
| [`test_e2e_text_analysis_pipeline.py`](test_e2e_text_analysis_pipeline.py) | **13** | **Funil Ponta a Ponta**: Validação completa do pipeline de texto (NLP determinístico $\rightarrow$ Ambiguidade $\rightarrow$ Fallback LLM $\rightarrow$ Degradação Graciosa) e teste de integração ao vivo (`test_funnel_live_api_execution_if_configured`) contra a API real (OpenAI / Omniroute / Gemini). |
|
||||
| [`test_llm_fallback.py`](test_llm_fallback.py) | **9** | **Módulo LLMFallbackAdapter (Tier 3)**: Validação da engenharia de prompt (`build_prompt`), detecção de disponibilidade, parsing JSON (puro e em blocos ````json ... ````), acionamento por limiar de confiança e resiliência a exceções. |
|
||||
| [`test_convert_article_to_markdown.py`](test_convert_article_to_markdown.py) | **67** | **Conversor JSON $\rightarrow$ Markdown (Spec 005)**: 100% dos requisitos funcionais (FR-001 a FR-018), matriz de prioridade de 11 campos de metadados, isolamento estrito de extratores, conversão ATX, remoção de headers duplicados e gravação atômica transacional. |
|
||||
| [`test_select_article_extractor.py`](test_select_article_extractor.py) | **30** | **Seletor Determinístico de Extrator (Spec 004)**: Cálculo de consenso $F_1$ n-gram, pontuação multi-critério (título, autor, data, corpo, imagens), isolamento de empates e enriquecimento com `selected_extractor`. |
|
||||
| [`test_extract_article_contents.py`](test_extract_article_contents.py) | **15** | **Extrator Multimotor (Spec 003)**: Scraping simultâneo via Trafilatura, Newspaper4k e Readability, isolamento de falhas individuais e fallbacks intra-motor (`markdown`/`html` $\rightarrow$ `text`). |
|
||||
| [`test_extract_google_news.py`](test_extract_google_news.py) | **16** | **Extrator Google News RSS (Spec 002)**: Parsing de feed RSS, motor anti-bot Foxcape headless, resolução paralela de URLs reais intermediárias e suporte a locales internacionais. |
|
||||
| [`test_benchmark_24.py`](test_benchmark_24.py) | **24** | **Matriz Canônica de Benchmark (Spec 001)**: 24 cenários controlados cobrindo **6 idiomas** (`pt`, `en`, `es`, `de`, `it`, `fr`) $\times$ **4 categorias de decisão** (`DIRECT_INHERENT`, `CONTEXTUAL_INHERENT`, `TANGENTIAL`, `NOT_RELATED`). |
|
||||
| [`test_adversarial.py`](test_adversarial.py) | **7** | **Testes Adversariais e Edge Cases**: Injeção de caracteres especiais, emojis, textos maciços, documentos sem quebra de linha, falsos cognatos e metáforas linguísticas. |
|
||||
| [`test_classifier.py`](test_classifier.py) | **5** | **Núcleo de Regras Determinísticas**: Asserções fundamentais do classificador Tier 1 sobre o ECP da Petrobras. |
|
||||
| [`test_cli.py`](test_cli.py) | **5** | **Interface CLI Básica**: Contratos de argumentos (`--ecp`, `--content`, `--output`), saída padrão formatada e captura de erros em `stderr`. |
|
||||
| [`test_language.py`](test_language.py) | **8** | **Módulo de Idioma e Normalização**: Identificação de ISO language, remoção de diacríticos/acentos e tokenização consciente de limites de palavras (*word boundaries*). |
|
||||
| [`test_models.py`](test_models.py) | **7** | **Modelos e Validações de Schema**: Parsing de ECP Snapshot JSON, dataclasses de nós do grafo (`RelatedEntity`) e serialização de `ClassificationResult`. |
|
||||
| [`test_adapters.py`](test_adapters.py) | **3** | **Interfaces de Adaptadores**: Contratos base das classes `BaseNLPAdapter`, `LocalEmbeddingsAdapter` e `LLMFallbackAdapter`. |
|
||||
| **TOTAL** | **247** | **100% de Aprovação (247/247 passing)** |
|
||||
|
||||
---
|
||||
|
||||
## 🚀 Como Executar os Testes
|
||||
|
||||
### 1. Executar Toda a Suíte do Projeto (247 testes)
|
||||
```bash
|
||||
pytest -v
|
||||
```
|
||||
|
||||
### 2. Executar por Módulo Específico
|
||||
|
||||
```bash
|
||||
# Suíte Exaustiva de Classificação e Fallback (QA Sênior - 38 testes)
|
||||
pytest tests/test_classify_exhaustive_suite.py -v
|
||||
|
||||
# Funil Ponta a Ponta e Integração Real com API (13 testes)
|
||||
pytest tests/test_e2e_text_analysis_pipeline.py -v
|
||||
|
||||
# Testes do Adaptador de Fallback LLM Tier 3 (9 testes)
|
||||
pytest tests/test_llm_fallback.py -v
|
||||
|
||||
# Conversor de Artigo para Markdown (67 testes)
|
||||
pytest tests/test_convert_article_to_markdown.py -v
|
||||
|
||||
# Seletor Determinístico de Extrator (30 testes)
|
||||
pytest tests/test_select_article_extractor.py -v
|
||||
|
||||
# Extrator de Conteúdo Multimotor (15 testes)
|
||||
pytest tests/test_extract_article_contents.py -v
|
||||
|
||||
# Extrator de Manchetes Google News (16 testes)
|
||||
pytest tests/test_extract_google_news.py -v
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 🔑 Configuração para Testes com LLM ao Vivo
|
||||
|
||||
Para que o teste `test_funnel_live_api_execution_if_configured` e a CLI `classify.py --enable-llm` executem chamadas reais na rede, configure o arquivo `.env` na raiz do projeto:
|
||||
|
||||
```env
|
||||
OPENAI_API_KEY="sk-..."
|
||||
OPENAI_BASE_URL="https://omniroute.app.andreferraro.com/v1"
|
||||
OPENAI_MODEL="cgpt-web/gpt-5.5"
|
||||
```
|
||||
*(Ou utilize `GEMINI_API_KEY="..."` para utilizar a API do Google Gemini).*
|
||||
|
||||
Se as chaves não estiverem configuradas, os testes unitários continuam executando 100% dos cenários através de mocks e stubs isolados sem quebrar o CI.
|
||||
Reference in New Issue
Block a user