test(e2e): add comprehensive Senior QA E2E text analysis and LLM fallback funnel suite

This commit is contained in:
2026-08-21 11:06:57 -03:00
parent bae144055e
commit a874b98dac
17 changed files with 2260 additions and 815 deletions
+7 -3
View File
@@ -500,7 +500,8 @@ TextNLPClassifierApp/
│ ├── test_extract_article_contents.py
│ ├── test_select_article_extractor.py
│ ├── test_convert_article_to_markdown.py # Testes da conversão para Markdown
│ └── test_llm_fallback.py # Testes do Tier 3 LLM Fallback
│ ├── test_llm_fallback.py # Testes do Tier 3 LLM Fallback
│ └── test_e2e_text_analysis_pipeline.py # Suíte E2E do Funil de Análise e Fallback
├── requirements.txt # Dependências do projeto
├── pyproject.toml # Configurações de ferramentas (pytest, ruff, mypy)
└── README.md # Documentação principal
@@ -510,12 +511,15 @@ TextNLPClassifierApp/
## 🧪 Testes e Qualidade de Código
O repositório possui **196 testes automatizados** com 100% de aprovação cobrindo testes unitários, de regressão, de integração, Golden Fixtures exatas, testes de sensibilidade de mutação, testes de fallback para LLM (Tier 3) e testes End-to-End (E2E) via CLI subprocess:
O repositório possui **209 testes automatizados** com 100% de aprovação cobrindo testes unitários, de regressão, de integração, Golden Fixtures exatas, testes de sensibilidade de mutação, testes de fallback para LLM (Tier 3), validações de degradação graciosa e testes End-to-End (E2E) via CLI subprocess:
```bash
# Executar toda a suíte de testes do projeto (196 testes)
# Executar toda a suíte de testes do projeto (209 testes)
pytest -v
# Executar a Suíte E2E do Funil de Análise de Texto e Fallback para LLM
pytest tests/test_e2e_text_analysis_pipeline.py -v
# Executar os testes do Fallback para LLM (Tier 3)
pytest tests/test_llm_fallback.py -v
+3 -3
View File
@@ -99,7 +99,7 @@
"97": "🧠 TextNLPClassifierApp",
"98": "Extraction Pipeline Checklist: Article Content Multi-Engine Extractor",
"99": "parametrize",
"100": "models.py",
"100": "main",
"101": "ECPSnapshot",
"102": "Feature Specification: Multilingual NLP Entity Inherence Classifier (POC)",
"103": "4. Requisitos Funcionais (FR)",
@@ -151,7 +151,7 @@
"149": "sample_rss_xml",
"150": "13. Estratégia de testes",
"151": "6. Contrato de entrada",
"152": "InherenceClassifier",
"152": "LLMFallbackAdapter",
"153": "convert_html_to_markdown",
"154": "JSON Schema Contract: Deterministic Article Content Selection",
"155": "5. Escopo",
@@ -164,7 +164,7 @@
"162": "test_normalize_date_iso_8601_variants",
"163": "test_metadata_priority_original_url_all_fallbacks",
"164": "test_normalize_scalar_non_string_types",
"165": ".disambiguate",
"165": "InherenceClassifier",
"166": "remove_duplicate_initial_h1",
"167": "test_normalize_scalar_whitespace_collapsing"
}
+1 -1
View File
@@ -1 +1 @@
{"0": "36bdb6f09c457f7c", "1": "8c5bf6244cf710c6", "2": "efbcc9c62a3ee78b", "3": "8599153989b07faa", "4": "b5952a1f7fee9f20", "5": "5b8462a3f82d188c", "6": "80f79e9e2011a3e3", "7": "4654167fd211d027", "8": "50acfa00fe353440", "9": "c6d2f770737823f1", "10": "44f2ca451aea24be", "11": "feaac5ab67a8c17a", "12": "b71bd92e5edbf2e0", "13": "219d65ba6d2689e4", "14": "8e30bb8112fd02d1", "15": "03906ab80b99db85", "16": "5d51c60ba1bc2be0", "17": "a1da914f522dcd21", "18": "fbad840891b90569", "19": "0686ff2d6fe29fb3", "20": "060baa9e1924b465", "21": "a5c8f2c3080b8243", "22": "0d76852f1d29eeb1", "23": "6ff68619f2d72924", "24": "3da11675eee7ec46", "25": "a6696589e9556f97", "26": "6c752999e8a4d4b6", "27": "2d4e13ea2111d750", "28": "4b60cb0ee1ac186a", "29": "f56fbca9bb8235ec", "30": "c7beed940704509f", "31": "38be2d254fb31ae8", "32": "ee5596fcf7e7c0b3", "33": "e4d4e0a440bc599f", "34": "c897e49c001acdae", "35": "3aad272a2cf5d495", "36": "0a197439d306b956", "37": "f43acf5c8b1329af", "38": "6775efafc9b33338", "39": "8176a164778526f9", "40": "66b69189c0acc3ff", "41": "0322ff824966a4d8", "42": "784c9e3d336a7f53", "43": "4b8bb6c3f7b64856", "44": "18c0ff3e6225bcb2", "45": "b18defdea7d53e37", "46": "54c0fceb01591230", "47": "0fad42a4989aa7e3", "48": "0237e1e02ee47a27", "49": "0d0f9f015921feef", "50": "8d0c81e5ca23e9a6", "51": "f79963571b9c15ee", "52": "5935824c825606cb", "53": "9685f9cbe158e50b", "54": "3d5ab759f350bc79", "55": "d549f24931a990e9", "56": "3cc031dcb648797c", "57": "a0ab88e6c629251d", "58": "76bd6412e2a22ecd", "59": "54827845564490c9", "60": "0a9736c416c0c6b9", "61": "77358620ac528153", "62": "3b0c585df09df48a", "63": "7e78cd3b28828c20", "64": "1c0c958231735f61", "65": "60b0f81225f62f69", "66": "920754c65cc94b88", "67": "df911472140a9b94", "68": "8e17bc11bcea91b9", "69": "7e905b75e4f28b95", "70": "a28424eca5d36c55", "71": "2cdb53d5b6051ab6", "72": "e42fbd3dc744e730", "73": "7fe2cac980de160c", "74": "2b1343a6a9db1487", "75": "54a1bb232f1d4ceb", "76": "442ba11d31ec0e0a", "77": "852a25b8b95bf8d1", "78": "1810ab370b9cd608", "79": "0fc5dca02a3f02f6", "80": "6ff8a97e63c9a2f3", "81": "a38f84ae3d895236", "82": "08e48bd11f9714df", "83": "5095122914e83cf5", "84": "1aef305bd7d7d63f", "85": "f8bfd0cfe9e8b478", "86": "410d15a346bd5894", "87": "6b41d288cfd834ab", "88": "5aa6db96312a8811", "89": "80225792bb62ba04", "90": "fd291228c3311f40", "91": "d4579c5b7aa2742a", "92": "7b9ba7c3bff11361", "93": "71cd9c1fa4a857f0", "94": "34cd980be3c32d21", "95": "970093453f3b7d90", "96": "9e96780a2b7c4bd6", "97": "b7c10b0e09caac0b", "98": "089ea6a55861c693", "99": "cb6165a7dc822d29", "100": "8c97d8c400895e15", "101": "bd22248d0516e5d4", "102": "6aa00d5a83295f11", "103": "f58668f5b10ccdeb", "104": "4ec787414cc6f50b", "105": "1cf3077fd45d874a", "106": "edcd5d9bb3c4b00f", "107": "37f2f47110fe3eaa", "108": "b7ad5abb1da8cf8d", "109": "cb48a9c4f54efa38", "110": "f6dd36fd7f3edbe5", "111": "2925b620f0b1fd17", "112": "d8b3099917c3b711", "113": "3bb61caa0302c804", "114": "0d4f1d08dd056bb9", "115": "4ac2dcddeec2ff11", "116": "07da9aae9668f573", "117": "196f63e0c4536d30", "118": "ade84262e3cfac12", "119": "ebe4e5e0c42c613f", "120": "27256931b19a2867", "121": "5396e68ca6c185ad", "122": "aa8a1de55696b666", "123": "96618c9a362af46c", "124": "83f104cbb62fd03e", "125": "6db738fb27190349", "126": "6a087a22cbcef972", "127": "85fd71a0cad8d3a5", "128": "22dd4feed96c4229", "129": "c4d2f60f532e6f16", "130": "f6b0aa8a1568926b", "131": "142d0db70bad18fe", "132": "67ea4284cbc02c54", "133": "d899cfc86c7a4a27", "134": "ba9464410a9b4168", "135": "0d496a12149eca27", "136": "521f5c7b9d566b4d", "137": "9e37828bdd2ba8c5", "138": "ec03c97194c56f91", "139": "f4e6d5dfa30034c5", "140": "d9b47fa423cf0748", "141": "1e0330b8757f333e", "142": "f35d75e1194c008d", "143": "4d2ae7190b514a34", "144": "4a98716cabf43f86", "145": "56b2431193739c38", "146": "edc785fd71bb0675", "147": "4c7347f8f86e1fbd", "148": "8ca77cc4fd6fd437", "149": "8968e9e7d55afcbe", "150": "f4f4ce1a1180ddb1", "151": "e426746f6e9ee15f", "152": "b2337918dbdaf257", "153": "a3593e6f45bafb20", "154": "56747bad6345d66b", "155": "a8e7498fa7e257df", "156": "56e7b2355898077f", "157": "f2fc88f7d8214711", "158": "c966f6f8570c8c29", "159": "5f6094aa385f3bfe", "160": "2834e7d59672e756", "161": "cc6e436d94fd0033", "162": "64f33a2fc8969cd2", "163": "26ac1c0a00eabce1", "164": "cdcea44a6805ae55", "165": "a6e35f4937a5000c", "166": "85c97dab928b9b1b", "167": "7ab5695391e32126"}
{"0": "36bdb6f09c457f7c", "1": "8c5bf6244cf710c6", "2": "efbcc9c62a3ee78b", "3": "8599153989b07faa", "4": "b5952a1f7fee9f20", "5": "5b8462a3f82d188c", "6": "80f79e9e2011a3e3", "7": "4654167fd211d027", "8": "50acfa00fe353440", "9": "c6d2f770737823f1", "10": "44f2ca451aea24be", "11": "feaac5ab67a8c17a", "12": "b71bd92e5edbf2e0", "13": "219d65ba6d2689e4", "14": "8e30bb8112fd02d1", "15": "03906ab80b99db85", "16": "5d51c60ba1bc2be0", "17": "a1da914f522dcd21", "18": "fbad840891b90569", "19": "0686ff2d6fe29fb3", "20": "060baa9e1924b465", "21": "a5c8f2c3080b8243", "22": "0d76852f1d29eeb1", "23": "6ff68619f2d72924", "24": "3da11675eee7ec46", "25": "a6696589e9556f97", "26": "6c752999e8a4d4b6", "27": "2d4e13ea2111d750", "28": "4b60cb0ee1ac186a", "29": "f56fbca9bb8235ec", "30": "c7beed940704509f", "31": "38be2d254fb31ae8", "32": "ee5596fcf7e7c0b3", "33": "e4d4e0a440bc599f", "34": "c897e49c001acdae", "35": "3aad272a2cf5d495", "36": "0a197439d306b956", "37": "f43acf5c8b1329af", "38": "6775efafc9b33338", "39": "8176a164778526f9", "40": "66b69189c0acc3ff", "41": "0322ff824966a4d8", "42": "784c9e3d336a7f53", "43": "4b8bb6c3f7b64856", "44": "18c0ff3e6225bcb2", "45": "d5eb5f4efd73cafb", "46": "57116576996271f5", "47": "0fad42a4989aa7e3", "48": "0237e1e02ee47a27", "49": "0d0f9f015921feef", "50": "8d0c81e5ca23e9a6", "51": "f79963571b9c15ee", "52": "5935824c825606cb", "53": "9685f9cbe158e50b", "54": "3d5ab759f350bc79", "55": "d549f24931a990e9", "56": "3cc031dcb648797c", "57": "a0ab88e6c629251d", "58": "76bd6412e2a22ecd", "59": "54827845564490c9", "60": "0a9736c416c0c6b9", "61": "77358620ac528153", "62": "3b0c585df09df48a", "63": "7e78cd3b28828c20", "64": "1c0c958231735f61", "65": "60b0f81225f62f69", "66": "920754c65cc94b88", "67": "df911472140a9b94", "68": "8e17bc11bcea91b9", "69": "7e905b75e4f28b95", "70": "a28424eca5d36c55", "71": "2cdb53d5b6051ab6", "72": "e42fbd3dc744e730", "73": "7fe2cac980de160c", "74": "2b1343a6a9db1487", "75": "54a1bb232f1d4ceb", "76": "442ba11d31ec0e0a", "77": "852a25b8b95bf8d1", "78": "1810ab370b9cd608", "79": "0fc5dca02a3f02f6", "80": "6ff8a97e63c9a2f3", "81": "a38f84ae3d895236", "82": "08e48bd11f9714df", "83": "5095122914e83cf5", "84": "1aef305bd7d7d63f", "85": "f8bfd0cfe9e8b478", "86": "410d15a346bd5894", "87": "6b41d288cfd834ab", "88": "5aa6db96312a8811", "89": "80225792bb62ba04", "90": "fd291228c3311f40", "91": "d4579c5b7aa2742a", "92": "7b9ba7c3bff11361", "93": "71cd9c1fa4a857f0", "94": "34cd980be3c32d21", "95": "970093453f3b7d90", "96": "9e96780a2b7c4bd6", "97": "b7c10b0e09caac0b", "98": "089ea6a55861c693", "99": "cb6165a7dc822d29", "100": "7bdb2c6abfbde762", "101": "3e1a3e8ca5030d57", "102": "6aa00d5a83295f11", "103": "f58668f5b10ccdeb", "104": "4ec787414cc6f50b", "105": "1cf3077fd45d874a", "106": "edcd5d9bb3c4b00f", "107": "37f2f47110fe3eaa", "108": "b7ad5abb1da8cf8d", "109": "cb48a9c4f54efa38", "110": "f6dd36fd7f3edbe5", "111": "2925b620f0b1fd17", "112": "d8b3099917c3b711", "113": "3bb61caa0302c804", "114": "0d4f1d08dd056bb9", "115": "4ac2dcddeec2ff11", "116": "07da9aae9668f573", "117": "196f63e0c4536d30", "118": "ade84262e3cfac12", "119": "ebe4e5e0c42c613f", "120": "16f0543249fafdd8", "121": "5396e68ca6c185ad", "122": "d1eeebf358bcab60", "123": "96618c9a362af46c", "124": "83f104cbb62fd03e", "125": "6db738fb27190349", "126": "6a087a22cbcef972", "127": "85fd71a0cad8d3a5", "128": "22dd4feed96c4229", "129": "c4d2f60f532e6f16", "130": "f6b0aa8a1568926b", "131": "142d0db70bad18fe", "132": "67ea4284cbc02c54", "133": "d899cfc86c7a4a27", "134": "ba9464410a9b4168", "135": "0d496a12149eca27", "136": "521f5c7b9d566b4d", "137": "9e37828bdd2ba8c5", "138": "ec03c97194c56f91", "139": "f4e6d5dfa30034c5", "140": "d9b47fa423cf0748", "141": "1e0330b8757f333e", "142": "f35d75e1194c008d", "143": "4d2ae7190b514a34", "144": "4a98716cabf43f86", "145": "56b2431193739c38", "146": "edc785fd71bb0675", "147": "4c7347f8f86e1fbd", "148": "8ca77cc4fd6fd437", "149": "8968e9e7d55afcbe", "150": "f4f4ce1a1180ddb1", "151": "e426746f6e9ee15f", "152": "4c30720833331d86", "153": "a3593e6f45bafb20", "154": "56747bad6345d66b", "155": "a8e7498fa7e257df", "156": "56e7b2355898077f", "157": "f2fc88f7d8214711", "158": "c966f6f8570c8c29", "159": "5f6094aa385f3bfe", "160": "2834e7d59672e756", "161": "cc6e436d94fd0033", "162": "64f33a2fc8969cd2", "163": "26ac1c0a00eabce1", "164": "cdcea44a6805ae55", "165": "1e217fe7a21a2501", "166": "85c97dab928b9b1b", "167": "7ab5695391e32126"}
@@ -148,7 +148,7 @@
"146": "Specification Quality Checklist: Convert Article JSON to Markdown",
"147": "CLI Contract: `convert_article_to_markdown.py`",
"148": "9. Interface CLI",
"149": "get_hl_gl_ceid",
"149": "sample_rss_xml",
"150": "13. Estratégia de testes",
"151": "6. Contrato de entrada",
"152": "InherenceClassifier",
@@ -164,7 +164,7 @@
"162": "test_normalize_date_iso_8601_variants",
"163": "test_metadata_priority_original_url_all_fallbacks",
"164": "test_normalize_scalar_non_string_types",
"165": "LLMFallbackAdapter",
"165": ".disambiguate",
"166": "remove_duplicate_initial_h1",
"167": "test_normalize_scalar_whitespace_collapsing"
}
+19 -19
View File
@@ -1,7 +1,7 @@
# Graph Report - TextNLPClassifierApp (2026-08-21)
## Corpus Check
- 201 files · ~110,162 words
- 201 files · ~110,615 words
- Verdict: corpus is large enough that graph structure adds value.
## Summary
@@ -10,7 +10,7 @@
- Token cost: 0 input · 0 output
## Graph Freshness
- Built from commit: `cb33dafa`
- Built from commit: `31152d50`
- Run `git rev-parse HEAD` and compare to check if the graph is stale.
- Run `graphify update .` after code changes (no API cost).
@@ -157,7 +157,7 @@
- Specification Quality Checklist: Convert Article JSON to Markdown
- CLI Contract: `convert_article_to_markdown.py`
- 9. Interface CLI
- get_hl_gl_ceid
- sample_rss_xml
- 13. Estratégia de testes
- 6. Contrato de entrada
- InherenceClassifier
@@ -173,7 +173,7 @@
- test_normalize_date_iso_8601_variants
- test_metadata_priority_original_url_all_fallbacks
- test_normalize_scalar_non_string_types
- LLMFallbackAdapter
- .disambiguate
- remove_duplicate_initial_h1
- test_normalize_scalar_whitespace_collapsing
@@ -192,7 +192,7 @@
## Surprising Connections (you probably didn't know these)
- `main()` --uses--> `ECPSnapshot` [INFERRED]
classify.py → src/models.py
- `test_e2e_extract_google_news_live_pipeline()` --uses--> `ExtractionResult` [INFERRED]
- `test_extract_google_news_orchestration_mocked()` --uses--> `ExtractionResult` [INFERRED]
tests/test_extract_google_news.py → scripts/extract_google_news.py
- `test_llm_adapter_interface()` --calls--> `LLMFallbackAdapter` [EXTRACTED]
tests/test_adapters.py → src/adapters/llm.py
@@ -371,16 +371,16 @@ Cohesion: 0.08
Nodes (24): 1. Visão geral (arquitetura), 2.1 DTO de entrada (`googlenews_etl/application/dtos/extract_news_dto.py`), 2.2 Value Object de validação (`googlenews_etl/domain/entities/search_query.py`), 2. Entrada, 3.1 O caso de uso (`googlenews_etl/application/use_cases/extract_news_use_case.py`), 3.2 A porta (`googlenews_etl/domain/ports/news_extractor_port.py`), 3.3.1 Inicialização: sessão HTTP com impersonação de browser, 3.3.2 Mapeamento idioma → parâmetros `hl`/`gl` (`_get_hl_gl`) (+16 more)
### Community 82 - "extract_google_news.py"
Cohesion: 0.20
Nodes (14): extract_google_news(), _fetch_rss_content(), NewsArticle, _normalize_text_for_comparison(), parse_google_news_rss(), Remove pontuação e espaços extras para comparação de redundância., Parseia o XML do RSS do Google News e extrai os itens estruturados., Resolve em paralelo as URLs intermediárias do Google News para os links finais… (+6 more)
Cohesion: 0.15
Nodes (18): extract_google_news(), _fetch_rss_content(), get_hl_gl_ceid(), NewsArticle, _normalize_text_for_comparison(), parse_google_news_rss(), Mapeia idioma e locale para os parâmetros hl, gl e ceid do Google News., Remove pontuação e espaços extras para comparação de redundância. (+10 more)
### Community 83 - "ExtractionResult"
Cohesion: 0.29
Nodes (5): ExtractionResult, Any, Resultado consolidado da extração., Valida a consolidação do ExtractionResult a partir da busca mockada com URLs…, test_extract_google_news_orchestration_mocked()
Nodes (5): ExtractionResult, Any, Resultado consolidado da extração., Valida E2E o fluxo completo de busca, parsing e resolução de URLs reais ao vivo., test_e2e_extract_google_news_live_pipeline()
### Community 84 - "test_extract_google_news.py"
Cohesion: 0.16
Nodes (14): Resolve a URL intermediária do Google News para a URL real do veículo., resolve_article_url(), fixture, Testes unitários e de integração para o Extrator de Manchetes do Google News.…, Valida o parsing do feed RSS, higienização de tags HTML e deduplicação., Valida fallback gracioso de URL quando não é link do Google News ou em erro., Valida resolução bem-sucedida de URL do Google News para o portal destino., Valida E2E que o decodificador resolve uma URL real do Google News para o… (+6 more)
Cohesion: 0.15
Nodes (15): Resolve a URL intermediária do Google News para a URL real do veículo., resolve_article_url(), Testes unitários e de integração para o Extrator de Manchetes do Google News.…, Valida fallback gracioso de URL quando não é link do Google News ou em erro., Valida resolução bem-sucedida de URL do Google News para o portal destino., Valida E2E que o decodificador resolve uma URL real do Google News para o…, Valida o mapeamento padrão de idiomas para pares (hl, gl, ceid)., Valida a sobrescrita geográfica quando o argumento locale é especificado. (+7 more)
### Community 85 - "Implementation Tasks: Google News Headlines Extractor"
Cohesion: 0.14
@@ -400,7 +400,7 @@ Nodes (7): Architecture & Pipeline, Documentation (this feature), Implementation
### Community 90 - "SearchQuery"
Cohesion: 0.20
Nodes (6): Value Object com parâmetros de busca validados., SearchQuery, Valida E2E o fluxo completo de busca, parsing e resolução de URLs reais ao vivo., Valida as regras de negócio e limites de SearchQuery., test_e2e_extract_google_news_live_pipeline(), test_search_query_validation()
Nodes (6): Value Object com parâmetros de busca validados., SearchQuery, Valida a consolidação do ExtractionResult a partir da busca mockada com URLs…, Valida as regras de negócio e limites de SearchQuery., test_extract_google_news_orchestration_mocked(), test_search_query_validation()
### Community 91 - "1. Technical Decisions & Tradeoffs"
Cohesion: 0.25
@@ -622,9 +622,9 @@ Nodes (5): 1. Script Signature, 2. Command-Line Arguments, 3. Exit Codes, 4. Sta
Cohesion: 0.40
Nodes (5): 9.1 Script, 9.2 Argumentos, 9.3 Exemplos, 9.4 Saída do processo, 9. Interface CLI
### Community 149 - "get_hl_gl_ceid"
Cohesion: 0.25
Nodes (8): get_hl_gl_ceid(), Mapeia idioma e locale para os parâmetros hl, gl e ceid do Google News., Valida o mapeamento padrão de idiomas para pares (hl, gl, ceid)., Valida a sobrescrita geográfica quando o argumento locale é especificado., Valida fallback dinâmico para idiomas regionais não listados explicitamente., test_get_hl_gl_ceid_default_mappings(), test_get_hl_gl_ceid_dynamic_fallback(), test_get_hl_gl_ceid_with_custom_locale()
### Community 149 - "sample_rss_xml"
Cohesion: 0.67
Nodes (3): fixture, Fixture que fornece o conteúdo do XML de exemplo para testes offline., sample_rss_xml()
### Community 150 - "13. Estratégia de testes"
Cohesion: 0.50
@@ -635,8 +635,8 @@ Cohesion: 0.50
Nodes (4): 6.1 Formato, 6.2 Valores aceitos para `selected_extractor`, 6.3 Campos obrigatórios após a resolução, 6. Contrato de entrada
### Community 152 - "InherenceClassifier"
Cohesion: 0.14
Nodes (24): InherenceClassifier, Tier 1 Deterministic NLP Entity Inherence Classifier with optional Tier 2 /…, DecisionCategory, Unit tests for deterministic classification decision logic., test_contextual_inherent(), test_direct_inherent(), test_negative_anchor_suppression(), test_not_related() (+16 more)
Cohesion: 0.11
Nodes (31): LLMFallbackAdapter, Optional adapter for LLM fallback boundary disambiguation., InherenceClassifier, Any, Tier 1 Deterministic NLP Entity Inherence Classifier with optional Tier 2 /…, DecisionCategory, Unit tests for deterministic classification decision logic., test_contextual_inherent() (+23 more)
### Community 153 - "convert_html_to_markdown"
Cohesion: 0.33
@@ -650,9 +650,9 @@ Nodes (3): 1. Input JSON Schema, 2. Output JSON Schema, JSON Schema Contract: De
Cohesion: 0.67
Nodes (3): 5.1 Incluído, 5.2 Fora do escopo, 5. Escopo
### Community 165 - "LLMFallbackAdapter"
Cohesion: 0.13
Nodes (11): LLMFallbackAdapter, Optional adapter for LLM fallback boundary disambiguation., Returns True if an API key or custom provider function is configured., Constructs a structured disambiguation prompt for the LLM., Executes LLM fallback for ambiguous boundary cases. Returns a refined…, Parses and validates structured JSON response from LLM., Any, Valida detecção de disponibilidade por chave de API ou provider customizado. (+3 more)
### Community 165 - ".disambiguate"
Cohesion: 0.25
Nodes (4): Executes LLM fallback for ambiguous boundary cases. Returns a refined…, Parses and validates structured JSON response from LLM., Returns True if an API key or custom provider function is configured., Constructs an expert-engineered prompt for multilingual entity inherence…
### Community 166 - "remove_duplicate_initial_h1"
Cohesion: 0.50
File diff suppressed because it is too large Load Diff
+6 -6
View File
@@ -330,9 +330,9 @@
"semantic_hash": ""
},
"src/adapters/llm.py": {
"mtime": 1787320047.0352886,
"seen": 1787320239.0258572,
"ast_hash": "ffcf26cb4e89395aa8771ddb4f5de785",
"mtime": 1787320797.2485664,
"seen": 1787320818.2953389,
"ast_hash": "a5cd6f66048ee1d443c2c91ae9947a14",
"semantic_hash": ""
},
"src/classifier.py": {
@@ -912,9 +912,9 @@
"semantic_hash": ""
},
"tests/test_llm_fallback.py": {
"mtime": 1787320089.2120655,
"seen": 1787320239.0310187,
"ast_hash": "74bc17c4668551e77214d6581baf206c",
"mtime": 1787320797.249565,
"seen": 1787320818.2967606,
"ast_hash": "e5d98de814ceeecd8bd601a7c206e92d",
"semantic_hash": ""
}
}
+37 -37
View File
@@ -1,16 +1,16 @@
# Graph Report - TextNLPClassifierApp (2026-08-21)
## Corpus Check
- 201 files · ~110,615 words
- 202 files · ~112,197 words
- Verdict: corpus is large enough that graph structure adds value.
## Summary
- 1554 nodes · 1970 edges · 168 communities (120 shown, 48 thin omitted)
- Extraction: 97% EXTRACTED · 3% INFERRED · 0% AMBIGUOUS · INFERRED: 59 edges (avg confidence: 0.95)
- 1579 nodes · 2030 edges · 168 communities (120 shown, 48 thin omitted)
- Extraction: 96% EXTRACTED · 4% INFERRED · 0% AMBIGUOUS · INFERRED: 73 edges (avg confidence: 0.95)
- Token cost: 0 input · 0 output
## Graph Freshness
- Built from commit: `31152d50`
- Built from commit: `bae14405`
- Run `git rev-parse HEAD` and compare to check if the graph is stale.
- Run `graphify update .` after code changes (no API cost).
@@ -110,7 +110,7 @@
- 🧠 TextNLPClassifierApp
- Extraction Pipeline Checklist: Article Content Multi-Engine Extractor
- parametrize
- models.py
- main
- ECPSnapshot
- Feature Specification: Multilingual NLP Entity Inherence Classifier (POC)
- 4. Requisitos Funcionais (FR)
@@ -160,7 +160,7 @@
- sample_rss_xml
- 13. Estratégia de testes
- 6. Contrato de entrada
- InherenceClassifier
- LLMFallbackAdapter
- convert_html_to_markdown
- JSON Schema Contract: Deterministic Article Content Selection
- 5. Escopo
@@ -173,16 +173,16 @@
- test_normalize_date_iso_8601_variants
- test_metadata_priority_original_url_all_fallbacks
- test_normalize_scalar_non_string_types
- .disambiguate
- InherenceClassifier
- remove_duplicate_initial_h1
- test_normalize_scalar_whitespace_collapsing
## God Nodes (most connected - your core abstractions)
1. `ECPSnapshot` - 40 edges
2. `InherenceClassifier` - 29 edges
3. `DecisionCategory` - 28 edges
4. `LLMFallbackAdapter` - 26 edges
5. `ClassificationResult` - 24 edges
1. `ECPSnapshot` - 49 edges
2. `InherenceClassifier` - 36 edges
3. `DecisionCategory` - 35 edges
4. `LLMFallbackAdapter` - 32 edges
5. `ClassificationResult` - 26 edges
6. `select_article_extractor()` - 23 edges
7. `ExtractorName` - 21 edges
8. `PRD — Conversão de artigo JSON para Markdown` - 16 edges
@@ -347,12 +347,12 @@ Cohesion: 0.29
Nodes (6): 1.1 Arguments & Options, 1. Command Line Interface, 2.1 Exit Codes, 2.2 Standard Output (`stdout`) / Standard Error (`stderr`), 2. Standard Streams & Exit Codes, CLI Contract & Interface Specification (POC)
### Community 45 - "ClassificationResult"
Cohesion: 0.11
Nodes (17): ABC, BaseNLPAdapter, Base abstract adapter interface for optional Tier 2 / Tier 3 NLP enhancers., Abstract interface for pluggable NLP classification adapters., Return True if the underlying provider or model is installed and configured., Compute semantic similarity score between text and a set of candidate terms., Optionally refine an ambiguous classification result., LocalEmbeddingsAdapter (+9 more)
Cohesion: 0.10
Nodes (20): ABC, BaseNLPAdapter, Base abstract adapter interface for optional Tier 2 / Tier 3 NLP enhancers., Abstract interface for pluggable NLP classification adapters., Return True if the underlying provider or model is installed and configured., Compute semantic similarity score between text and a set of candidate terms., Optionally refine an ambiguous classification result., LocalEmbeddingsAdapter (+12 more)
### Community 46 - "test_adversarial.py"
Cohesion: 0.10
Nodes (19): Any, RelatedEntity, Adversarial and robustness test suite for Multilingual NLP Entity Inherence…, Run CLI via subprocess without --output and verify stdout is pure parseable…, Run CLI via subprocess with empty content and verify error code and exit code., Content about city/state governance of São Paulo against ECP for São Paulo FC., Run CLI via subprocess with missing target_name and verify error payload., Run CLI via subprocess with corrupted JSON and verify error payload. (+11 more)
Cohesion: 0.11
Nodes (19): RelatedEntity, Adversarial and robustness test suite for Multilingual NLP Entity Inherence…, Run CLI via subprocess without --output and verify stdout is pure parseable…, Run CLI via subprocess with empty content and verify error code and exit code., Run CLI via subprocess with missing target_name and verify error payload., Run CLI via subprocess with corrupted JSON and verify error payload., Content about apple fruit/culinary recipe against Apple Inc. tech entity., High-weight related entity mentioned in passing without required domain anchors. (+11 more)
### Community 47 - "classifier.py"
Cohesion: 0.16
@@ -434,13 +434,13 @@ Nodes (34): 1. Requirement Completeness, 2. Requirement Clarity & Non-Ambiguity,
Cohesion: 0.22
Nodes (9): parametrize, Garante aceitação de URLs absolutas com esquema HTTP e HTTPS válidos., Garante rejeição de esquemas não permitidos, URLs relativas e strings vazias., Garante que a ausência de corpo no extrator selecionado NUNCA faça fallback…, Garante que todos os placeholders documentados no PRD sejam descartados…, test_normalize_scalar_placeholders_discarded(), test_resolve_article_body_strict_isolation_all_extractors(), test_validate_url_invalid_schemes() (+1 more)
### Community 100 - "models.py"
Cohesion: 0.15
Nodes (17): emit_error(), main(), parse_args(), Namespace, ClassificationError, ErrorCode, MatchedGraphEntity, Enum (+9 more)
### Community 100 - "main"
Cohesion: 0.17
Nodes (15): emit_error(), main(), parse_args(), Namespace, ClassificationError, ErrorCode, Enum, str (+7 more)
### Community 101 - "ECPSnapshot"
Cohesion: 0.24
Nodes (10): ECPSnapshot, classifier(), fixture, parametrize, Controlled 24-case benchmark suite for Multilingual NLP Entity Inherence…, test_benchmark_case(), Unit tests for ECP models, schema validation, and structured error handling., test_ecp_snapshot_defaults() (+2 more)
Cohesion: 0.18
Nodes (12): ECPSnapshot, Any, classifier(), fixture, parametrize, Controlled 24-case benchmark suite for Multilingual NLP Entity Inherence…, test_benchmark_case(), Unit tests for ECP models, schema validation, and structured error handling. (+4 more)
### Community 102 - "Feature Specification: Multilingual NLP Entity Inherence Classifier (POC)"
Cohesion: 0.14
@@ -520,7 +520,7 @@ Nodes (16): detect_language(), extract_words(), normalize_text(), Lightweight mu
### Community 122 - "test_select_article_extractor.py"
Cohesion: 0.18
Nodes (16): generate_shingles(), normalize_text(), Executa a normalização determinística para comparação: 1. Decodificar entidades…, Gera conjunto de shingles ordenados de tamanho window_size (padrão 5). - Se…, Suíte de Testes Automatizados para o Seletor Determinístico de Extrator. Cobre…, Garante que marcação de imagem Markdown ![alt](url) seja descartada e link…, E2E: Executa scripts/select_article_extractor.py como subprocesso real na linha…, test_e2e_cli_subprocess_real_execution() (+8 more)
Nodes (16): generate_shingles(), normalize_text(), Executa a normalização determinística para comparação: 1. Decodificar entidades…, Gera conjunto de shingles ordenados de tamanho window_size (padrão 5). - Se…, Suíte de Testes Automatizados para o Seletor Determinístico de Extrator. Cobre…, Garante que marcação de imagem Markdown ![alt](url) seja descartada e link…, E2E: Executa CLI com arquivo inexistente e valida código 1 e mensagem no stderr., test_e2e_cli_subprocess_missing_file() (+8 more)
### Community 123 - "Feature Specification: Deterministic Content Selection"
Cohesion: 0.17
@@ -634,9 +634,9 @@ Nodes (4): 13.1 Testes unitários, 13.2 Testes de integração do CLI, 13.3 Caso
Cohesion: 0.50
Nodes (4): 6.1 Formato, 6.2 Valores aceitos para `selected_extractor`, 6.3 Campos obrigatórios após a resolução, 6. Contrato de entrada
### Community 152 - "InherenceClassifier"
Cohesion: 0.11
Nodes (31): LLMFallbackAdapter, Optional adapter for LLM fallback boundary disambiguation., InherenceClassifier, Any, Tier 1 Deterministic NLP Entity Inherence Classifier with optional Tier 2 /…, DecisionCategory, Unit tests for deterministic classification decision logic., test_contextual_inherent() (+23 more)
### Community 152 - "LLMFallbackAdapter"
Cohesion: 0.08
Nodes (26): LLMFallbackAdapter, Executes LLM fallback for ambiguous boundary cases. Returns a refined…, Parses and validates structured JSON response from LLM., Optional adapter for LLM fallback boundary disambiguation., Returns True if an API key or custom provider function is configured., Constructs an expert-engineered prompt for multilingual entity inherence…, Any, Suíte de Testes para o Adaptador de Fallback para LLM (Tier 3) do Classificador… (+18 more)
### Community 153 - "convert_html_to_markdown"
Cohesion: 0.33
@@ -650,9 +650,9 @@ Nodes (3): 1. Input JSON Schema, 2. Output JSON Schema, JSON Schema Contract: De
Cohesion: 0.67
Nodes (3): 5.1 Incluído, 5.2 Fora do escopo, 5. Escopo
### Community 165 - ".disambiguate"
Cohesion: 0.25
Nodes (4): Executes LLM fallback for ambiguous boundary cases. Returns a refined…, Parses and validates structured JSON response from LLM., Returns True if an API key or custom provider function is configured., Constructs an expert-engineered prompt for multilingual entity inherence…
### Community 165 - "InherenceClassifier"
Cohesion: 0.11
Nodes (30): InherenceClassifier, Tier 1 Deterministic NLP Entity Inherence Classifier with optional Tier 2 /…, DecisionCategory, Content about city/state governance of São Paulo against ECP for São Paulo FC., test_adversarial_sao_paulo_city_vs_fc(), Unit tests for deterministic classification decision logic., test_contextual_inherent(), test_direct_inherent() (+22 more)
### Community 166 - "remove_duplicate_initial_h1"
Cohesion: 0.50
@@ -667,16 +667,16 @@ Nodes (4): Remove o primeiro título H1 do corpo somente quando ele for igual ao
_Questions this graph is uniquely positioned to answer:_
- **Why does `PRD — Conversão de artigo JSON para Markdown` connect `PRD — Conversão de artigo JSON para Markdown` to `8. Regras funcionais`, `12. Critérios de aceite`, `11. Requisitos não funcionais`, `9. Interface CLI`, `13. Estratégia de testes`, `6. Contrato de entrada`, `5. Escopo`?**
_High betweenness centrality (0.008) - this node is a cross-community bridge._
- **Why does `Implementation Plan: Convert Article JSON to Markdown` connect `Implementation Plan: Convert Article JSON to Markdown` to `005-convert-json-markdown/plan.md`?**
_High betweenness centrality (0.004) - this node is a cross-community bridge._
_High betweenness centrality (0.006) - this node is a cross-community bridge._
- **Why does `Tasks: Convert Article JSON to Markdown` connect `Tasks: Convert Article JSON to Markdown` to `005-convert-json-markdown/plan.md`?**
_High betweenness centrality (0.004) - this node is a cross-community bridge._
- **Are the 10 inferred relationships involving `ECPSnapshot` (e.g. with `main()` and `BaseNLPAdapter`) actually correct?**
_`ECPSnapshot` has 10 INFERRED edges - model-reasoned connections that need verification._
- **Are the 6 inferred relationships involving `InherenceClassifier` (e.g. with `LocalEmbeddingsAdapter` and `LLMFallbackAdapter`) actually correct?**
_`InherenceClassifier` has 6 INFERRED edges - model-reasoned connections that need verification._
- **Are the 18 inferred relationships involving `DecisionCategory` (e.g. with `LLMFallbackAdapter` and `InherenceClassifier`) actually correct?**
_`DecisionCategory` has 18 INFERRED edges - model-reasoned connections that need verification._
- **Why does `ECPSnapshot` connect `ECPSnapshot` to `main`, `InherenceClassifier`, `ClassificationResult`, `test_adversarial.py`, `classifier.py`, `LLMFallbackAdapter`?**
_High betweenness centrality (0.004) - this node is a cross-community bridge._
- **Are the 16 inferred relationships involving `ECPSnapshot` (e.g. with `main()` and `BaseNLPAdapter`) actually correct?**
_`ECPSnapshot` has 16 INFERRED edges - model-reasoned connections that need verification._
- **Are the 7 inferred relationships involving `InherenceClassifier` (e.g. with `LocalEmbeddingsAdapter` and `LLMFallbackAdapter`) actually correct?**
_`InherenceClassifier` has 7 INFERRED edges - model-reasoned connections that need verification._
- **Are the 24 inferred relationships involving `DecisionCategory` (e.g. with `LLMFallbackAdapter` and `InherenceClassifier`) actually correct?**
_`DecisionCategory` has 24 INFERRED edges - model-reasoned connections that need verification._
- **Are the 4 inferred relationships involving `LLMFallbackAdapter` (e.g. with `ClassificationResult` and `DecisionCategory`) actually correct?**
_`LLMFallbackAdapter` has 4 INFERRED edges - model-reasoned connections that need verification._
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
+1 -1
View File
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
+1332 -356
View File
File diff suppressed because it is too large Load Diff
+12 -6
View File
@@ -330,9 +330,9 @@
"semantic_hash": ""
},
"src/adapters/llm.py": {
"mtime": 1787320797.2485664,
"seen": 1787320818.2953389,
"ast_hash": "a5cd6f66048ee1d443c2c91ae9947a14",
"mtime": 1787321086.752706,
"seen": 1787321205.5144775,
"ast_hash": "21ac74a13ac5dfad7db498b17165f8b3",
"semantic_hash": ""
},
"src/classifier.py": {
@@ -654,9 +654,9 @@
"semantic_hash": ""
},
"README.md": {
"mtime": 1787320219.5852203,
"seen": 1787320239.0933797,
"ast_hash": "ce59670fbaebc5e408a30a1009a58d12",
"mtime": 1787321187.0812356,
"seen": 1787321205.5201268,
"ast_hash": "aedfaf7a245288227952a2e28e7e7b13",
"semantic_hash": ""
},
"scripts/extract_article_contents.py": {
@@ -916,5 +916,11 @@
"seen": 1787320818.2967606,
"ast_hash": "e5d98de814ceeecd8bd601a7c206e92d",
"semantic_hash": ""
},
"tests/test_e2e_text_analysis_pipeline.py": {
"mtime": 1787321086.751707,
"seen": 1787321205.5156026,
"ast_hash": "3a2d47f2ffcf8371ffdf90bb797b5346",
"semantic_hash": ""
}
}
+74 -6
View File
@@ -1,19 +1,44 @@
"""Optional LLM fallback adapter (Tier 3).
Disabled by default. Provides fallback interface for boundary disambiguation
without requiring external API keys for core POC execution.
supporting OpenAI and Gemini APIs with automatic .env loading and custom provider functions.
"""
from __future__ import annotations
import json
import os
import urllib.error
import urllib.request
from pathlib import Path
from typing import Callable, Optional
from src.adapters.base import BaseNLPAdapter
from src.models import ClassificationResult, DecisionCategory, ECPSnapshot
def _load_env_file() -> None:
"""Carrega variáveis do arquivo .env na raiz do projeto se existir."""
for parent in [Path.cwd(), Path(__file__).parent.parent.parent]:
env_file = parent / ".env"
if env_file.is_file():
try:
for line in env_file.read_text(encoding="utf-8").splitlines():
line = line.strip()
if line and not line.startswith("#") and "=" in line:
key, val = line.split("=", 1)
key = key.strip()
val = val.strip().strip("'\"")
if key and key not in os.environ:
os.environ[key] = val
except Exception:
pass
break
_load_env_file()
class LLMFallbackAdapter(BaseNLPAdapter):
"""Optional adapter for LLM fallback boundary disambiguation."""
@@ -23,13 +48,14 @@ class LLMFallbackAdapter(BaseNLPAdapter):
api_key: str | None = None,
provider_fn: Optional[Callable[[str], str]] = None,
) -> None:
self.model_name = model_name
self.api_key = api_key or os.environ.get("OPENAI_API_KEY")
self.model_name = os.environ.get("OPENAI_MODEL", model_name)
self.openai_api_key = api_key or os.environ.get("OPENAI_API_KEY")
self.gemini_api_key = os.environ.get("GEMINI_API_KEY") or os.environ.get("GOOGLE_API_KEY")
self.provider_fn = provider_fn
def is_available(self) -> bool:
"""Returns True if an API key or custom provider function is configured."""
return bool(self.api_key or self.provider_fn)
return bool(self.openai_api_key or self.gemini_api_key or self.provider_fn)
def evaluate_similarity(self, text: str, terms: list[str]) -> float:
return 0.0
@@ -135,12 +161,54 @@ Respond ONLY with a valid JSON object matching this schema:
prompt = self.build_prompt(ecp, content_md, initial_result)
# If custom provider function is provided (e.g. for testing or custom runtime)
# 1. Custom provider function (testing or custom runtime)
if self.provider_fn is not None:
raw_response = self.provider_fn(prompt)
return self._parse_llm_response(raw_response, initial_result)
# Stub default for POC when only API key string is present without active SDK
# 2. Real OpenAI execution
if self.openai_api_key:
try:
from openai import OpenAI
client = OpenAI(api_key=self.openai_api_key)
response = client.chat.completions.create(
model=self.model_name,
messages=[{"role": "user", "content": prompt}],
response_format={"type": "json_object"},
temperature=0.0,
)
raw_text = response.choices[0].message.content or ""
return self._parse_llm_response(raw_text, initial_result)
except Exception as e:
initial_result.warnings.append(f"OpenAI fallback invocation error: {e}")
return None
# 3. Real Gemini execution via REST API
if self.gemini_api_key:
try:
gemini_model = os.environ.get("GEMINI_MODEL", "gemini-2.5-flash")
url = f"https://generativelanguage.googleapis.com/v1beta/models/{gemini_model}:generateContent?key={self.gemini_api_key}"
payload = {
"contents": [{"parts": [{"text": prompt}]}],
"generationConfig": {
"responseMimeType": "application/json",
"temperature": 0.0,
},
}
req = urllib.request.Request(
url,
data=json.dumps(payload).encode("utf-8"),
headers={"Content-Type": "application/json"},
)
with urllib.request.urlopen(req, timeout=30) as resp:
data = json.loads(resp.read().decode("utf-8"))
raw_text = data["candidates"][0]["content"]["parts"][0]["text"]
return self._parse_llm_response(raw_text, initial_result)
except Exception as e:
initial_result.warnings.append(f"Gemini fallback invocation error: {e}")
return None
return None
def _parse_llm_response(
+388
View File
@@ -0,0 +1,388 @@
"""
Suíte de Testes E2E e de Integração Completa para Análise de Texto e Classificação de Inerência (QA Sênior).
Valida todo o funil de análise de texto:
1. Sucesso no NLP Determinístico (Tier 1 DIRECT_INHERENT com bypass de LLM).
2. Insucesso / Rejeição no NLP (Tier 1 NOT_RELATED por âncoras negativas e homônimos).
3. Ambiguidade / Limiar detectada no NLP e resolvida com sucesso no LLM (Tier 3 Upgrade).
4. Ambiguidade confirmada pelo LLM como TANGENTIAL (Tier 3 Confirmation).
5. Insucesso / Falha de API de LLM com degradação graciosa para Tier 1.
6. Cobertura Multilíngue E2E nos 6 idiomas (PT, EN, ES, DE, IT, FR).
7. Execução E2E via CLI subprocess com contratos de entrada, saída e flags.
8. Chamada real ao vivo a provedores de LLM (OpenAI/Gemini) quando credenciais estiverem disponíveis.
"""
from __future__ import annotations
import json
import subprocess
import sys
from pathlib import Path
import pytest
from src.adapters.llm import LLMFallbackAdapter
from src.classifier import InherenceClassifier
from src.models import (
ClassificationResult,
DecisionCategory,
ECPSnapshot,
RelatedEntity,
)
CLASSIFY_CLI = Path(__file__).parent.parent / "classify.py"
# ==============================================================================
# Fixtures e Helpers para o Funil de Teste E2E
# ==============================================================================
@pytest.fixture
def ecp_river_plate() -> ECPSnapshot:
"""Fixture ECP oficial para o Club Atlético River Plate."""
return ECPSnapshot(
target_entity_id="ecp_river_plate",
target_name="Club Atlético River Plate",
aliases=[
"Club Atlético River Plate",
"River Plate",
"River",
"El Millonario",
"La Banda",
"CARP",
],
domain="Futebol / Esportes",
anchors=[
"fútbol",
"Copa Libertadores",
"Libertadores",
"Copa Sudamericana",
"Sudamericana",
"Monumental",
"Estadio Monumental",
"Eduardo Coudet",
"Coudet",
"Nicolás Otamendi",
"Otamendi",
"Rafael Santos Borré",
],
negative_anchors=[
"River Plate de Montevideo",
"River Plate de Asunción",
"River de Piauí",
"Rio da Prata",
"Bacia do Rio da Prata",
],
related_entities=[
RelatedEntity(
entity_id="estadio_monumental",
name="Estadio Mâs Monumental",
relation_type="HOME_VENUE_OF",
weight=0.95,
aliases=["Monumental", "El Monumental"],
),
RelatedEntity(
entity_id="copa_sudamericana",
name="Copa Sudamericana",
relation_type="COMPETES_IN",
weight=0.85,
aliases=["Sudamericana"],
),
],
)
# ==============================================================================
# 1. Funil de Sucesso NLP Determinístico (Tier 1)
# ==============================================================================
def test_funnel_nlp_deterministic_success(ecp_river_plate: ECPSnapshot):
"""
Cenário 1: Artigo com alta densidade de âncoras do River Plate.
Oráculo: Decisão DIRECT_INHERENT, confiança alta (>=0.95), LLM não é chamado.
"""
content = """
# River Plate vence com autoridade na Copa Sudamericana
Em noite histórica no Estadio Monumental, o River dominou a partida sob o comando de Eduardo Coudet.
Otamendi e Borré marcaram os gols que garantiram a classificação na Sudamericana.
"""
llm_called = {"status": False}
def mock_llm(prompt: str) -> str:
llm_called["status"] = True
return json.dumps({"decision": "DIRECT_INHERENT", "confidence": 0.99})
adapter = LLMFallbackAdapter(provider_fn=mock_llm)
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
result = classifier.classify(ecp_river_plate, content)
assert result.decision == DecisionCategory.DIRECT_INHERENT
assert result.is_inherent is True
assert result.confidence >= 0.95
assert "River Plate" in result.matched_anchors or "River" in result.matched_anchors
assert len(result.graph_matches) >= 1
# Otimização de custo: LLM NÃO deve ser chamado em casos determinísticos claros
assert llm_called["status"] is False
# ==============================================================================
# 2. Funil de Insucesso / Rejeição NLP (Tier 1 Negativas e Homônimos)
# ==============================================================================
def test_funnel_nlp_deterministic_rejection_homonym(ecp_river_plate: ECPSnapshot):
"""
Cenário 2: Artigo sobre a Bacia do Rio da Prata ou clube homônimo do Uruguai.
Oráculo: Decisão NOT_RELATED, is_inherent=False, âncoras negativas detectadas.
"""
content = """
# Expedição ambiental navega pela Bacia do Rio da Prata
Pesquisadores mapearam a biodiversidade fluvial e os sedimentos do Rio da Prata durante o verão.
"""
classifier = InherenceClassifier(enable_llm=False)
result = classifier.classify(ecp_river_plate, content)
assert result.decision == DecisionCategory.NOT_RELATED
assert result.is_inherent is False
assert any("Rio da Prata" in neg for neg in result.negative_matches)
assert "Negative anchor" in result.rationale
# ==============================================================================
# 3. Funil de Ambiguidade NLP -> Resolução com Sucesso no LLM (Tier 3)
# ==============================================================================
def test_funnel_nlp_ambiguity_resolved_by_llm_upgrade(ecp_river_plate: ECPSnapshot):
"""
Cenário 3: Menção isolada do clube ('River') em contexto com poucas âncoras explícitas.
Tier 1 preliminar: TANGENTIAL (baixa densidade).
LLM Fallback: Analisa o contexto profundo e eleva para DIRECT_INHERENT.
"""
ambiguous_content = """
# Bastidores do mercado sul-americano
A diretoria do River finalizou os últimos detalhes contratuais para a renovação de jovens promessas.
"""
mock_llm_response = json.dumps(
{
"analysis_summary": "O artigo trata da gestão de elenco e renovações contratuais do clube River Plate.",
"decision": "DIRECT_INHERENT",
"confidence": 0.94,
"rationale": "A análise contextual profunda comprova que a matéria é focada na administração do River Plate.",
}
)
adapter = LLMFallbackAdapter(provider_fn=lambda prompt: mock_llm_response)
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
result = classifier.classify(ecp_river_plate, ambiguous_content)
assert result.decision == DecisionCategory.DIRECT_INHERENT
assert result.is_inherent is True
assert result.confidence == 0.94
assert "[Tier 3 LLM]" in result.rationale
assert "[Tier 3 LLM Override applied]" in result.warnings
# ==============================================================================
# 4. Funil de Ambiguidade NLP -> Confirmação de Tangencial no LLM
# ==============================================================================
def test_funnel_nlp_ambiguity_confirmed_tangential_by_llm(ecp_river_plate: ECPSnapshot):
"""
Cenário 4: Menção metafórica ou turística a um local próximo.
Tier 1 preliminar: TANGENTIAL.
LLM Fallback: Confirma que é meramente periférico/ilustrativo.
"""
tangential_content = """
# Melhores restaurantes do bairro de Núñez em Buenos Aires
Ao visitar a capital portenha, próximo de onde fica o River, você encontra excelentes opções gastronômicas.
"""
mock_llm_response = json.dumps(
{
"analysis_summary": "Guia gastronômico sobre o bairro de Núñez com citação geográfica casual ao clube.",
"decision": "TANGENTIAL",
"confidence": 0.96,
"rationale": "A entidade é usada apenas como ponto de referência geográfica em um artigo sobre restaurantes.",
}
)
adapter = LLMFallbackAdapter(provider_fn=lambda prompt: mock_llm_response)
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
result = classifier.classify(ecp_river_plate, tangential_content)
assert result.decision == DecisionCategory.TANGENTIAL
assert result.is_inherent is False
assert result.confidence == 0.96
assert "[Tier 3 LLM]" in result.rationale
# ==============================================================================
# 5. Funil de Insucesso / Degradação Graciosa em Falha do LLM
# ==============================================================================
def test_funnel_llm_failure_graceful_degradation(ecp_river_plate: ECPSnapshot):
"""
Cenário 5: LLM configurado, caso ambíguo, mas a API externa sofre timeout/500.
Oráculo: Mantém o resultado do Tier 1 determinístico com warning detalhado e sem quebrar.
"""
def broken_llm(prompt: str) -> str:
raise TimeoutError("Conexão com serviço de LLM excedeu 30 segundos.")
adapter = LLMFallbackAdapter(provider_fn=broken_llm)
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
content = "O River esteve presente no evento de inauguração da praça."
result = classifier.classify(ecp_river_plate, content)
# Mantém o Tier 1 determinístico
assert result.decision == DecisionCategory.TANGENTIAL
assert result.is_inherent is False
assert any("LLM fallback failed" in w for w in result.warnings)
# ==============================================================================
# 6. Cobertura Multilíngue nos 6 Idiomas (PT, EN, ES, DE, IT, FR)
# ==============================================================================
@pytest.mark.parametrize(
"lang_code,content,expected_lang",
[
("pt", "# Petrobras anuncia perfuração no pré-sal com tecnologia nacional.", "pt"),
("en", "# Apple unveils new generative AI features for upcoming devices.", "en"),
(
"es",
"# River Plate prepara su viaje a Bogotá para disputar el torneo continental.",
"es",
),
(
"de",
"# Volkswagen investiert Milliarden in neue Batterie-Fabriken in Deutschland.",
"de",
),
("it", "# Ferrari conquista la pole position nel Gran Premio di Monza.", "it"),
(
"fr",
"# L'entreprise TotalEnergies accélère ses investissements solaires en France.",
"fr",
),
],
)
def test_funnel_multilingual_language_detection(
lang_code: str, content: str, expected_lang: str, ecp_river_plate: ECPSnapshot
):
"""Garante a identificação precisa de idioma e integridade nos 6 idiomas suportados."""
classifier = InherenceClassifier(enable_llm=False)
result = classifier.classify(ecp_river_plate, content)
assert result.detected_language == expected_lang
# ==============================================================================
# 7. Execução E2E via CLI Subprocess
# ==============================================================================
def test_funnel_cli_subprocess_end_to_end(tmp_path: Path):
"""Valida o contrato CLI completo classify.py com saída em arquivo JSON e flags ativas."""
ecp_path = tmp_path / "ecp.json"
ecp_path.write_text(
json.dumps(
{
"target_entity_id": "ecp_test_e2e",
"target_name": "Clube Teste",
"aliases": ["Clube Teste", "Clube"],
"domain": "Esportes",
"anchors": ["campeonato", "vitória", "torneio"],
}
),
encoding="utf-8",
)
doc_path = tmp_path / "artigo.md"
doc_path.write_text(
"# Clube Teste comemora vitória histórica no campeonato\n\nEquipe foi campeã do torneio.",
encoding="utf-8",
)
out_path = tmp_path / "resultado.json"
res = subprocess.run(
[
sys.executable,
str(CLASSIFY_CLI),
"--ecp",
str(ecp_path),
"--content",
str(doc_path),
"--output",
str(out_path),
"--enable-llm",
],
capture_output=True,
text=True,
)
assert res.returncode == 0, f"Erro na CLI: {res.stderr}"
assert out_path.exists()
payload = json.loads(out_path.read_text(encoding="utf-8"))
assert payload["decision"] == "DIRECT_INHERENT"
assert payload["is_inherent"] is True
assert payload["confidence"] >= 0.85
assert isinstance(payload["matched_anchors"], list)
assert isinstance(payload["evidence"], list)
# ==============================================================================
# 8. Teste Live Opt-In com API Real (OpenAI / Gemini) se .env Estiver Presente
# ==============================================================================
def test_funnel_live_api_execution_if_configured():
"""
Executa chamada ao vivo contra OpenAI ou Gemini caso OPENAI_API_KEY ou GEMINI_API_KEY
esteja configurada no ambiente ou no arquivo .env.
"""
adapter = LLMFallbackAdapter()
if not (adapter.openai_api_key or adapter.gemini_api_key):
pytest.skip(
"Chaves de API reais (OPENAI_API_KEY ou GEMINI_API_KEY) não configuradas no .env"
)
ecp = ECPSnapshot(
target_entity_id="ecp_live_test",
target_name="Club Atlético River Plate",
aliases=["River Plate", "River"],
domain="Futebol",
anchors=["Monumental", "Libertadores"],
)
initial_res = ClassificationResult(
decision=DecisionCategory.TANGENTIAL,
is_inherent=False,
confidence=0.40,
detected_language="es",
matched_anchors=["River"],
negative_matches=[],
graph_matches=[],
evidence=["River"],
rationale="Passing mention detected by Tier 1.",
warnings=[],
)
# Texto de teste para a API ao vivo
content = "O River Plate empatou em 1 a 1 em Bogotá com gols de Otamendi na Copa Sul-Americana."
refined = adapter.disambiguate(ecp, content, initial_res)
assert refined is not None
assert refined.decision in [
DecisionCategory.DIRECT_INHERENT,
DecisionCategory.CONTEXTUAL_INHERENT,
]
assert refined.is_inherent is True
assert "[Tier 3 LLM]" in refined.rationale