test(qa): add exhaustive 38-scenario test suite covering happy, sad, borderline, LLM fallback, CLI contracts, and multilingual matrix
This commit is contained in:
@@ -501,7 +501,8 @@ TextNLPClassifierApp/
|
||||
│ ├── test_select_article_extractor.py
|
||||
│ ├── test_convert_article_to_markdown.py # Testes da conversão para Markdown
|
||||
│ ├── test_llm_fallback.py # Testes do Tier 3 LLM Fallback
|
||||
│ └── test_e2e_text_analysis_pipeline.py # Suíte E2E do Funil de Análise e Fallback
|
||||
│ ├── test_e2e_text_analysis_pipeline.py # Suíte E2E do Funil de Análise e Fallback
|
||||
│ └── test_classify_exhaustive_suite.py # Suíte Exaustiva de Casos Felizes/Infelizes (QA Sênior)
|
||||
├── requirements.txt # Dependências do projeto
|
||||
├── pyproject.toml # Configurações de ferramentas (pytest, ruff, mypy)
|
||||
└── README.md # Documentação principal
|
||||
@@ -511,12 +512,15 @@ TextNLPClassifierApp/
|
||||
|
||||
## 🧪 Testes e Qualidade de Código
|
||||
|
||||
O repositório possui **209 testes automatizados** com 100% de aprovação cobrindo testes unitários, de regressão, de integração, Golden Fixtures exatas, testes de sensibilidade de mutação, testes de fallback para LLM (Tier 3), validações de degradação graciosa e testes End-to-End (E2E) via CLI subprocess:
|
||||
O repositório possui **247 testes automatizados** com 100% de aprovação cobrindo testes unitários, de regressão, de integração, Golden Fixtures exatas, testes de sensibilidade de mutação, testes de fallback para LLM (Tier 3), validações de degradação graciosa, matriz multilíngue e testes End-to-End (E2E) via CLI subprocess:
|
||||
|
||||
```bash
|
||||
# Executar toda a suíte de testes do projeto (209 testes)
|
||||
# Executar toda a suíte de testes do projeto (247 testes)
|
||||
pytest -v
|
||||
|
||||
# Executar a Suíte Exaustiva de Classificação e Fallback (38 testes)
|
||||
pytest tests/test_classify_exhaustive_suite.py -v
|
||||
|
||||
# Executar a Suíte E2E do Funil de Análise de Texto e Fallback para LLM
|
||||
pytest tests/test_e2e_text_analysis_pipeline.py -v
|
||||
|
||||
|
||||
@@ -46,7 +46,7 @@
|
||||
"44": "2. Standard Streams & Exit Codes",
|
||||
"45": "ClassificationResult",
|
||||
"46": "test_adversarial.py",
|
||||
"47": "classifier.py",
|
||||
"47": "LLMFallbackAdapter",
|
||||
"48": "test_convert_article_to_markdown.py",
|
||||
"49": "content_northvolt_de.md",
|
||||
"50": "content_presal_pt.md",
|
||||
@@ -100,7 +100,7 @@
|
||||
"98": "Extraction Pipeline Checklist: Article Content Multi-Engine Extractor",
|
||||
"99": "parametrize",
|
||||
"100": "main",
|
||||
"101": "ECPSnapshot",
|
||||
"101": "classifier.py",
|
||||
"102": "Feature Specification: Multilingual NLP Entity Inherence Classifier (POC)",
|
||||
"103": "4. Requisitos Funcionais (FR)",
|
||||
"104": "Tasks: Article Content Multi-Engine Extractor",
|
||||
@@ -120,7 +120,7 @@
|
||||
"118": "Tasks: Deterministic Article Content Selection",
|
||||
"119": "select_article_extractor",
|
||||
"120": "process_batch",
|
||||
"121": "detect_language",
|
||||
"121": "test_models.py",
|
||||
"122": "test_select_article_extractor.py",
|
||||
"123": "Feature Specification: Deterministic Content Selection",
|
||||
"124": "2. Entity Descriptions & Fields",
|
||||
@@ -148,10 +148,10 @@
|
||||
"146": "Specification Quality Checklist: Convert Article JSON to Markdown",
|
||||
"147": "CLI Contract: `convert_article_to_markdown.py`",
|
||||
"148": "9. Interface CLI",
|
||||
"149": "sample_rss_xml",
|
||||
"149": "get_hl_gl_ceid",
|
||||
"150": "13. Estratégia de testes",
|
||||
"151": "6. Contrato de entrada",
|
||||
"152": "LLMFallbackAdapter",
|
||||
"152": "ECPSnapshot",
|
||||
"153": "convert_html_to_markdown",
|
||||
"154": "JSON Schema Contract: Deterministic Article Content Selection",
|
||||
"155": "5. Escopo",
|
||||
@@ -166,5 +166,7 @@
|
||||
"164": "test_normalize_scalar_non_string_types",
|
||||
"165": "InherenceClassifier",
|
||||
"166": "remove_duplicate_initial_h1",
|
||||
"167": "test_normalize_scalar_whitespace_collapsing"
|
||||
"167": "test_normalize_scalar_whitespace_collapsing",
|
||||
"168": ".disambiguate",
|
||||
"169": "test_funnel_cli_subprocess_end_to_end"
|
||||
}
|
||||
|
||||
@@ -1 +1 @@
|
||||
{"0": "36bdb6f09c457f7c", "1": "8c5bf6244cf710c6", "2": "efbcc9c62a3ee78b", "3": "8599153989b07faa", "4": "b5952a1f7fee9f20", "5": "5b8462a3f82d188c", "6": "80f79e9e2011a3e3", "7": "4654167fd211d027", "8": "50acfa00fe353440", "9": "c6d2f770737823f1", "10": "44f2ca451aea24be", "11": "feaac5ab67a8c17a", "12": "b71bd92e5edbf2e0", "13": "219d65ba6d2689e4", "14": "8e30bb8112fd02d1", "15": "03906ab80b99db85", "16": "5d51c60ba1bc2be0", "17": "a1da914f522dcd21", "18": "fbad840891b90569", "19": "0686ff2d6fe29fb3", "20": "060baa9e1924b465", "21": "a5c8f2c3080b8243", "22": "0d76852f1d29eeb1", "23": "6ff68619f2d72924", "24": "3da11675eee7ec46", "25": "a6696589e9556f97", "26": "6c752999e8a4d4b6", "27": "2d4e13ea2111d750", "28": "4b60cb0ee1ac186a", "29": "f56fbca9bb8235ec", "30": "c7beed940704509f", "31": "38be2d254fb31ae8", "32": "ee5596fcf7e7c0b3", "33": "e4d4e0a440bc599f", "34": "c897e49c001acdae", "35": "3aad272a2cf5d495", "36": "0a197439d306b956", "37": "f43acf5c8b1329af", "38": "6775efafc9b33338", "39": "8176a164778526f9", "40": "66b69189c0acc3ff", "41": "0322ff824966a4d8", "42": "784c9e3d336a7f53", "43": "4b8bb6c3f7b64856", "44": "18c0ff3e6225bcb2", "45": "d5eb5f4efd73cafb", "46": "57116576996271f5", "47": "0fad42a4989aa7e3", "48": "0237e1e02ee47a27", "49": "0d0f9f015921feef", "50": "8d0c81e5ca23e9a6", "51": "f79963571b9c15ee", "52": "5935824c825606cb", "53": "9685f9cbe158e50b", "54": "3d5ab759f350bc79", "55": "d549f24931a990e9", "56": "3cc031dcb648797c", "57": "a0ab88e6c629251d", "58": "76bd6412e2a22ecd", "59": "54827845564490c9", "60": "0a9736c416c0c6b9", "61": "77358620ac528153", "62": "3b0c585df09df48a", "63": "7e78cd3b28828c20", "64": "1c0c958231735f61", "65": "60b0f81225f62f69", "66": "920754c65cc94b88", "67": "df911472140a9b94", "68": "8e17bc11bcea91b9", "69": "7e905b75e4f28b95", "70": "a28424eca5d36c55", "71": "2cdb53d5b6051ab6", "72": "e42fbd3dc744e730", "73": "7fe2cac980de160c", "74": "2b1343a6a9db1487", "75": "54a1bb232f1d4ceb", "76": "442ba11d31ec0e0a", "77": "852a25b8b95bf8d1", "78": "1810ab370b9cd608", "79": "0fc5dca02a3f02f6", "80": "6ff8a97e63c9a2f3", "81": "a38f84ae3d895236", "82": "08e48bd11f9714df", "83": "5095122914e83cf5", "84": "1aef305bd7d7d63f", "85": "f8bfd0cfe9e8b478", "86": "410d15a346bd5894", "87": "6b41d288cfd834ab", "88": "5aa6db96312a8811", "89": "80225792bb62ba04", "90": "fd291228c3311f40", "91": "d4579c5b7aa2742a", "92": "7b9ba7c3bff11361", "93": "71cd9c1fa4a857f0", "94": "34cd980be3c32d21", "95": "970093453f3b7d90", "96": "9e96780a2b7c4bd6", "97": "b7c10b0e09caac0b", "98": "089ea6a55861c693", "99": "cb6165a7dc822d29", "100": "7bdb2c6abfbde762", "101": "3e1a3e8ca5030d57", "102": "6aa00d5a83295f11", "103": "f58668f5b10ccdeb", "104": "4ec787414cc6f50b", "105": "1cf3077fd45d874a", "106": "edcd5d9bb3c4b00f", "107": "37f2f47110fe3eaa", "108": "b7ad5abb1da8cf8d", "109": "cb48a9c4f54efa38", "110": "f6dd36fd7f3edbe5", "111": "2925b620f0b1fd17", "112": "d8b3099917c3b711", "113": "3bb61caa0302c804", "114": "0d4f1d08dd056bb9", "115": "4ac2dcddeec2ff11", "116": "07da9aae9668f573", "117": "196f63e0c4536d30", "118": "ade84262e3cfac12", "119": "ebe4e5e0c42c613f", "120": "16f0543249fafdd8", "121": "5396e68ca6c185ad", "122": "d1eeebf358bcab60", "123": "96618c9a362af46c", "124": "83f104cbb62fd03e", "125": "6db738fb27190349", "126": "6a087a22cbcef972", "127": "85fd71a0cad8d3a5", "128": "22dd4feed96c4229", "129": "c4d2f60f532e6f16", "130": "f6b0aa8a1568926b", "131": "142d0db70bad18fe", "132": "67ea4284cbc02c54", "133": "d899cfc86c7a4a27", "134": "ba9464410a9b4168", "135": "0d496a12149eca27", "136": "521f5c7b9d566b4d", "137": "9e37828bdd2ba8c5", "138": "ec03c97194c56f91", "139": "f4e6d5dfa30034c5", "140": "d9b47fa423cf0748", "141": "1e0330b8757f333e", "142": "f35d75e1194c008d", "143": "4d2ae7190b514a34", "144": "4a98716cabf43f86", "145": "56b2431193739c38", "146": "edc785fd71bb0675", "147": "4c7347f8f86e1fbd", "148": "8ca77cc4fd6fd437", "149": "8968e9e7d55afcbe", "150": "f4f4ce1a1180ddb1", "151": "e426746f6e9ee15f", "152": "4c30720833331d86", "153": "a3593e6f45bafb20", "154": "56747bad6345d66b", "155": "a8e7498fa7e257df", "156": "56e7b2355898077f", "157": "f2fc88f7d8214711", "158": "c966f6f8570c8c29", "159": "5f6094aa385f3bfe", "160": "2834e7d59672e756", "161": "cc6e436d94fd0033", "162": "64f33a2fc8969cd2", "163": "26ac1c0a00eabce1", "164": "cdcea44a6805ae55", "165": "1e217fe7a21a2501", "166": "85c97dab928b9b1b", "167": "7ab5695391e32126"}
|
||||
{"0": "36bdb6f09c457f7c", "1": "8c5bf6244cf710c6", "2": "efbcc9c62a3ee78b", "3": "8599153989b07faa", "4": "b5952a1f7fee9f20", "5": "5b8462a3f82d188c", "6": "80f79e9e2011a3e3", "7": "4654167fd211d027", "8": "50acfa00fe353440", "9": "c6d2f770737823f1", "10": "44f2ca451aea24be", "11": "feaac5ab67a8c17a", "12": "b71bd92e5edbf2e0", "13": "219d65ba6d2689e4", "14": "8e30bb8112fd02d1", "15": "03906ab80b99db85", "16": "5d51c60ba1bc2be0", "17": "a1da914f522dcd21", "18": "fbad840891b90569", "19": "0686ff2d6fe29fb3", "20": "060baa9e1924b465", "21": "a5c8f2c3080b8243", "22": "0d76852f1d29eeb1", "23": "6ff68619f2d72924", "24": "3da11675eee7ec46", "25": "a6696589e9556f97", "26": "6c752999e8a4d4b6", "27": "2d4e13ea2111d750", "28": "4b60cb0ee1ac186a", "29": "f56fbca9bb8235ec", "30": "c7beed940704509f", "31": "38be2d254fb31ae8", "32": "ee5596fcf7e7c0b3", "33": "e4d4e0a440bc599f", "34": "c897e49c001acdae", "35": "3aad272a2cf5d495", "36": "0a197439d306b956", "37": "f43acf5c8b1329af", "38": "6775efafc9b33338", "39": "8176a164778526f9", "40": "66b69189c0acc3ff", "41": "0322ff824966a4d8", "42": "784c9e3d336a7f53", "43": "4b8bb6c3f7b64856", "44": "18c0ff3e6225bcb2", "45": "943c894b7e96a921", "46": "b30963ae66d7e3c9", "47": "85bec2d6e2b742cf", "48": "0237e1e02ee47a27", "49": "0d0f9f015921feef", "50": "8d0c81e5ca23e9a6", "51": "f79963571b9c15ee", "52": "5935824c825606cb", "53": "9685f9cbe158e50b", "54": "3d5ab759f350bc79", "55": "d549f24931a990e9", "56": "3cc031dcb648797c", "57": "a0ab88e6c629251d", "58": "76bd6412e2a22ecd", "59": "54827845564490c9", "60": "0a9736c416c0c6b9", "61": "77358620ac528153", "62": "3b0c585df09df48a", "63": "7e78cd3b28828c20", "64": "1c0c958231735f61", "65": "60b0f81225f62f69", "66": "920754c65cc94b88", "67": "df911472140a9b94", "68": "8e17bc11bcea91b9", "69": "7e905b75e4f28b95", "70": "a28424eca5d36c55", "71": "2cdb53d5b6051ab6", "72": "e42fbd3dc744e730", "73": "7fe2cac980de160c", "74": "2b1343a6a9db1487", "75": "54a1bb232f1d4ceb", "76": "442ba11d31ec0e0a", "77": "852a25b8b95bf8d1", "78": "1810ab370b9cd608", "79": "0fc5dca02a3f02f6", "80": "6ff8a97e63c9a2f3", "81": "a38f84ae3d895236", "82": "dc6ddc157a3b9efb", "83": "a05140495d7a0353", "84": "24ca89fec34df075", "85": "f8bfd0cfe9e8b478", "86": "410d15a346bd5894", "87": "6b41d288cfd834ab", "88": "5aa6db96312a8811", "89": "80225792bb62ba04", "90": "e18a0a239fe528ba", "91": "d4579c5b7aa2742a", "92": "7b9ba7c3bff11361", "93": "71cd9c1fa4a857f0", "94": "34cd980be3c32d21", "95": "970093453f3b7d90", "96": "9e96780a2b7c4bd6", "97": "b7c10b0e09caac0b", "98": "089ea6a55861c693", "99": "cb6165a7dc822d29", "100": "8cb2e59dcf557313", "101": "75f2ab420202693e", "102": "6aa00d5a83295f11", "103": "f58668f5b10ccdeb", "104": "4ec787414cc6f50b", "105": "1cf3077fd45d874a", "106": "edcd5d9bb3c4b00f", "107": "37f2f47110fe3eaa", "108": "b7ad5abb1da8cf8d", "109": "cb48a9c4f54efa38", "110": "f6dd36fd7f3edbe5", "111": "2925b620f0b1fd17", "112": "d8b3099917c3b711", "113": "3bb61caa0302c804", "114": "0d4f1d08dd056bb9", "115": "4ac2dcddeec2ff11", "116": "07da9aae9668f573", "117": "196f63e0c4536d30", "118": "ade84262e3cfac12", "119": "ebe4e5e0c42c613f", "120": "27256931b19a2867", "121": "3d7cd9541766116e", "122": "aa8a1de55696b666", "123": "96618c9a362af46c", "124": "83f104cbb62fd03e", "125": "6db738fb27190349", "126": "6a087a22cbcef972", "127": "85fd71a0cad8d3a5", "128": "22dd4feed96c4229", "129": "c4d2f60f532e6f16", "130": "f6b0aa8a1568926b", "131": "142d0db70bad18fe", "132": "67ea4284cbc02c54", "133": "d899cfc86c7a4a27", "134": "ba9464410a9b4168", "135": "0d496a12149eca27", "136": "521f5c7b9d566b4d", "137": "9e37828bdd2ba8c5", "138": "ec03c97194c56f91", "139": "f4e6d5dfa30034c5", "140": "d9b47fa423cf0748", "141": "1e0330b8757f333e", "142": "f35d75e1194c008d", "143": "4d2ae7190b514a34", "144": "4a98716cabf43f86", "145": "56b2431193739c38", "146": "edc785fd71bb0675", "147": "4c7347f8f86e1fbd", "148": "8ca77cc4fd6fd437", "149": "09850697b717469a", "150": "f4f4ce1a1180ddb1", "151": "e426746f6e9ee15f", "152": "73cf7c102ed797ed", "153": "a3593e6f45bafb20", "154": "56747bad6345d66b", "155": "a8e7498fa7e257df", "156": "56e7b2355898077f", "157": "f2fc88f7d8214711", "158": "c966f6f8570c8c29", "159": "5f6094aa385f3bfe", "160": "2834e7d59672e756", "161": "cc6e436d94fd0033", "162": "64f33a2fc8969cd2", "163": "26ac1c0a00eabce1", "164": "cdcea44a6805ae55", "165": "f3f95b2d8f95c75f", "166": "85c97dab928b9b1b", "167": "7ab5695391e32126", "168": "eafca6a072d4f435", "169": "a68dbc0869da4c5e"}
|
||||
@@ -99,7 +99,7 @@
|
||||
"97": "🧠 TextNLPClassifierApp",
|
||||
"98": "Extraction Pipeline Checklist: Article Content Multi-Engine Extractor",
|
||||
"99": "parametrize",
|
||||
"100": "models.py",
|
||||
"100": "main",
|
||||
"101": "ECPSnapshot",
|
||||
"102": "Feature Specification: Multilingual NLP Entity Inherence Classifier (POC)",
|
||||
"103": "4. Requisitos Funcionais (FR)",
|
||||
@@ -151,7 +151,7 @@
|
||||
"149": "sample_rss_xml",
|
||||
"150": "13. Estratégia de testes",
|
||||
"151": "6. Contrato de entrada",
|
||||
"152": "InherenceClassifier",
|
||||
"152": "LLMFallbackAdapter",
|
||||
"153": "convert_html_to_markdown",
|
||||
"154": "JSON Schema Contract: Deterministic Article Content Selection",
|
||||
"155": "5. Escopo",
|
||||
@@ -164,7 +164,7 @@
|
||||
"162": "test_normalize_date_iso_8601_variants",
|
||||
"163": "test_metadata_priority_original_url_all_fallbacks",
|
||||
"164": "test_normalize_scalar_non_string_types",
|
||||
"165": ".disambiguate",
|
||||
"165": "InherenceClassifier",
|
||||
"166": "remove_duplicate_initial_h1",
|
||||
"167": "test_normalize_scalar_whitespace_collapsing"
|
||||
}
|
||||
|
||||
@@ -1,16 +1,16 @@
|
||||
# Graph Report - TextNLPClassifierApp (2026-08-21)
|
||||
|
||||
## Corpus Check
|
||||
- 201 files · ~110,615 words
|
||||
- 202 files · ~112,197 words
|
||||
- Verdict: corpus is large enough that graph structure adds value.
|
||||
|
||||
## Summary
|
||||
- 1554 nodes · 1970 edges · 168 communities (120 shown, 48 thin omitted)
|
||||
- Extraction: 97% EXTRACTED · 3% INFERRED · 0% AMBIGUOUS · INFERRED: 59 edges (avg confidence: 0.95)
|
||||
- 1579 nodes · 2030 edges · 168 communities (120 shown, 48 thin omitted)
|
||||
- Extraction: 96% EXTRACTED · 4% INFERRED · 0% AMBIGUOUS · INFERRED: 73 edges (avg confidence: 0.95)
|
||||
- Token cost: 0 input · 0 output
|
||||
|
||||
## Graph Freshness
|
||||
- Built from commit: `31152d50`
|
||||
- Built from commit: `bae14405`
|
||||
- Run `git rev-parse HEAD` and compare to check if the graph is stale.
|
||||
- Run `graphify update .` after code changes (no API cost).
|
||||
|
||||
@@ -110,7 +110,7 @@
|
||||
- 🧠 TextNLPClassifierApp
|
||||
- Extraction Pipeline Checklist: Article Content Multi-Engine Extractor
|
||||
- parametrize
|
||||
- models.py
|
||||
- main
|
||||
- ECPSnapshot
|
||||
- Feature Specification: Multilingual NLP Entity Inherence Classifier (POC)
|
||||
- 4. Requisitos Funcionais (FR)
|
||||
@@ -160,7 +160,7 @@
|
||||
- sample_rss_xml
|
||||
- 13. Estratégia de testes
|
||||
- 6. Contrato de entrada
|
||||
- InherenceClassifier
|
||||
- LLMFallbackAdapter
|
||||
- convert_html_to_markdown
|
||||
- JSON Schema Contract: Deterministic Article Content Selection
|
||||
- 5. Escopo
|
||||
@@ -173,16 +173,16 @@
|
||||
- test_normalize_date_iso_8601_variants
|
||||
- test_metadata_priority_original_url_all_fallbacks
|
||||
- test_normalize_scalar_non_string_types
|
||||
- .disambiguate
|
||||
- InherenceClassifier
|
||||
- remove_duplicate_initial_h1
|
||||
- test_normalize_scalar_whitespace_collapsing
|
||||
|
||||
## God Nodes (most connected - your core abstractions)
|
||||
1. `ECPSnapshot` - 40 edges
|
||||
2. `InherenceClassifier` - 29 edges
|
||||
3. `DecisionCategory` - 28 edges
|
||||
4. `LLMFallbackAdapter` - 26 edges
|
||||
5. `ClassificationResult` - 24 edges
|
||||
1. `ECPSnapshot` - 49 edges
|
||||
2. `InherenceClassifier` - 36 edges
|
||||
3. `DecisionCategory` - 35 edges
|
||||
4. `LLMFallbackAdapter` - 32 edges
|
||||
5. `ClassificationResult` - 26 edges
|
||||
6. `select_article_extractor()` - 23 edges
|
||||
7. `ExtractorName` - 21 edges
|
||||
8. `PRD — Conversão de artigo JSON para Markdown` - 16 edges
|
||||
@@ -347,12 +347,12 @@ Cohesion: 0.29
|
||||
Nodes (6): 1.1 Arguments & Options, 1. Command Line Interface, 2.1 Exit Codes, 2.2 Standard Output (`stdout`) / Standard Error (`stderr`), 2. Standard Streams & Exit Codes, CLI Contract & Interface Specification (POC)
|
||||
|
||||
### Community 45 - "ClassificationResult"
|
||||
Cohesion: 0.11
|
||||
Nodes (17): ABC, BaseNLPAdapter, Base abstract adapter interface for optional Tier 2 / Tier 3 NLP enhancers., Abstract interface for pluggable NLP classification adapters., Return True if the underlying provider or model is installed and configured., Compute semantic similarity score between text and a set of candidate terms., Optionally refine an ambiguous classification result., LocalEmbeddingsAdapter (+9 more)
|
||||
Cohesion: 0.10
|
||||
Nodes (20): ABC, BaseNLPAdapter, Base abstract adapter interface for optional Tier 2 / Tier 3 NLP enhancers., Abstract interface for pluggable NLP classification adapters., Return True if the underlying provider or model is installed and configured., Compute semantic similarity score between text and a set of candidate terms., Optionally refine an ambiguous classification result., LocalEmbeddingsAdapter (+12 more)
|
||||
|
||||
### Community 46 - "test_adversarial.py"
|
||||
Cohesion: 0.10
|
||||
Nodes (19): Any, RelatedEntity, Adversarial and robustness test suite for Multilingual NLP Entity Inherence…, Run CLI via subprocess without --output and verify stdout is pure parseable…, Run CLI via subprocess with empty content and verify error code and exit code., Content about city/state governance of São Paulo against ECP for São Paulo FC., Run CLI via subprocess with missing target_name and verify error payload., Run CLI via subprocess with corrupted JSON and verify error payload. (+11 more)
|
||||
Cohesion: 0.11
|
||||
Nodes (19): RelatedEntity, Adversarial and robustness test suite for Multilingual NLP Entity Inherence…, Run CLI via subprocess without --output and verify stdout is pure parseable…, Run CLI via subprocess with empty content and verify error code and exit code., Run CLI via subprocess with missing target_name and verify error payload., Run CLI via subprocess with corrupted JSON and verify error payload., Content about apple fruit/culinary recipe against Apple Inc. tech entity., High-weight related entity mentioned in passing without required domain anchors. (+11 more)
|
||||
|
||||
### Community 47 - "classifier.py"
|
||||
Cohesion: 0.16
|
||||
@@ -434,13 +434,13 @@ Nodes (34): 1. Requirement Completeness, 2. Requirement Clarity & Non-Ambiguity,
|
||||
Cohesion: 0.22
|
||||
Nodes (9): parametrize, Garante aceitação de URLs absolutas com esquema HTTP e HTTPS válidos., Garante rejeição de esquemas não permitidos, URLs relativas e strings vazias., Garante que a ausência de corpo no extrator selecionado NUNCA faça fallback…, Garante que todos os placeholders documentados no PRD sejam descartados…, test_normalize_scalar_placeholders_discarded(), test_resolve_article_body_strict_isolation_all_extractors(), test_validate_url_invalid_schemes() (+1 more)
|
||||
|
||||
### Community 100 - "models.py"
|
||||
Cohesion: 0.15
|
||||
Nodes (17): emit_error(), main(), parse_args(), Namespace, ClassificationError, ErrorCode, MatchedGraphEntity, Enum (+9 more)
|
||||
### Community 100 - "main"
|
||||
Cohesion: 0.17
|
||||
Nodes (15): emit_error(), main(), parse_args(), Namespace, ClassificationError, ErrorCode, Enum, str (+7 more)
|
||||
|
||||
### Community 101 - "ECPSnapshot"
|
||||
Cohesion: 0.24
|
||||
Nodes (10): ECPSnapshot, classifier(), fixture, parametrize, Controlled 24-case benchmark suite for Multilingual NLP Entity Inherence…, test_benchmark_case(), Unit tests for ECP models, schema validation, and structured error handling., test_ecp_snapshot_defaults() (+2 more)
|
||||
Cohesion: 0.18
|
||||
Nodes (12): ECPSnapshot, Any, classifier(), fixture, parametrize, Controlled 24-case benchmark suite for Multilingual NLP Entity Inherence…, test_benchmark_case(), Unit tests for ECP models, schema validation, and structured error handling. (+4 more)
|
||||
|
||||
### Community 102 - "Feature Specification: Multilingual NLP Entity Inherence Classifier (POC)"
|
||||
Cohesion: 0.14
|
||||
@@ -520,7 +520,7 @@ Nodes (16): detect_language(), extract_words(), normalize_text(), Lightweight mu
|
||||
|
||||
### Community 122 - "test_select_article_extractor.py"
|
||||
Cohesion: 0.18
|
||||
Nodes (16): generate_shingles(), normalize_text(), Executa a normalização determinística para comparação: 1. Decodificar entidades…, Gera conjunto de shingles ordenados de tamanho window_size (padrão 5). - Se…, Suíte de Testes Automatizados para o Seletor Determinístico de Extrator. Cobre…, Garante que marcação de imagem Markdown  seja descartada e link…, E2E: Executa scripts/select_article_extractor.py como subprocesso real na linha…, test_e2e_cli_subprocess_real_execution() (+8 more)
|
||||
Nodes (16): generate_shingles(), normalize_text(), Executa a normalização determinística para comparação: 1. Decodificar entidades…, Gera conjunto de shingles ordenados de tamanho window_size (padrão 5). - Se…, Suíte de Testes Automatizados para o Seletor Determinístico de Extrator. Cobre…, Garante que marcação de imagem Markdown  seja descartada e link…, E2E: Executa CLI com arquivo inexistente e valida código 1 e mensagem no stderr., test_e2e_cli_subprocess_missing_file() (+8 more)
|
||||
|
||||
### Community 123 - "Feature Specification: Deterministic Content Selection"
|
||||
Cohesion: 0.17
|
||||
@@ -634,9 +634,9 @@ Nodes (4): 13.1 Testes unitários, 13.2 Testes de integração do CLI, 13.3 Caso
|
||||
Cohesion: 0.50
|
||||
Nodes (4): 6.1 Formato, 6.2 Valores aceitos para `selected_extractor`, 6.3 Campos obrigatórios após a resolução, 6. Contrato de entrada
|
||||
|
||||
### Community 152 - "InherenceClassifier"
|
||||
Cohesion: 0.11
|
||||
Nodes (31): LLMFallbackAdapter, Optional adapter for LLM fallback boundary disambiguation., InherenceClassifier, Any, Tier 1 Deterministic NLP Entity Inherence Classifier with optional Tier 2 /…, DecisionCategory, Unit tests for deterministic classification decision logic., test_contextual_inherent() (+23 more)
|
||||
### Community 152 - "LLMFallbackAdapter"
|
||||
Cohesion: 0.08
|
||||
Nodes (26): LLMFallbackAdapter, Executes LLM fallback for ambiguous boundary cases. Returns a refined…, Parses and validates structured JSON response from LLM., Optional adapter for LLM fallback boundary disambiguation., Returns True if an API key or custom provider function is configured., Constructs an expert-engineered prompt for multilingual entity inherence…, Any, Suíte de Testes para o Adaptador de Fallback para LLM (Tier 3) do Classificador… (+18 more)
|
||||
|
||||
### Community 153 - "convert_html_to_markdown"
|
||||
Cohesion: 0.33
|
||||
@@ -650,9 +650,9 @@ Nodes (3): 1. Input JSON Schema, 2. Output JSON Schema, JSON Schema Contract: De
|
||||
Cohesion: 0.67
|
||||
Nodes (3): 5.1 Incluído, 5.2 Fora do escopo, 5. Escopo
|
||||
|
||||
### Community 165 - ".disambiguate"
|
||||
Cohesion: 0.25
|
||||
Nodes (4): Executes LLM fallback for ambiguous boundary cases. Returns a refined…, Parses and validates structured JSON response from LLM., Returns True if an API key or custom provider function is configured., Constructs an expert-engineered prompt for multilingual entity inherence…
|
||||
### Community 165 - "InherenceClassifier"
|
||||
Cohesion: 0.11
|
||||
Nodes (30): InherenceClassifier, Tier 1 Deterministic NLP Entity Inherence Classifier with optional Tier 2 /…, DecisionCategory, Content about city/state governance of São Paulo against ECP for São Paulo FC., test_adversarial_sao_paulo_city_vs_fc(), Unit tests for deterministic classification decision logic., test_contextual_inherent(), test_direct_inherent() (+22 more)
|
||||
|
||||
### Community 166 - "remove_duplicate_initial_h1"
|
||||
Cohesion: 0.50
|
||||
@@ -667,16 +667,16 @@ Nodes (4): Remove o primeiro título H1 do corpo somente quando ele for igual ao
|
||||
_Questions this graph is uniquely positioned to answer:_
|
||||
|
||||
- **Why does `PRD — Conversão de artigo JSON para Markdown` connect `PRD — Conversão de artigo JSON para Markdown` to `8. Regras funcionais`, `12. Critérios de aceite`, `11. Requisitos não funcionais`, `9. Interface CLI`, `13. Estratégia de testes`, `6. Contrato de entrada`, `5. Escopo`?**
|
||||
_High betweenness centrality (0.008) - this node is a cross-community bridge._
|
||||
- **Why does `Implementation Plan: Convert Article JSON to Markdown` connect `Implementation Plan: Convert Article JSON to Markdown` to `005-convert-json-markdown/plan.md`?**
|
||||
_High betweenness centrality (0.004) - this node is a cross-community bridge._
|
||||
_High betweenness centrality (0.006) - this node is a cross-community bridge._
|
||||
- **Why does `Tasks: Convert Article JSON to Markdown` connect `Tasks: Convert Article JSON to Markdown` to `005-convert-json-markdown/plan.md`?**
|
||||
_High betweenness centrality (0.004) - this node is a cross-community bridge._
|
||||
- **Are the 10 inferred relationships involving `ECPSnapshot` (e.g. with `main()` and `BaseNLPAdapter`) actually correct?**
|
||||
_`ECPSnapshot` has 10 INFERRED edges - model-reasoned connections that need verification._
|
||||
- **Are the 6 inferred relationships involving `InherenceClassifier` (e.g. with `LocalEmbeddingsAdapter` and `LLMFallbackAdapter`) actually correct?**
|
||||
_`InherenceClassifier` has 6 INFERRED edges - model-reasoned connections that need verification._
|
||||
- **Are the 18 inferred relationships involving `DecisionCategory` (e.g. with `LLMFallbackAdapter` and `InherenceClassifier`) actually correct?**
|
||||
_`DecisionCategory` has 18 INFERRED edges - model-reasoned connections that need verification._
|
||||
- **Why does `ECPSnapshot` connect `ECPSnapshot` to `main`, `InherenceClassifier`, `ClassificationResult`, `test_adversarial.py`, `classifier.py`, `LLMFallbackAdapter`?**
|
||||
_High betweenness centrality (0.004) - this node is a cross-community bridge._
|
||||
- **Are the 16 inferred relationships involving `ECPSnapshot` (e.g. with `main()` and `BaseNLPAdapter`) actually correct?**
|
||||
_`ECPSnapshot` has 16 INFERRED edges - model-reasoned connections that need verification._
|
||||
- **Are the 7 inferred relationships involving `InherenceClassifier` (e.g. with `LocalEmbeddingsAdapter` and `LLMFallbackAdapter`) actually correct?**
|
||||
_`InherenceClassifier` has 7 INFERRED edges - model-reasoned connections that need verification._
|
||||
- **Are the 24 inferred relationships involving `DecisionCategory` (e.g. with `LLMFallbackAdapter` and `InherenceClassifier`) actually correct?**
|
||||
_`DecisionCategory` has 24 INFERRED edges - model-reasoned connections that need verification._
|
||||
- **Are the 4 inferred relationships involving `LLMFallbackAdapter` (e.g. with `ClassificationResult` and `DecisionCategory`) actually correct?**
|
||||
_`LLMFallbackAdapter` has 4 INFERRED edges - model-reasoned connections that need verification._
|
||||
+1332
-356
File diff suppressed because it is too large
Load Diff
@@ -330,9 +330,9 @@
|
||||
"semantic_hash": ""
|
||||
},
|
||||
"src/adapters/llm.py": {
|
||||
"mtime": 1787320797.2485664,
|
||||
"seen": 1787320818.2953389,
|
||||
"ast_hash": "a5cd6f66048ee1d443c2c91ae9947a14",
|
||||
"mtime": 1787321086.752706,
|
||||
"seen": 1787321205.5144775,
|
||||
"ast_hash": "21ac74a13ac5dfad7db498b17165f8b3",
|
||||
"semantic_hash": ""
|
||||
},
|
||||
"src/classifier.py": {
|
||||
@@ -654,9 +654,9 @@
|
||||
"semantic_hash": ""
|
||||
},
|
||||
"README.md": {
|
||||
"mtime": 1787320219.5852203,
|
||||
"seen": 1787320239.0933797,
|
||||
"ast_hash": "ce59670fbaebc5e408a30a1009a58d12",
|
||||
"mtime": 1787321187.0812356,
|
||||
"seen": 1787321205.5201268,
|
||||
"ast_hash": "aedfaf7a245288227952a2e28e7e7b13",
|
||||
"semantic_hash": ""
|
||||
},
|
||||
"scripts/extract_article_contents.py": {
|
||||
@@ -916,5 +916,11 @@
|
||||
"seen": 1787320818.2967606,
|
||||
"ast_hash": "e5d98de814ceeecd8bd601a7c206e92d",
|
||||
"semantic_hash": ""
|
||||
},
|
||||
"tests/test_e2e_text_analysis_pipeline.py": {
|
||||
"mtime": 1787321086.751707,
|
||||
"seen": 1787321205.5156026,
|
||||
"ast_hash": "3a2d47f2ffcf8371ffdf90bb797b5346",
|
||||
"semantic_hash": ""
|
||||
}
|
||||
}
|
||||
@@ -1,16 +1,16 @@
|
||||
# Graph Report - TextNLPClassifierApp (2026-08-21)
|
||||
|
||||
## Corpus Check
|
||||
- 202 files · ~112,197 words
|
||||
- 203 files · ~114,894 words
|
||||
- Verdict: corpus is large enough that graph structure adds value.
|
||||
|
||||
## Summary
|
||||
- 1579 nodes · 2030 edges · 168 communities (120 shown, 48 thin omitted)
|
||||
- Extraction: 96% EXTRACTED · 4% INFERRED · 0% AMBIGUOUS · INFERRED: 73 edges (avg confidence: 0.95)
|
||||
- 1651 nodes · 2220 edges · 170 communities (122 shown, 48 thin omitted)
|
||||
- Extraction: 94% EXTRACTED · 6% INFERRED · 0% AMBIGUOUS · INFERRED: 127 edges (avg confidence: 0.95)
|
||||
- Token cost: 0 input · 0 output
|
||||
|
||||
## Graph Freshness
|
||||
- Built from commit: `bae14405`
|
||||
- Built from commit: `a874b98d`
|
||||
- Run `git rev-parse HEAD` and compare to check if the graph is stale.
|
||||
- Run `graphify update .` after code changes (no API cost).
|
||||
|
||||
@@ -58,7 +58,7 @@
|
||||
- 2. Standard Streams & Exit Codes
|
||||
- ClassificationResult
|
||||
- test_adversarial.py
|
||||
- classifier.py
|
||||
- LLMFallbackAdapter
|
||||
- test_convert_article_to_markdown.py
|
||||
- content_northvolt_de.md
|
||||
- content_presal_pt.md
|
||||
@@ -111,7 +111,7 @@
|
||||
- Extraction Pipeline Checklist: Article Content Multi-Engine Extractor
|
||||
- parametrize
|
||||
- main
|
||||
- ECPSnapshot
|
||||
- classifier.py
|
||||
- Feature Specification: Multilingual NLP Entity Inherence Classifier (POC)
|
||||
- 4. Requisitos Funcionais (FR)
|
||||
- Tasks: Article Content Multi-Engine Extractor
|
||||
@@ -130,7 +130,7 @@
|
||||
- Tasks: Deterministic Article Content Selection
|
||||
- select_article_extractor
|
||||
- process_batch
|
||||
- detect_language
|
||||
- test_models.py
|
||||
- test_select_article_extractor.py
|
||||
- Feature Specification: Deterministic Content Selection
|
||||
- 2. Entity Descriptions & Fields
|
||||
@@ -157,10 +157,10 @@
|
||||
- Specification Quality Checklist: Convert Article JSON to Markdown
|
||||
- CLI Contract: `convert_article_to_markdown.py`
|
||||
- 9. Interface CLI
|
||||
- sample_rss_xml
|
||||
- get_hl_gl_ceid
|
||||
- 13. Estratégia de testes
|
||||
- 6. Contrato de entrada
|
||||
- LLMFallbackAdapter
|
||||
- ECPSnapshot
|
||||
- convert_html_to_markdown
|
||||
- JSON Schema Contract: Deterministic Article Content Selection
|
||||
- 5. Escopo
|
||||
@@ -176,35 +176,37 @@
|
||||
- InherenceClassifier
|
||||
- remove_duplicate_initial_h1
|
||||
- test_normalize_scalar_whitespace_collapsing
|
||||
- .disambiguate
|
||||
- test_funnel_cli_subprocess_end_to_end
|
||||
|
||||
## God Nodes (most connected - your core abstractions)
|
||||
1. `ECPSnapshot` - 49 edges
|
||||
2. `InherenceClassifier` - 36 edges
|
||||
3. `DecisionCategory` - 35 edges
|
||||
4. `LLMFallbackAdapter` - 32 edges
|
||||
5. `ClassificationResult` - 26 edges
|
||||
1. `ECPSnapshot` - 78 edges
|
||||
2. `InherenceClassifier` - 62 edges
|
||||
3. `DecisionCategory` - 62 edges
|
||||
4. `LLMFallbackAdapter` - 48 edges
|
||||
5. `ClassificationResult` - 29 edges
|
||||
6. `select_article_extractor()` - 23 edges
|
||||
7. `ExtractorName` - 21 edges
|
||||
8. `PRD — Conversão de artigo JSON para Markdown` - 16 edges
|
||||
9. `process_batch()` - 15 edges
|
||||
10. `8. Regras funcionais` - 15 edges
|
||||
8. `main()` - 20 edges
|
||||
9. `PRD — Conversão de artigo JSON para Markdown` - 16 edges
|
||||
10. `process_batch()` - 15 edges
|
||||
|
||||
## Surprising Connections (you probably didn't know these)
|
||||
- `main()` --uses--> `ECPSnapshot` [INFERRED]
|
||||
classify.py → src/models.py
|
||||
- `test_extract_google_news_orchestration_mocked()` --uses--> `ExtractionResult` [INFERRED]
|
||||
- `main()` --uses--> `ErrorCode` [INFERRED]
|
||||
classify.py → src/models.py
|
||||
- `test_e2e_extract_google_news_live_pipeline()` --uses--> `ExtractionResult` [INFERRED]
|
||||
tests/test_extract_google_news.py → scripts/extract_google_news.py
|
||||
- `test_llm_adapter_interface()` --calls--> `LLMFallbackAdapter` [EXTRACTED]
|
||||
tests/test_adapters.py → src/adapters/llm.py
|
||||
- `classifier()` --uses--> `InherenceClassifier` [INFERRED]
|
||||
tests/test_benchmark_24.py → src/classifier.py
|
||||
- `test_adversarial_apple_fruit_recipe()` --uses--> `DecisionCategory` [INFERRED]
|
||||
tests/test_adversarial.py → src/models.py
|
||||
|
||||
## Import Cycles
|
||||
- None detected.
|
||||
|
||||
## Communities (168 total, 48 thin omitted)
|
||||
## Communities (170 total, 48 thin omitted)
|
||||
|
||||
### Community 0 - "Task Planning"
|
||||
Cohesion: 0.07
|
||||
@@ -348,15 +350,15 @@ Nodes (6): 1.1 Arguments & Options, 1. Command Line Interface, 2.1 Exit Codes, 2
|
||||
|
||||
### Community 45 - "ClassificationResult"
|
||||
Cohesion: 0.10
|
||||
Nodes (20): ABC, BaseNLPAdapter, Base abstract adapter interface for optional Tier 2 / Tier 3 NLP enhancers., Abstract interface for pluggable NLP classification adapters., Return True if the underlying provider or model is installed and configured., Compute semantic similarity score between text and a set of candidate terms., Optionally refine an ambiguous classification result., LocalEmbeddingsAdapter (+12 more)
|
||||
Nodes (19): ABC, BaseNLPAdapter, Base abstract adapter interface for optional Tier 2 / Tier 3 NLP enhancers., Abstract interface for pluggable NLP classification adapters., Return True if the underlying provider or model is installed and configured., Compute semantic similarity score between text and a set of candidate terms., Optionally refine an ambiguous classification result., LocalEmbeddingsAdapter (+11 more)
|
||||
|
||||
### Community 46 - "test_adversarial.py"
|
||||
Cohesion: 0.11
|
||||
Nodes (19): RelatedEntity, Adversarial and robustness test suite for Multilingual NLP Entity Inherence…, Run CLI via subprocess without --output and verify stdout is pure parseable…, Run CLI via subprocess with empty content and verify error code and exit code., Run CLI via subprocess with missing target_name and verify error payload., Run CLI via subprocess with corrupted JSON and verify error payload., Content about apple fruit/culinary recipe against Apple Inc. tech entity., High-weight related entity mentioned in passing without required domain anchors. (+11 more)
|
||||
Cohesion: 0.10
|
||||
Nodes (21): RelatedEntity, Adversarial and robustness test suite for Multilingual NLP Entity Inherence…, Run CLI via subprocess without --output and verify stdout is pure parseable…, Run CLI via subprocess with empty content and verify error code and exit code., Content about city/state governance of São Paulo against ECP for São Paulo FC., Run CLI via subprocess with missing target_name and verify error payload., Run CLI via subprocess with corrupted JSON and verify error payload., High-weight related entity mentioned in passing without required domain anchors. (+13 more)
|
||||
|
||||
### Community 47 - "classifier.py"
|
||||
Cohesion: 0.16
|
||||
Nodes (15): count_phrase_occurrences(), match_phrase_in_text(), Core deterministic classification engine (Tier 1 core)., Check if a normalized phrase appears in normalized text with word boundary…, Count occurrences of a phrase in text., Classify inherence of content against an ECP snapshot., extract_evidence_snippets(), extract_sentences() (+7 more)
|
||||
### Community 47 - "LLMFallbackAdapter"
|
||||
Cohesion: 0.06
|
||||
Nodes (43): LLMFallbackAdapter, Optional adapter for LLM fallback boundary disambiguation., Any, parametrize, Suíte de Testes Exaustiva para o Classificador de Inerência (classify.py e…, Cenário 4.1: Caso ambíguo elevado para DIRECT_INHERENT pelo LLM., Cenário 4.2: Caso ambíguo elevado para CONTEXTUAL_INHERENT pelo LLM., Cenário 4.3: LLM confirma categoricamente que a menção é periférica /… (+35 more)
|
||||
|
||||
### Community 48 - "test_convert_article_to_markdown.py"
|
||||
Cohesion: 0.08
|
||||
@@ -371,16 +373,16 @@ Cohesion: 0.08
|
||||
Nodes (24): 1. Visão geral (arquitetura), 2.1 DTO de entrada (`googlenews_etl/application/dtos/extract_news_dto.py`), 2.2 Value Object de validação (`googlenews_etl/domain/entities/search_query.py`), 2. Entrada, 3.1 O caso de uso (`googlenews_etl/application/use_cases/extract_news_use_case.py`), 3.2 A porta (`googlenews_etl/domain/ports/news_extractor_port.py`), 3.3.1 Inicialização: sessão HTTP com impersonação de browser, 3.3.2 Mapeamento idioma → parâmetros `hl`/`gl` (`_get_hl_gl`) (+16 more)
|
||||
|
||||
### Community 82 - "extract_google_news.py"
|
||||
Cohesion: 0.15
|
||||
Nodes (18): extract_google_news(), _fetch_rss_content(), get_hl_gl_ceid(), NewsArticle, _normalize_text_for_comparison(), parse_google_news_rss(), Mapeia idioma e locale para os parâmetros hl, gl e ceid do Google News., Remove pontuação e espaços extras para comparação de redundância. (+10 more)
|
||||
Cohesion: 0.20
|
||||
Nodes (14): extract_google_news(), _fetch_rss_content(), NewsArticle, _normalize_text_for_comparison(), parse_google_news_rss(), Remove pontuação e espaços extras para comparação de redundância., Parseia o XML do RSS do Google News e extrai os itens estruturados., Resolve em paralelo as URLs intermediárias do Google News para os links finais… (+6 more)
|
||||
|
||||
### Community 83 - "ExtractionResult"
|
||||
Cohesion: 0.29
|
||||
Nodes (5): ExtractionResult, Any, Resultado consolidado da extração., Valida E2E o fluxo completo de busca, parsing e resolução de URLs reais ao vivo., test_e2e_extract_google_news_live_pipeline()
|
||||
Nodes (5): ExtractionResult, Any, Resultado consolidado da extração., Valida a consolidação do ExtractionResult a partir da busca mockada com URLs…, test_extract_google_news_orchestration_mocked()
|
||||
|
||||
### Community 84 - "test_extract_google_news.py"
|
||||
Cohesion: 0.15
|
||||
Nodes (15): Resolve a URL intermediária do Google News para a URL real do veículo., resolve_article_url(), Testes unitários e de integração para o Extrator de Manchetes do Google News.…, Valida fallback gracioso de URL quando não é link do Google News ou em erro., Valida resolução bem-sucedida de URL do Google News para o portal destino., Valida E2E que o decodificador resolve uma URL real do Google News para o…, Valida o mapeamento padrão de idiomas para pares (hl, gl, ceid)., Valida a sobrescrita geográfica quando o argumento locale é especificado. (+7 more)
|
||||
Cohesion: 0.16
|
||||
Nodes (14): Resolve a URL intermediária do Google News para a URL real do veículo., resolve_article_url(), fixture, Testes unitários e de integração para o Extrator de Manchetes do Google News.…, Valida o parsing do feed RSS, higienização de tags HTML e deduplicação., Valida fallback gracioso de URL quando não é link do Google News ou em erro., Valida resolução bem-sucedida de URL do Google News para o portal destino., Valida E2E que o decodificador resolve uma URL real do Google News para o… (+6 more)
|
||||
|
||||
### Community 85 - "Implementation Tasks: Google News Headlines Extractor"
|
||||
Cohesion: 0.14
|
||||
@@ -400,7 +402,7 @@ Nodes (7): Architecture & Pipeline, Documentation (this feature), Implementation
|
||||
|
||||
### Community 90 - "SearchQuery"
|
||||
Cohesion: 0.20
|
||||
Nodes (6): Value Object com parâmetros de busca validados., SearchQuery, Valida a consolidação do ExtractionResult a partir da busca mockada com URLs…, Valida as regras de negócio e limites de SearchQuery., test_extract_google_news_orchestration_mocked(), test_search_query_validation()
|
||||
Nodes (6): Value Object com parâmetros de busca validados., SearchQuery, Valida E2E o fluxo completo de busca, parsing e resolução de URLs reais ao vivo., Valida as regras de negócio e limites de SearchQuery., test_e2e_extract_google_news_live_pipeline(), test_search_query_validation()
|
||||
|
||||
### Community 91 - "1. Technical Decisions & Tradeoffs"
|
||||
Cohesion: 0.25
|
||||
@@ -435,12 +437,12 @@ Cohesion: 0.22
|
||||
Nodes (9): parametrize, Garante aceitação de URLs absolutas com esquema HTTP e HTTPS válidos., Garante rejeição de esquemas não permitidos, URLs relativas e strings vazias., Garante que a ausência de corpo no extrator selecionado NUNCA faça fallback…, Garante que todos os placeholders documentados no PRD sejam descartados…, test_normalize_scalar_placeholders_discarded(), test_resolve_article_body_strict_isolation_all_extractors(), test_validate_url_invalid_schemes() (+1 more)
|
||||
|
||||
### Community 100 - "main"
|
||||
Cohesion: 0.17
|
||||
Nodes (15): emit_error(), main(), parse_args(), Namespace, ClassificationError, ErrorCode, Enum, str (+7 more)
|
||||
Cohesion: 0.14
|
||||
Nodes (20): main(), Path, Cenário 6.1: Caminho de ECP inexistente -> Exit Code 1, error_code:…, Cenário 6.2: Arquivo ECP com sintaxe JSON corrompida., Cenário 6.3: Valida erro para falta de cada um dos campos obrigatórios do ECP., Cenário 6.4: Caminho de arquivo Markdown inexistente., Cenário 6.5: Arquivo Markdown vazio ou contendo apenas espaços em branco., Cenário 6.6: A flag -o / --output cria diretórios aninhados automaticamente. (+12 more)
|
||||
|
||||
### Community 101 - "ECPSnapshot"
|
||||
Cohesion: 0.18
|
||||
Nodes (12): ECPSnapshot, Any, classifier(), fixture, parametrize, Controlled 24-case benchmark suite for Multilingual NLP Entity Inherence…, test_benchmark_case(), Unit tests for ECP models, schema validation, and structured error handling. (+4 more)
|
||||
### Community 101 - "classifier.py"
|
||||
Cohesion: 0.17
|
||||
Nodes (12): emit_error(), parse_args(), Namespace, Core deterministic classification engine (Tier 1 core)., ErrorCode, MatchedGraphEntity, Enum, str (+4 more)
|
||||
|
||||
### Community 102 - "Feature Specification: Multilingual NLP Entity Inherence Classifier (POC)"
|
||||
Cohesion: 0.14
|
||||
@@ -514,13 +516,13 @@ Nodes (35): CandidateStatus, extract_candidate_data(), ExtractorName, Any, Enum,
|
||||
Cohesion: 0.11
|
||||
Nodes (24): atomic_save_json(), process_batch(), Path, Salva dados em JSON de forma atômica utilizando arquivo temporário e rename., Lê o JSON de entrada, valida a estrutura, processa todos os artigos e grava o…, Path, CT-012: A entrada já contém selected_extractor -> Recalcular e substituir…, CT-013: articles está vazio -> Gerar saída válida com articles vazio. (+16 more)
|
||||
|
||||
### Community 121 - "detect_language"
|
||||
Cohesion: 0.19
|
||||
Nodes (16): detect_language(), extract_words(), normalize_text(), Lightweight multilingual language detection and text normalization., Normalize text by converting to lowercase and stripping combining diacritical…, Tokenize text into lowercase alphanumeric words., Detect the ISO-639-1 language code of text among supported languages (pt, en,…, Unit tests for language detection and text normalization. (+8 more)
|
||||
### Community 121 - "test_models.py"
|
||||
Cohesion: 0.07
|
||||
Nodes (37): count_phrase_occurrences(), match_phrase_in_text(), Check if a normalized phrase appears in normalized text with word boundary…, Count occurrences of a phrase in text., Classify inherence of content against an ECP snapshot., detect_language(), extract_words(), normalize_text() (+29 more)
|
||||
|
||||
### Community 122 - "test_select_article_extractor.py"
|
||||
Cohesion: 0.18
|
||||
Nodes (16): generate_shingles(), normalize_text(), Executa a normalização determinística para comparação: 1. Decodificar entidades…, Gera conjunto de shingles ordenados de tamanho window_size (padrão 5). - Se…, Suíte de Testes Automatizados para o Seletor Determinístico de Extrator. Cobre…, Garante que marcação de imagem Markdown  seja descartada e link…, E2E: Executa CLI com arquivo inexistente e valida código 1 e mensagem no stderr., test_e2e_cli_subprocess_missing_file() (+8 more)
|
||||
Nodes (16): generate_shingles(), normalize_text(), Executa a normalização determinística para comparação: 1. Decodificar entidades…, Gera conjunto de shingles ordenados de tamanho window_size (padrão 5). - Se…, Suíte de Testes Automatizados para o Seletor Determinístico de Extrator. Cobre…, Garante que marcação de imagem Markdown  seja descartada e link…, E2E: Executa scripts/select_article_extractor.py como subprocesso real na linha…, test_e2e_cli_subprocess_real_execution() (+8 more)
|
||||
|
||||
### Community 123 - "Feature Specification: Deterministic Content Selection"
|
||||
Cohesion: 0.17
|
||||
@@ -622,9 +624,9 @@ Nodes (5): 1. Script Signature, 2. Command-Line Arguments, 3. Exit Codes, 4. Sta
|
||||
Cohesion: 0.40
|
||||
Nodes (5): 9.1 Script, 9.2 Argumentos, 9.3 Exemplos, 9.4 Saída do processo, 9. Interface CLI
|
||||
|
||||
### Community 149 - "sample_rss_xml"
|
||||
Cohesion: 0.67
|
||||
Nodes (3): fixture, Fixture que fornece o conteúdo do XML de exemplo para testes offline., sample_rss_xml()
|
||||
### Community 149 - "get_hl_gl_ceid"
|
||||
Cohesion: 0.25
|
||||
Nodes (8): get_hl_gl_ceid(), Mapeia idioma e locale para os parâmetros hl, gl e ceid do Google News., Valida o mapeamento padrão de idiomas para pares (hl, gl, ceid)., Valida a sobrescrita geográfica quando o argumento locale é especificado., Valida fallback dinâmico para idiomas regionais não listados explicitamente., test_get_hl_gl_ceid_default_mappings(), test_get_hl_gl_ceid_dynamic_fallback(), test_get_hl_gl_ceid_with_custom_locale()
|
||||
|
||||
### Community 150 - "13. Estratégia de testes"
|
||||
Cohesion: 0.50
|
||||
@@ -634,9 +636,9 @@ Nodes (4): 13.1 Testes unitários, 13.2 Testes de integração do CLI, 13.3 Caso
|
||||
Cohesion: 0.50
|
||||
Nodes (4): 6.1 Formato, 6.2 Valores aceitos para `selected_extractor`, 6.3 Campos obrigatórios após a resolução, 6. Contrato de entrada
|
||||
|
||||
### Community 152 - "LLMFallbackAdapter"
|
||||
Cohesion: 0.08
|
||||
Nodes (26): LLMFallbackAdapter, Executes LLM fallback for ambiguous boundary cases. Returns a refined…, Parses and validates structured JSON response from LLM., Optional adapter for LLM fallback boundary disambiguation., Returns True if an API key or custom provider function is configured., Constructs an expert-engineered prompt for multilingual entity inherence…, Any, Suíte de Testes para o Adaptador de Fallback para LLM (Tier 3) do Classificador… (+18 more)
|
||||
### Community 152 - "ECPSnapshot"
|
||||
Cohesion: 0.13
|
||||
Nodes (20): ECPSnapshot, parametrize, test_benchmark_case(), Suíte de Testes para o Adaptador de Fallback para LLM (Tier 3) do Classificador…, Valida extração de JSON quando a resposta do LLM vem formatada em bloco…, Valida que respostas corrompidas ou JSONs sem campos obrigatórios retornem None…, Garante que o classificador dispare o Tier 3 LLM para casos ambíguos…, Garante que casos claros (alta confiança e alta densidade de âncoras) NÃO… (+12 more)
|
||||
|
||||
### Community 153 - "convert_html_to_markdown"
|
||||
Cohesion: 0.33
|
||||
@@ -651,13 +653,21 @@ Cohesion: 0.67
|
||||
Nodes (3): 5.1 Incluído, 5.2 Fora do escopo, 5. Escopo
|
||||
|
||||
### Community 165 - "InherenceClassifier"
|
||||
Cohesion: 0.11
|
||||
Nodes (30): InherenceClassifier, Tier 1 Deterministic NLP Entity Inherence Classifier with optional Tier 2 /…, DecisionCategory, Content about city/state governance of São Paulo against ECP for São Paulo FC., test_adversarial_sao_paulo_city_vs_fc(), Unit tests for deterministic classification decision logic., test_contextual_inherent(), test_direct_inherent() (+22 more)
|
||||
Cohesion: 0.07
|
||||
Nodes (45): InherenceClassifier, Tier 1 Deterministic NLP Entity Inherence Classifier with optional Tier 2 /…, DecisionCategory, Content about apple fruit/culinary recipe against Apple Inc. tech entity., test_adversarial_apple_fruit_recipe(), Unit tests for deterministic classification decision logic., test_contextual_inherent(), test_direct_inherent() (+37 more)
|
||||
|
||||
### Community 166 - "remove_duplicate_initial_h1"
|
||||
Cohesion: 0.50
|
||||
Nodes (4): Remove o primeiro título H1 do corpo somente quando ele for igual ao título…, remove_duplicate_initial_h1(), Testa remoção de H1 inicial coincidente com título com variações de espaços e…, test_remove_duplicate_initial_h1_exact_and_variations()
|
||||
|
||||
### Community 168 - ".disambiguate"
|
||||
Cohesion: 0.25
|
||||
Nodes (4): Executes LLM fallback for ambiguous boundary cases. Returns a refined…, Parses and validates structured JSON response from LLM., Returns True if an API key or custom provider function is configured., Constructs an expert-engineered prompt for multilingual entity inherence…
|
||||
|
||||
### Community 169 - "test_funnel_cli_subprocess_end_to_end"
|
||||
Cohesion: 0.67
|
||||
Nodes (3): Path, Valida o contrato CLI completo classify.py com saída em arquivo JSON e flags…, test_funnel_cli_subprocess_end_to_end()
|
||||
|
||||
## Knowledge Gaps
|
||||
- **696 isolated node(s):** `text-nlp-classifier`, `MatchedGraphEntity`, `graphify`, `Usage`, `What graphify is for` (+691 more)
|
||||
These have ≤1 connection - possible missing edges or undocumented components.
|
||||
@@ -666,17 +676,17 @@ Nodes (4): Remove o primeiro título H1 do corpo somente quando ele for igual ao
|
||||
## Suggested Questions
|
||||
_Questions this graph is uniquely positioned to answer:_
|
||||
|
||||
- **Why does `ECPSnapshot` connect `ECPSnapshot` to `main`, `classifier.py`, `InherenceClassifier`, `.disambiguate`, `ClassificationResult`, `test_adversarial.py`, `LLMFallbackAdapter`, `test_models.py`?**
|
||||
_High betweenness centrality (0.009) - this node is a cross-community bridge._
|
||||
- **Why does `PRD — Conversão de artigo JSON para Markdown` connect `PRD — Conversão de artigo JSON para Markdown` to `8. Regras funcionais`, `12. Critérios de aceite`, `11. Requisitos não funcionais`, `9. Interface CLI`, `13. Estratégia de testes`, `6. Contrato de entrada`, `5. Escopo`?**
|
||||
_High betweenness centrality (0.006) - this node is a cross-community bridge._
|
||||
- **Why does `Tasks: Convert Article JSON to Markdown` connect `Tasks: Convert Article JSON to Markdown` to `005-convert-json-markdown/plan.md`?**
|
||||
_High betweenness centrality (0.004) - this node is a cross-community bridge._
|
||||
- **Why does `ECPSnapshot` connect `ECPSnapshot` to `main`, `InherenceClassifier`, `ClassificationResult`, `test_adversarial.py`, `classifier.py`, `LLMFallbackAdapter`?**
|
||||
- **Why does `InherenceClassifier` connect `InherenceClassifier` to `main`, `classifier.py`, `ClassificationResult`, `test_adversarial.py`, `LLMFallbackAdapter`, `ECPSnapshot`, `test_models.py`?**
|
||||
_High betweenness centrality (0.004) - this node is a cross-community bridge._
|
||||
- **Are the 16 inferred relationships involving `ECPSnapshot` (e.g. with `main()` and `BaseNLPAdapter`) actually correct?**
|
||||
_`ECPSnapshot` has 16 INFERRED edges - model-reasoned connections that need verification._
|
||||
- **Are the 7 inferred relationships involving `InherenceClassifier` (e.g. with `LocalEmbeddingsAdapter` and `LLMFallbackAdapter`) actually correct?**
|
||||
_`InherenceClassifier` has 7 INFERRED edges - model-reasoned connections that need verification._
|
||||
- **Are the 24 inferred relationships involving `DecisionCategory` (e.g. with `LLMFallbackAdapter` and `InherenceClassifier`) actually correct?**
|
||||
_`DecisionCategory` has 24 INFERRED edges - model-reasoned connections that need verification._
|
||||
- **Are the 42 inferred relationships involving `ECPSnapshot` (e.g. with `main()` and `BaseNLPAdapter`) actually correct?**
|
||||
_`ECPSnapshot` has 42 INFERRED edges - model-reasoned connections that need verification._
|
||||
- **Are the 8 inferred relationships involving `InherenceClassifier` (e.g. with `LocalEmbeddingsAdapter` and `LLMFallbackAdapter`) actually correct?**
|
||||
_`InherenceClassifier` has 8 INFERRED edges - model-reasoned connections that need verification._
|
||||
- **Are the 50 inferred relationships involving `DecisionCategory` (e.g. with `LLMFallbackAdapter` and `InherenceClassifier`) actually correct?**
|
||||
_`DecisionCategory` has 50 INFERRED edges - model-reasoned connections that need verification._
|
||||
- **Are the 4 inferred relationships involving `LLMFallbackAdapter` (e.g. with `ClassificationResult` and `DecisionCategory`) actually correct?**
|
||||
_`LLMFallbackAdapter` has 4 INFERRED edges - model-reasoned connections that need verification._
|
||||
+1
File diff suppressed because one or more lines are too long
+1
File diff suppressed because one or more lines are too long
+1
File diff suppressed because one or more lines are too long
Vendored
+1
-1
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
+3815
-804
File diff suppressed because it is too large
Load Diff
@@ -336,9 +336,9 @@
|
||||
"semantic_hash": ""
|
||||
},
|
||||
"src/classifier.py": {
|
||||
"mtime": 1787320047.0362887,
|
||||
"seen": 1787320239.025866,
|
||||
"ast_hash": "d6cc674d407a99ab52f6d5156f9b3d8e",
|
||||
"mtime": 1787321309.8981817,
|
||||
"seen": 1787321481.1505442,
|
||||
"ast_hash": "a4e5dafed12aa4eaf096988b2c6a8ae0",
|
||||
"semantic_hash": ""
|
||||
},
|
||||
"src/language.py": {
|
||||
@@ -654,9 +654,9 @@
|
||||
"semantic_hash": ""
|
||||
},
|
||||
"README.md": {
|
||||
"mtime": 1787321187.0812356,
|
||||
"seen": 1787321205.5201268,
|
||||
"ast_hash": "aedfaf7a245288227952a2e28e7e7b13",
|
||||
"mtime": 1787321467.0542295,
|
||||
"seen": 1787321481.1561577,
|
||||
"ast_hash": "0cde8e800125cbba61a1d7de9d9d2c9e",
|
||||
"semantic_hash": ""
|
||||
},
|
||||
"scripts/extract_article_contents.py": {
|
||||
@@ -922,5 +922,11 @@
|
||||
"seen": 1787321205.5156026,
|
||||
"ast_hash": "3a2d47f2ffcf8371ffdf90bb797b5346",
|
||||
"semantic_hash": ""
|
||||
},
|
||||
"tests/test_classify_exhaustive_suite.py": {
|
||||
"mtime": 1787321371.2658408,
|
||||
"seen": 1787321481.15137,
|
||||
"ast_hash": "08e3c680d8669b2849a19d73b27b869a",
|
||||
"semantic_hash": ""
|
||||
}
|
||||
}
|
||||
+1
-1
@@ -48,7 +48,7 @@ class InherenceClassifier:
|
||||
llm_adapter: Optional[Any] = None,
|
||||
) -> None:
|
||||
self.enable_embeddings = enable_embeddings
|
||||
self.enable_llm = enable_llm or (llm_adapter is not None)
|
||||
self.enable_llm = enable_llm
|
||||
self._embeddings_adapter = None
|
||||
self._llm_adapter = llm_adapter
|
||||
|
||||
|
||||
@@ -0,0 +1,806 @@
|
||||
"""
|
||||
Suíte de Testes Exaustiva para o Classificador de Inerência (classify.py e src/).
|
||||
|
||||
Cobre 100% dos caminhos felizes, infelizes, limiares, de ambiguidade,
|
||||
fallback de LLM (OpenAI e Gemini), resiliência de API, erros de contrato CLI
|
||||
e suporte aos 6 idiomas conforme a metodologia da skill-suite-tests.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
from classify import main
|
||||
from src.adapters.llm import LLMFallbackAdapter
|
||||
from src.classifier import InherenceClassifier
|
||||
from src.models import (
|
||||
ClassificationResult,
|
||||
DecisionCategory,
|
||||
ECPSnapshot,
|
||||
RelatedEntity,
|
||||
)
|
||||
|
||||
CLASSIFY_CLI = Path(__file__).parent.parent / "classify.py"
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# Fixtures Universais
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def ecp_tech_corp() -> ECPSnapshot:
|
||||
return ECPSnapshot(
|
||||
target_entity_id="ent_tech_corp",
|
||||
target_name="TechCorp Global",
|
||||
aliases=["TechCorp", "TechCorp Global", "TCG"],
|
||||
domain="Tecnologia e Cloud",
|
||||
anchors=[
|
||||
"cloud",
|
||||
"computação em nuvem",
|
||||
"software",
|
||||
"inteligência artificial",
|
||||
"datacenter",
|
||||
],
|
||||
negative_anchors=["TechCorp Calçados", "TechCorp Imóveis", "homônimo"],
|
||||
related_entities=[
|
||||
RelatedEntity(
|
||||
entity_id="ent_cloud_subsidiary",
|
||||
name="CloudPlatform Solutions",
|
||||
relation_type="SUBSIDIARY_OF",
|
||||
weight=0.90,
|
||||
aliases=["CloudPlatform"],
|
||||
scope="cloud_services",
|
||||
),
|
||||
RelatedEntity(
|
||||
entity_id="ent_ceo_tech",
|
||||
name="Alan Turing Silva",
|
||||
relation_type="CEO_OF",
|
||||
weight=0.80,
|
||||
aliases=["Alan Turing"],
|
||||
scope="executive",
|
||||
),
|
||||
],
|
||||
)
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 1. Casos Felizes (Happy Paths) - NLP Determinístico (Tier 1)
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_happy_path_direct_inherent_with_canonical_and_anchors(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 1.1: Nome canônico + múltiplas âncoras temáticas -> DIRECT_INHERENT com alta confiança."""
|
||||
classifier = InherenceClassifier()
|
||||
content = (
|
||||
"# TechCorp Global anuncia novo datacenter de computação em nuvem\n\n"
|
||||
"A TechCorp Global investiu 500 milhões para expandir sua infraestrutura de software "
|
||||
"e inteligência artificial na América Latina."
|
||||
)
|
||||
res = classifier.classify(ecp_tech_corp, content)
|
||||
assert res.decision == DecisionCategory.DIRECT_INHERENT
|
||||
assert res.is_inherent is True
|
||||
assert res.confidence >= 0.90
|
||||
assert "TechCorp Global" in res.matched_anchors or "TechCorp" in res.matched_anchors
|
||||
assert len(res.evidence) >= 1
|
||||
|
||||
|
||||
def test_happy_path_direct_inherent_via_alias_and_acronym(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 1.2: Apenas o alias / sigla 'TCG' é mencionado, com âncoras do domínio."""
|
||||
classifier = InherenceClassifier()
|
||||
content = (
|
||||
"# Inovação em Cloud\n\n"
|
||||
"A TCG lançou hoje uma nova plataforma de software baseada em computação em nuvem."
|
||||
)
|
||||
res = classifier.classify(ecp_tech_corp, content)
|
||||
assert res.decision == DecisionCategory.DIRECT_INHERENT
|
||||
assert res.is_inherent is True
|
||||
assert res.confidence >= 0.85
|
||||
|
||||
|
||||
def test_happy_path_direct_inherent_by_repetition_without_heavy_anchors(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 1.3: O nome 'TechCorp' aparece 3 vezes no texto, satisfazendo a regra de menção múltipla."""
|
||||
classifier = InherenceClassifier()
|
||||
content = (
|
||||
"# Relatório Corporativo Trimestral\n\n"
|
||||
"A TechCorp divulgou seus resultados. A TechCorp superou as estimativas de analistas. "
|
||||
"O conselho da TechCorp aprovou dividendos extraordinários."
|
||||
)
|
||||
res = classifier.classify(ecp_tech_corp, content)
|
||||
assert res.decision == DecisionCategory.DIRECT_INHERENT
|
||||
assert res.is_inherent is True
|
||||
assert res.confidence >= 0.85
|
||||
|
||||
|
||||
def test_happy_path_contextual_inherent_via_subsidiary_graph_entity(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 1.4: Menção da subsidiária 'CloudPlatform Solutions' com âncoras de cloud."""
|
||||
classifier = InherenceClassifier()
|
||||
content = (
|
||||
"# Expansão de Infraestrutura de Nuvem\n\n"
|
||||
"A CloudPlatform Solutions ativou novos servidores em seu datacenter de computação em nuvem."
|
||||
)
|
||||
res = classifier.classify(ecp_tech_corp, content)
|
||||
assert res.decision == DecisionCategory.CONTEXTUAL_INHERENT
|
||||
assert res.is_inherent is True
|
||||
assert res.confidence >= 0.75
|
||||
assert len(res.graph_matches) >= 1
|
||||
assert res.graph_matches[0]["name"] == "CloudPlatform Solutions"
|
||||
|
||||
|
||||
def test_happy_path_contextual_inherent_via_executive_graph_entity(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 1.5: Menção ao CEO no grafo + âncoras de tecnologia."""
|
||||
classifier = InherenceClassifier()
|
||||
content = (
|
||||
"# Discurso na Conferência de Tecnologia\n\n"
|
||||
"O executivo Alan Turing Silva discursou sobre o futuro da inteligência artificial e software."
|
||||
)
|
||||
res = classifier.classify(ecp_tech_corp, content)
|
||||
assert res.decision == DecisionCategory.CONTEXTUAL_INHERENT
|
||||
assert res.is_inherent is True
|
||||
assert any(g["name"] == "Alan Turing Silva" for g in res.graph_matches)
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 2. Casos Infelizes e Rejeições (Sad Paths) - NLP Determinístico (Tier 1)
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_sad_path_not_related_completely_off_topic(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 2.1: Conteúdo totalmente desvinculado (culinária/jardinagem)."""
|
||||
classifier = InherenceClassifier()
|
||||
content = (
|
||||
"# Receita de Pão Caseiro Fácil\n\n"
|
||||
"Misture a farinha, o fermento biológico seco e a água morna. "
|
||||
"Deixe a massa descansar por 40 minutos em local aquecido."
|
||||
)
|
||||
res = classifier.classify(ecp_tech_corp, content)
|
||||
assert res.decision == DecisionCategory.NOT_RELATED
|
||||
assert res.is_inherent is False
|
||||
assert res.confidence >= 0.90
|
||||
assert len(res.matched_anchors) == 0
|
||||
|
||||
|
||||
def test_sad_path_not_related_generic_domain_without_target_or_graph(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 2.2: Artigo cita muitas âncoras ('cloud', 'software'), mas NÃO cita a TechCorp nem o grafo."""
|
||||
classifier = InherenceClassifier()
|
||||
content = (
|
||||
"# O Mercado Global de Computação em Nuvem\n\n"
|
||||
"O setor de computação em nuvem, datacenter e inteligência artificial cresceu 25% este ano."
|
||||
)
|
||||
res = classifier.classify(ecp_tech_corp, content)
|
||||
assert res.decision == DecisionCategory.NOT_RELATED
|
||||
assert res.is_inherent is False
|
||||
assert (
|
||||
"General domain topics mentioned, but target entity or related entities are absent."
|
||||
in res.rationale
|
||||
)
|
||||
|
||||
|
||||
def test_sad_path_not_related_negative_anchor_dominance(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 2.3: Homônimo 'TechCorp Calçados' dispara âncora negativa dominante."""
|
||||
classifier = InherenceClassifier()
|
||||
content = (
|
||||
"# Feira de Moda e Varejo\n\n"
|
||||
"A TechCorp Calçados apresentou sua nova linha de sandálias de couro para o verão."
|
||||
)
|
||||
res = classifier.classify(ecp_tech_corp, content)
|
||||
assert res.decision == DecisionCategory.NOT_RELATED
|
||||
assert res.is_inherent is False
|
||||
assert "TechCorp Calçados" in res.negative_matches
|
||||
|
||||
|
||||
def test_sad_path_not_related_negative_anchor_ties_with_positive_anchor(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 2.4: 1 âncora negativa e 1 positiva -> prioridade de segurança rejeita para NOT_RELATED."""
|
||||
classifier = InherenceClassifier()
|
||||
content = "A TechCorp Calçados adotou um novo software interno de gestão."
|
||||
res = classifier.classify(ecp_tech_corp, content)
|
||||
assert res.decision == DecisionCategory.NOT_RELATED
|
||||
assert res.is_inherent is False
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 3. Casos Limiares e Ambiguidades (Borderline / Tangential)
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_borderline_tangential_single_passing_mention(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 3.1: Menção única isolada sem âncoras temáticas -> TANGENTIAL com baixa confiança."""
|
||||
classifier = InherenceClassifier()
|
||||
content = "Estávamos caminhando pela avenida e vimos a placa da TechCorp ao longe na esquina."
|
||||
res = classifier.classify(ecp_tech_corp, content)
|
||||
assert res.decision == DecisionCategory.TANGENTIAL
|
||||
assert res.is_inherent is False
|
||||
assert res.confidence == 0.40
|
||||
assert any("Low contextual density" in w for w in res.warnings)
|
||||
|
||||
|
||||
def test_borderline_tangential_graph_entity_in_isolation(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 3.2: Entidade do grafo mencionada sem contexto de domínio -> TANGENTIAL."""
|
||||
classifier = InherenceClassifier()
|
||||
content = "Alan Turing Silva participou de uma corrida beneficente no parque no domingo."
|
||||
res = classifier.classify(ecp_tech_corp, content)
|
||||
assert res.decision == DecisionCategory.TANGENTIAL
|
||||
assert res.is_inherent is False
|
||||
assert res.confidence == 0.45
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 4. Suíte Abrangente de Fallback para LLM (Tier 3)
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_llm_happy_path_upgrade_tangential_to_direct_inherent(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 4.1: Caso ambíguo elevado para DIRECT_INHERENT pelo LLM."""
|
||||
mock_resp = json.dumps(
|
||||
{
|
||||
"analysis_summary": "Artigo detalha o projeto estratégico secreto da TechCorp.",
|
||||
"decision": "DIRECT_INHERENT",
|
||||
"confidence": 0.95,
|
||||
"rationale": "Embora a redação use linguagem coloquial, o artigo foca inteiramente na estratégia da TechCorp.",
|
||||
}
|
||||
)
|
||||
adapter = LLMFallbackAdapter(provider_fn=lambda p: mock_resp)
|
||||
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
|
||||
|
||||
content = "A diretoria da TechCorp finalizou as negociações confidenciais da rodada."
|
||||
res = classifier.classify(ecp_tech_corp, content)
|
||||
|
||||
assert res.decision == DecisionCategory.DIRECT_INHERENT
|
||||
assert res.is_inherent is True
|
||||
assert res.confidence == 0.95
|
||||
assert "[Tier 3 LLM]" in res.rationale
|
||||
assert "[Tier 3 LLM Override applied]" in res.warnings
|
||||
|
||||
|
||||
def test_llm_happy_path_upgrade_tangential_to_contextual_inherent(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 4.2: Caso ambíguo elevado para CONTEXTUAL_INHERENT pelo LLM."""
|
||||
mock_resp = json.dumps(
|
||||
{
|
||||
"analysis_summary": "Matéria sobre fusão de fornecedores onde a TechCorp é impactada diretamente.",
|
||||
"decision": "CONTEXTUAL_INHERENT",
|
||||
"confidence": 0.88,
|
||||
"rationale": "A TechCorp é parte material do ecossistema afetado pela fusão anunciada.",
|
||||
}
|
||||
)
|
||||
adapter = LLMFallbackAdapter(provider_fn=lambda p: mock_resp)
|
||||
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
|
||||
|
||||
content = "O consórcio fornecedor foi reestruturado e envolverá contratos com a TechCorp."
|
||||
res = classifier.classify(ecp_tech_corp, content)
|
||||
|
||||
assert res.decision == DecisionCategory.CONTEXTUAL_INHERENT
|
||||
assert res.is_inherent is True
|
||||
assert res.confidence == 0.88
|
||||
|
||||
|
||||
def test_llm_happy_path_confirmation_of_tangential(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 4.3: LLM confirma categoricamente que a menção é periférica / irrelevante."""
|
||||
mock_resp = json.dumps(
|
||||
{
|
||||
"analysis_summary": "Crônica sobre trânsito urbano com citação lateral a um outdoor da TechCorp.",
|
||||
"decision": "TANGENTIAL",
|
||||
"confidence": 0.97,
|
||||
"rationale": "A empresa é apenas uma referência visual casual sem relação com a narrativa de trânsito.",
|
||||
}
|
||||
)
|
||||
adapter = LLMFallbackAdapter(provider_fn=lambda p: mock_resp)
|
||||
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
|
||||
|
||||
content = "O tráfego estava parado bem em frente ao painel da TechCorp na autoestrada."
|
||||
res = classifier.classify(ecp_tech_corp, content)
|
||||
|
||||
assert res.decision == DecisionCategory.TANGENTIAL
|
||||
assert res.is_inherent is False
|
||||
assert res.confidence == 0.97
|
||||
|
||||
|
||||
def test_llm_happy_path_rejection_to_not_related(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 4.4: LLM identifica homônimo não mapeado nas regras determinísticas e rebaixa para NOT_RELATED."""
|
||||
mock_resp = json.dumps(
|
||||
{
|
||||
"analysis_summary": "Artigo sobre uma banda de rock indie com nome idêntico.",
|
||||
"decision": "NOT_RELATED",
|
||||
"confidence": 0.99,
|
||||
"rationale": "O texto refere-se a um grupo musical e não à empresa de tecnologia.",
|
||||
}
|
||||
)
|
||||
adapter = LLMFallbackAdapter(provider_fn=lambda p: mock_resp)
|
||||
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
|
||||
|
||||
content = "A banda TechCorp tocou seus novos acordes no festival de música independente."
|
||||
res = classifier.classify(ecp_tech_corp, content)
|
||||
|
||||
assert res.decision == DecisionCategory.NOT_RELATED
|
||||
assert res.is_inherent is False
|
||||
assert res.confidence == 0.99
|
||||
|
||||
|
||||
def test_llm_sad_path_llm_disabled_by_default_never_invokes_adapter(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 4.5: Quando enable_llm=False (padrão), o LLM NUNCA é chamado mesmo em caso limiar."""
|
||||
called = {"status": False}
|
||||
|
||||
def tracking_fn(p: str) -> str:
|
||||
called["status"] = True
|
||||
return "{}"
|
||||
|
||||
adapter = LLMFallbackAdapter(provider_fn=tracking_fn)
|
||||
classifier = InherenceClassifier(enable_llm=False, llm_adapter=adapter)
|
||||
|
||||
content = "Menção isolada da TechCorp sem contexto algum."
|
||||
res = classifier.classify(ecp_tech_corp, content)
|
||||
|
||||
assert res.decision == DecisionCategory.TANGENTIAL
|
||||
assert called["status"] is False
|
||||
|
||||
|
||||
def test_llm_sad_path_flag_enabled_without_api_key_or_provider(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 4.6: enable_llm=True mas sem chaves no ambiente -> degrada sem quebrar, retém Tier 1."""
|
||||
with patch.dict("os.environ", {}, clear=True):
|
||||
adapter = LLMFallbackAdapter(api_key="", provider_fn=None)
|
||||
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
|
||||
|
||||
content = "Menção isolada da TechCorp em relatório breve."
|
||||
res = classifier.classify(ecp_tech_corp, content)
|
||||
|
||||
assert res.decision == DecisionCategory.TANGENTIAL
|
||||
assert res.is_inherent is False
|
||||
|
||||
|
||||
def test_llm_sad_path_network_timeout_graceful_degradation(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 4.7: API do LLM sofre TimeoutError -> retém Tier 1 e registra aviso em warnings."""
|
||||
|
||||
def timeout_fn(p: str) -> str:
|
||||
raise TimeoutError("Conexão com gateway do LLM excedeu tempo limite de 30s.")
|
||||
|
||||
adapter = LLMFallbackAdapter(provider_fn=timeout_fn)
|
||||
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
|
||||
|
||||
content = "A TechCorp esteve presente no evento de premiação."
|
||||
res = classifier.classify(ecp_tech_corp, content)
|
||||
|
||||
assert res.decision == DecisionCategory.TANGENTIAL
|
||||
assert any("LLM fallback failed" in w for w in res.warnings)
|
||||
|
||||
|
||||
def test_llm_sad_path_http_500_server_error_graceful_degradation(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 4.8: API do LLM retorna erro 500 / ConnectionError -> retém Tier 1 com aviso."""
|
||||
|
||||
def error_500_fn(p: str) -> str:
|
||||
raise ConnectionError("HTTP 500: Internal Server Error do provedor de IA.")
|
||||
|
||||
adapter = LLMFallbackAdapter(provider_fn=error_500_fn)
|
||||
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
|
||||
|
||||
content = "Menção da TechCorp em comunicado à imprensa."
|
||||
res = classifier.classify(ecp_tech_corp, content)
|
||||
|
||||
assert res.decision == DecisionCategory.TANGENTIAL
|
||||
assert any("LLM fallback failed" in w for w in res.warnings)
|
||||
|
||||
|
||||
def test_llm_sad_path_malformed_json_and_non_json_strings(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 4.9: LLM retorna texto livre ou JSON quebrado -> parser ignora com segurança."""
|
||||
|
||||
def make_bad_provider(resp_text: str):
|
||||
def _prov(prompt: str) -> str:
|
||||
return resp_text
|
||||
|
||||
return _prov
|
||||
|
||||
for bad_resp in [
|
||||
"Não tenho certeza sobre este documento.",
|
||||
"{json_quebrado_sem_aspas: true",
|
||||
"```json\n{invalido: 123}\n```",
|
||||
]:
|
||||
adapter = LLMFallbackAdapter(provider_fn=make_bad_provider(bad_resp))
|
||||
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
|
||||
|
||||
content = "Menção isolada da TechCorp."
|
||||
res = classifier.classify(ecp_tech_corp, content)
|
||||
assert res.decision == DecisionCategory.TANGENTIAL
|
||||
|
||||
|
||||
def test_llm_sad_path_missing_decision_key_in_json(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 4.10: LLM retorna JSON válido mas sem o campo obrigatório 'decision'."""
|
||||
adapter = LLMFallbackAdapter(
|
||||
provider_fn=lambda p: json.dumps({"confidence": 0.90, "rationale": "Faltou a decisao"})
|
||||
)
|
||||
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
|
||||
|
||||
content = "Menção isolada da TechCorp."
|
||||
res = classifier.classify(ecp_tech_corp, content)
|
||||
assert res.decision == DecisionCategory.TANGENTIAL
|
||||
|
||||
|
||||
def test_llm_sad_path_unknown_hallucinated_decision_enum(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 4.11: LLM alucina uma categoria inexistente (ex: 'SUPER_INHERENT')."""
|
||||
adapter = LLMFallbackAdapter(
|
||||
provider_fn=lambda p: json.dumps({"decision": "SUPER_INHERENT", "confidence": 0.99})
|
||||
)
|
||||
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
|
||||
|
||||
content = "Menção isolada da TechCorp."
|
||||
res = classifier.classify(ecp_tech_corp, content)
|
||||
assert res.decision == DecisionCategory.TANGENTIAL
|
||||
|
||||
|
||||
def test_llm_resilience_confidence_clipping(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 4.12: LLM retorna confidence fora do intervalo [0.0, 1.0] -> clippa com segurança."""
|
||||
|
||||
def make_clipping_provider(c_val: float):
|
||||
def _prov(prompt: str) -> str:
|
||||
return json.dumps(
|
||||
{
|
||||
"decision": "DIRECT_INHERENT",
|
||||
"confidence": c_val,
|
||||
"rationale": "Teste de clipping.",
|
||||
}
|
||||
)
|
||||
|
||||
return _prov
|
||||
|
||||
for raw_conf, expected_conf in [(1.5, 1.0), (-0.5, 0.0), (0.85432, 0.8543)]:
|
||||
adapter = LLMFallbackAdapter(provider_fn=make_clipping_provider(raw_conf))
|
||||
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
|
||||
res = classifier.classify(ecp_tech_corp, "Menção da TechCorp.")
|
||||
assert res.confidence == expected_conf
|
||||
|
||||
|
||||
def test_llm_optimization_clear_case_bypasses_llm(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 4.13: Caso claro de alta densidade NÃO chama LLM mesmo com enable_llm=True."""
|
||||
called = {"status": False}
|
||||
|
||||
def tracking_fn(p: str) -> str:
|
||||
called["status"] = True
|
||||
return json.dumps({"decision": "DIRECT_INHERENT"})
|
||||
|
||||
adapter = LLMFallbackAdapter(provider_fn=tracking_fn)
|
||||
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
|
||||
|
||||
content = (
|
||||
"# TechCorp Global anuncia nova inteligência artificial para computação em nuvem\n\n"
|
||||
"A TechCorp Global ativou hoje novos clusters de datacenter com software avançado."
|
||||
)
|
||||
res = classifier.classify(ecp_tech_corp, content)
|
||||
|
||||
assert res.decision == DecisionCategory.DIRECT_INHERENT
|
||||
assert called["status"] is False # LLM NÃO foi acionado
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 5. Provedores Reais de LLM (OpenAI Mock e Gemini REST Mock)
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_llm_provider_openai_client_execution(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 5.1: Simula execução bem-sucedida via cliente OpenAI SDK."""
|
||||
mock_chat_completion = MagicMock()
|
||||
mock_choice = MagicMock()
|
||||
mock_choice.message.content = json.dumps(
|
||||
{
|
||||
"decision": "DIRECT_INHERENT",
|
||||
"confidence": 0.96,
|
||||
"rationale": "OpenAI validou o contexto corporativo com precisão.",
|
||||
}
|
||||
)
|
||||
mock_chat_completion.choices = [mock_choice]
|
||||
|
||||
mock_openai_instance = MagicMock()
|
||||
mock_openai_instance.chat.completions.create.return_value = mock_chat_completion
|
||||
|
||||
with patch("openai.OpenAI", return_value=mock_openai_instance):
|
||||
adapter = LLMFallbackAdapter(api_key="sk-mock-openai-key")
|
||||
initial_res = ClassificationResult(
|
||||
decision=DecisionCategory.TANGENTIAL,
|
||||
is_inherent=False,
|
||||
confidence=0.40,
|
||||
detected_language="pt",
|
||||
matched_anchors=[],
|
||||
negative_matches=[],
|
||||
graph_matches=[],
|
||||
evidence=[],
|
||||
rationale="Passing.",
|
||||
warnings=[],
|
||||
)
|
||||
res = adapter.disambiguate(ecp_tech_corp, "Artigo sobre TechCorp.", initial_res)
|
||||
assert res is not None
|
||||
assert res.decision == DecisionCategory.DIRECT_INHERENT
|
||||
assert res.confidence == 0.96
|
||||
|
||||
|
||||
def test_llm_provider_gemini_rest_execution(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 5.2: Simula execução bem-sucedida via API REST do Google Gemini."""
|
||||
gemini_payload = {
|
||||
"candidates": [
|
||||
{
|
||||
"content": {
|
||||
"parts": [
|
||||
{
|
||||
"text": json.dumps(
|
||||
{
|
||||
"decision": "DIRECT_INHERENT",
|
||||
"confidence": 0.98,
|
||||
"rationale": "Gemini 2.5 Flash confirmou aderência direta ao tópico.",
|
||||
}
|
||||
)
|
||||
}
|
||||
]
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
mock_response = MagicMock()
|
||||
mock_response.read.return_value = json.dumps(gemini_payload).encode("utf-8")
|
||||
mock_response.__enter__.return_value = mock_response
|
||||
|
||||
with patch("urllib.request.urlopen", return_value=mock_response):
|
||||
with patch.dict("os.environ", {"GEMINI_API_KEY": "mock-gemini-key"}):
|
||||
adapter = LLMFallbackAdapter(api_key="")
|
||||
initial_res = ClassificationResult(
|
||||
decision=DecisionCategory.TANGENTIAL,
|
||||
is_inherent=False,
|
||||
confidence=0.40,
|
||||
detected_language="pt",
|
||||
matched_anchors=[],
|
||||
negative_matches=[],
|
||||
graph_matches=[],
|
||||
evidence=[],
|
||||
rationale="Passing.",
|
||||
warnings=[],
|
||||
)
|
||||
res = adapter.disambiguate(ecp_tech_corp, "Artigo sobre TechCorp.", initial_res)
|
||||
assert res is not None
|
||||
assert res.decision == DecisionCategory.DIRECT_INHERENT
|
||||
assert res.confidence == 0.98
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 6. Suíte de Contrato e Erros da CLI classify.py
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_cli_error_ecp_file_does_not_exist(tmp_path: Path, capsys):
|
||||
"""Cenário 6.1: Caminho de ECP inexistente -> Exit Code 1, error_code: invalid_ecp_json."""
|
||||
content_file = tmp_path / "valid.md"
|
||||
content_file.write_text("# Conteúdo válido", encoding="utf-8")
|
||||
|
||||
exit_code = main(["--ecp", str(tmp_path / "nao_existe.json"), "--content", str(content_file)])
|
||||
assert exit_code == 1
|
||||
|
||||
captured = capsys.readouterr()
|
||||
err_json = json.loads(captured.err)
|
||||
assert err_json["error_code"] == "invalid_ecp_json"
|
||||
|
||||
|
||||
def test_cli_error_ecp_corrupted_json_syntax(tmp_path: Path, capsys):
|
||||
"""Cenário 6.2: Arquivo ECP com sintaxe JSON corrompida."""
|
||||
bad_ecp = tmp_path / "corrupt.json"
|
||||
bad_ecp.write_text("{ target_name: 'sem_aspas' ", encoding="utf-8")
|
||||
content_file = tmp_path / "valid.md"
|
||||
content_file.write_text("# Conteúdo válido", encoding="utf-8")
|
||||
|
||||
exit_code = main(["--ecp", str(bad_ecp), "--content", str(content_file)])
|
||||
assert exit_code == 1
|
||||
|
||||
captured = capsys.readouterr()
|
||||
err_json = json.loads(captured.err)
|
||||
assert err_json["error_code"] == "invalid_ecp_json"
|
||||
|
||||
|
||||
def test_cli_error_ecp_missing_each_required_field(tmp_path: Path, capsys):
|
||||
"""Cenário 6.3: Valida erro para falta de cada um dos campos obrigatórios do ECP."""
|
||||
required_fields = ["target_entity_id", "target_name", "aliases", "domain", "anchors"]
|
||||
|
||||
base_ecp = {
|
||||
"target_entity_id": "ent_1",
|
||||
"target_name": "Nome",
|
||||
"aliases": ["Alias"],
|
||||
"domain": "Domínio",
|
||||
"anchors": ["Âncora"],
|
||||
}
|
||||
content_file = tmp_path / "valid.md"
|
||||
content_file.write_text("# Conteúdo válido", encoding="utf-8")
|
||||
|
||||
for field in required_fields:
|
||||
bad_data = base_ecp.copy()
|
||||
del bad_data[field]
|
||||
bad_file = tmp_path / f"missing_{field}.json"
|
||||
bad_file.write_text(json.dumps(bad_data), encoding="utf-8")
|
||||
|
||||
exit_code = main(["--ecp", str(bad_file), "--content", str(content_file)])
|
||||
assert exit_code == 1
|
||||
|
||||
captured = capsys.readouterr()
|
||||
err_json = json.loads(captured.err)
|
||||
assert err_json["error_code"] == "missing_required_field"
|
||||
assert field in err_json["message"]
|
||||
|
||||
|
||||
def test_cli_error_content_file_does_not_exist(tmp_path: Path, capsys):
|
||||
"""Cenário 6.4: Caminho de arquivo Markdown inexistente."""
|
||||
ecp_file = tmp_path / "ecp.json"
|
||||
ecp_file.write_text(
|
||||
json.dumps(
|
||||
{
|
||||
"target_entity_id": "ent_1",
|
||||
"target_name": "TechCorp",
|
||||
"aliases": ["TechCorp"],
|
||||
"domain": "Tech",
|
||||
"anchors": ["cloud"],
|
||||
}
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
exit_code = main(["--ecp", str(ecp_file), "--content", str(tmp_path / "doc_fantasma.md")])
|
||||
assert exit_code == 1
|
||||
|
||||
captured = capsys.readouterr()
|
||||
err_json = json.loads(captured.err)
|
||||
assert err_json["error_code"] == "invalid_markdown"
|
||||
|
||||
|
||||
def test_cli_error_empty_and_whitespace_content(tmp_path: Path, capsys):
|
||||
"""Cenário 6.5: Arquivo Markdown vazio ou contendo apenas espaços em branco."""
|
||||
ecp_file = tmp_path / "ecp.json"
|
||||
ecp_file.write_text(
|
||||
json.dumps(
|
||||
{
|
||||
"target_entity_id": "ent_1",
|
||||
"target_name": "TechCorp",
|
||||
"aliases": ["TechCorp"],
|
||||
"domain": "Tech",
|
||||
"anchors": ["cloud"],
|
||||
}
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
for empty_text in ["", " \n\n\t \n "]:
|
||||
empty_file = tmp_path / "empty.md"
|
||||
empty_file.write_text(empty_text, encoding="utf-8")
|
||||
|
||||
exit_code = main(["--ecp", str(ecp_file), "--content", str(empty_file)])
|
||||
assert exit_code == 1
|
||||
|
||||
captured = capsys.readouterr()
|
||||
err_json = json.loads(captured.err)
|
||||
assert err_json["error_code"] == "empty_content"
|
||||
|
||||
|
||||
def test_cli_output_file_creates_nested_directories(tmp_path: Path):
|
||||
"""Cenário 6.6: A flag -o / --output cria diretórios aninhados automaticamente."""
|
||||
ecp_file = tmp_path / "ecp.json"
|
||||
ecp_file.write_text(
|
||||
json.dumps(
|
||||
{
|
||||
"target_entity_id": "ent_1",
|
||||
"target_name": "TechCorp",
|
||||
"aliases": ["TechCorp"],
|
||||
"domain": "Tech",
|
||||
"anchors": ["cloud"],
|
||||
}
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
content_file = tmp_path / "content.md"
|
||||
content_file.write_text("# TechCorp\n\nTechCorp cloud computing.", encoding="utf-8")
|
||||
|
||||
nested_out = tmp_path / "deep" / "nested" / "folder" / "resultado.json"
|
||||
|
||||
exit_code = main(
|
||||
["--ecp", str(ecp_file), "--content", str(content_file), "-o", str(nested_out)]
|
||||
)
|
||||
assert exit_code == 0
|
||||
assert nested_out.exists()
|
||||
|
||||
payload = json.loads(nested_out.read_text(encoding="utf-8"))
|
||||
assert payload["decision"] == "DIRECT_INHERENT"
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 7. Matriz Multilíngue Completa (6 Idiomas)
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"lang,target,aliases,domain,anchors,content,expected_decision,expected_lang",
|
||||
[
|
||||
# Português
|
||||
(
|
||||
"pt",
|
||||
"Petrobras",
|
||||
["Petrobras"],
|
||||
"Energia",
|
||||
["pré-sal", "petróleo", "refinaria"],
|
||||
"# Petrobras bate recorde de produção no pré-sal com novas plataformas.",
|
||||
DecisionCategory.DIRECT_INHERENT,
|
||||
"pt",
|
||||
),
|
||||
# Inglês
|
||||
(
|
||||
"en",
|
||||
"Apple Inc.",
|
||||
["Apple", "Apple Inc."],
|
||||
"Technology",
|
||||
["iPhone", "MacBook", "iOS", "silicon"],
|
||||
"# Apple unveils new MacBook Pro with M4 silicon and advanced iOS features.",
|
||||
DecisionCategory.DIRECT_INHERENT,
|
||||
"en",
|
||||
),
|
||||
# Espanhol
|
||||
(
|
||||
"es",
|
||||
"River Plate",
|
||||
["River Plate", "River"],
|
||||
"Fútbol",
|
||||
["Monumental", "Libertadores", "Sudamericana"],
|
||||
"# River Plate se prepara para disputar el torneo continental en el Estadio Monumental.",
|
||||
DecisionCategory.DIRECT_INHERENT,
|
||||
"es",
|
||||
),
|
||||
# Alemão (Compostos e Diacríticos)
|
||||
(
|
||||
"de",
|
||||
"Volkswagen AG",
|
||||
["Volkswagen", "VW"],
|
||||
"Automobilindustrie",
|
||||
["Elektroauto", "Batteriefabrik", "Produktion"],
|
||||
"# Volkswagen investiert Milliarden in eine neue Batteriefabrik für Elektroautos in Deutschland.",
|
||||
DecisionCategory.DIRECT_INHERENT,
|
||||
"de",
|
||||
),
|
||||
# Italiano
|
||||
(
|
||||
"it",
|
||||
"Scuderia Ferrari",
|
||||
["Ferrari", "Scuderia Ferrari"],
|
||||
"Automobilismo",
|
||||
["Monza", "Gran Premio", "motore", "pole position"],
|
||||
"# La Ferrari conquista una straordinaria pole position nel Gran Premio di Monza.",
|
||||
DecisionCategory.DIRECT_INHERENT,
|
||||
"it",
|
||||
),
|
||||
# Francês (Elisão e Apóstrofos)
|
||||
(
|
||||
"fr",
|
||||
"TotalEnergies",
|
||||
["TotalEnergies", "Total"],
|
||||
"Énergie",
|
||||
["énergie solaire", "pétrole", "renouvelable", "électricité"],
|
||||
"# L'entreprise TotalEnergies accélère ses investissements dans l'énergie solaire et l'électricité en France.",
|
||||
DecisionCategory.DIRECT_INHERENT,
|
||||
"fr",
|
||||
),
|
||||
],
|
||||
)
|
||||
def test_multilingual_matrix_6_languages(
|
||||
lang: str,
|
||||
target: str,
|
||||
aliases: list[str],
|
||||
domain: str,
|
||||
anchors: list[str],
|
||||
content: str,
|
||||
expected_decision: DecisionCategory,
|
||||
expected_lang: str,
|
||||
):
|
||||
"""Garante a precisão e robustez do classificador nos 6 idiomas suportados pela POC."""
|
||||
ecp = ECPSnapshot(
|
||||
target_entity_id=f"ent_{lang}",
|
||||
target_name=target,
|
||||
aliases=aliases,
|
||||
domain=domain,
|
||||
anchors=anchors,
|
||||
)
|
||||
classifier = InherenceClassifier()
|
||||
res = classifier.classify(ecp, content)
|
||||
|
||||
assert res.decision == expected_decision
|
||||
assert res.is_inherent is True
|
||||
assert res.detected_language == expected_lang
|
||||
assert res.confidence >= 0.85
|
||||
Reference in New Issue
Block a user