test(qa): add exhaustive 38-scenario test suite covering happy, sad, borderline, LLM fallback, CLI contracts, and multilingual matrix

This commit is contained in:
2026-08-21 11:11:35 -03:00
parent a874b98dac
commit 2cdd3547b2
17 changed files with 6111 additions and 1287 deletions
+7 -3
View File
@@ -501,7 +501,8 @@ TextNLPClassifierApp/
│ ├── test_select_article_extractor.py │ ├── test_select_article_extractor.py
│ ├── test_convert_article_to_markdown.py # Testes da conversão para Markdown │ ├── test_convert_article_to_markdown.py # Testes da conversão para Markdown
│ ├── test_llm_fallback.py # Testes do Tier 3 LLM Fallback │ ├── test_llm_fallback.py # Testes do Tier 3 LLM Fallback
│ └── test_e2e_text_analysis_pipeline.py # Suíte E2E do Funil de Análise e Fallback │ ├── test_e2e_text_analysis_pipeline.py # Suíte E2E do Funil de Análise e Fallback
│ └── test_classify_exhaustive_suite.py # Suíte Exaustiva de Casos Felizes/Infelizes (QA Sênior)
├── requirements.txt # Dependências do projeto ├── requirements.txt # Dependências do projeto
├── pyproject.toml # Configurações de ferramentas (pytest, ruff, mypy) ├── pyproject.toml # Configurações de ferramentas (pytest, ruff, mypy)
└── README.md # Documentação principal └── README.md # Documentação principal
@@ -511,12 +512,15 @@ TextNLPClassifierApp/
## 🧪 Testes e Qualidade de Código ## 🧪 Testes e Qualidade de Código
O repositório possui **209 testes automatizados** com 100% de aprovação cobrindo testes unitários, de regressão, de integração, Golden Fixtures exatas, testes de sensibilidade de mutação, testes de fallback para LLM (Tier 3), validações de degradação graciosa e testes End-to-End (E2E) via CLI subprocess: O repositório possui **247 testes automatizados** com 100% de aprovação cobrindo testes unitários, de regressão, de integração, Golden Fixtures exatas, testes de sensibilidade de mutação, testes de fallback para LLM (Tier 3), validações de degradação graciosa, matriz multilíngue e testes End-to-End (E2E) via CLI subprocess:
```bash ```bash
# Executar toda a suíte de testes do projeto (209 testes) # Executar toda a suíte de testes do projeto (247 testes)
pytest -v pytest -v
# Executar a Suíte Exaustiva de Classificação e Fallback (38 testes)
pytest tests/test_classify_exhaustive_suite.py -v
# Executar a Suíte E2E do Funil de Análise de Texto e Fallback para LLM # Executar a Suíte E2E do Funil de Análise de Texto e Fallback para LLM
pytest tests/test_e2e_text_analysis_pipeline.py -v pytest tests/test_e2e_text_analysis_pipeline.py -v
+8 -6
View File
@@ -46,7 +46,7 @@
"44": "2. Standard Streams & Exit Codes", "44": "2. Standard Streams & Exit Codes",
"45": "ClassificationResult", "45": "ClassificationResult",
"46": "test_adversarial.py", "46": "test_adversarial.py",
"47": "classifier.py", "47": "LLMFallbackAdapter",
"48": "test_convert_article_to_markdown.py", "48": "test_convert_article_to_markdown.py",
"49": "content_northvolt_de.md", "49": "content_northvolt_de.md",
"50": "content_presal_pt.md", "50": "content_presal_pt.md",
@@ -100,7 +100,7 @@
"98": "Extraction Pipeline Checklist: Article Content Multi-Engine Extractor", "98": "Extraction Pipeline Checklist: Article Content Multi-Engine Extractor",
"99": "parametrize", "99": "parametrize",
"100": "main", "100": "main",
"101": "ECPSnapshot", "101": "classifier.py",
"102": "Feature Specification: Multilingual NLP Entity Inherence Classifier (POC)", "102": "Feature Specification: Multilingual NLP Entity Inherence Classifier (POC)",
"103": "4. Requisitos Funcionais (FR)", "103": "4. Requisitos Funcionais (FR)",
"104": "Tasks: Article Content Multi-Engine Extractor", "104": "Tasks: Article Content Multi-Engine Extractor",
@@ -120,7 +120,7 @@
"118": "Tasks: Deterministic Article Content Selection", "118": "Tasks: Deterministic Article Content Selection",
"119": "select_article_extractor", "119": "select_article_extractor",
"120": "process_batch", "120": "process_batch",
"121": "detect_language", "121": "test_models.py",
"122": "test_select_article_extractor.py", "122": "test_select_article_extractor.py",
"123": "Feature Specification: Deterministic Content Selection", "123": "Feature Specification: Deterministic Content Selection",
"124": "2. Entity Descriptions & Fields", "124": "2. Entity Descriptions & Fields",
@@ -148,10 +148,10 @@
"146": "Specification Quality Checklist: Convert Article JSON to Markdown", "146": "Specification Quality Checklist: Convert Article JSON to Markdown",
"147": "CLI Contract: `convert_article_to_markdown.py`", "147": "CLI Contract: `convert_article_to_markdown.py`",
"148": "9. Interface CLI", "148": "9. Interface CLI",
"149": "sample_rss_xml", "149": "get_hl_gl_ceid",
"150": "13. Estratégia de testes", "150": "13. Estratégia de testes",
"151": "6. Contrato de entrada", "151": "6. Contrato de entrada",
"152": "LLMFallbackAdapter", "152": "ECPSnapshot",
"153": "convert_html_to_markdown", "153": "convert_html_to_markdown",
"154": "JSON Schema Contract: Deterministic Article Content Selection", "154": "JSON Schema Contract: Deterministic Article Content Selection",
"155": "5. Escopo", "155": "5. Escopo",
@@ -166,5 +166,7 @@
"164": "test_normalize_scalar_non_string_types", "164": "test_normalize_scalar_non_string_types",
"165": "InherenceClassifier", "165": "InherenceClassifier",
"166": "remove_duplicate_initial_h1", "166": "remove_duplicate_initial_h1",
"167": "test_normalize_scalar_whitespace_collapsing" "167": "test_normalize_scalar_whitespace_collapsing",
"168": ".disambiguate",
"169": "test_funnel_cli_subprocess_end_to_end"
} }
+1 -1
View File
@@ -1 +1 @@
{"0": "36bdb6f09c457f7c", "1": "8c5bf6244cf710c6", "2": "efbcc9c62a3ee78b", "3": "8599153989b07faa", "4": "b5952a1f7fee9f20", "5": "5b8462a3f82d188c", "6": "80f79e9e2011a3e3", "7": "4654167fd211d027", "8": "50acfa00fe353440", "9": "c6d2f770737823f1", "10": "44f2ca451aea24be", "11": "feaac5ab67a8c17a", "12": "b71bd92e5edbf2e0", "13": "219d65ba6d2689e4", "14": "8e30bb8112fd02d1", "15": "03906ab80b99db85", "16": "5d51c60ba1bc2be0", "17": "a1da914f522dcd21", "18": "fbad840891b90569", "19": "0686ff2d6fe29fb3", "20": "060baa9e1924b465", "21": "a5c8f2c3080b8243", "22": "0d76852f1d29eeb1", "23": "6ff68619f2d72924", "24": "3da11675eee7ec46", "25": "a6696589e9556f97", "26": "6c752999e8a4d4b6", "27": "2d4e13ea2111d750", "28": "4b60cb0ee1ac186a", "29": "f56fbca9bb8235ec", "30": "c7beed940704509f", "31": "38be2d254fb31ae8", "32": "ee5596fcf7e7c0b3", "33": "e4d4e0a440bc599f", "34": "c897e49c001acdae", "35": "3aad272a2cf5d495", "36": "0a197439d306b956", "37": "f43acf5c8b1329af", "38": "6775efafc9b33338", "39": "8176a164778526f9", "40": "66b69189c0acc3ff", "41": "0322ff824966a4d8", "42": "784c9e3d336a7f53", "43": "4b8bb6c3f7b64856", "44": "18c0ff3e6225bcb2", "45": "d5eb5f4efd73cafb", "46": "57116576996271f5", "47": "0fad42a4989aa7e3", "48": "0237e1e02ee47a27", "49": "0d0f9f015921feef", "50": "8d0c81e5ca23e9a6", "51": "f79963571b9c15ee", "52": "5935824c825606cb", "53": "9685f9cbe158e50b", "54": "3d5ab759f350bc79", "55": "d549f24931a990e9", "56": "3cc031dcb648797c", "57": "a0ab88e6c629251d", "58": "76bd6412e2a22ecd", "59": "54827845564490c9", "60": "0a9736c416c0c6b9", "61": "77358620ac528153", "62": "3b0c585df09df48a", "63": "7e78cd3b28828c20", "64": "1c0c958231735f61", "65": "60b0f81225f62f69", "66": "920754c65cc94b88", "67": "df911472140a9b94", "68": "8e17bc11bcea91b9", "69": "7e905b75e4f28b95", "70": "a28424eca5d36c55", "71": "2cdb53d5b6051ab6", "72": "e42fbd3dc744e730", "73": "7fe2cac980de160c", "74": "2b1343a6a9db1487", "75": "54a1bb232f1d4ceb", "76": "442ba11d31ec0e0a", "77": "852a25b8b95bf8d1", "78": "1810ab370b9cd608", "79": "0fc5dca02a3f02f6", "80": "6ff8a97e63c9a2f3", "81": "a38f84ae3d895236", "82": "08e48bd11f9714df", "83": "5095122914e83cf5", "84": "1aef305bd7d7d63f", "85": "f8bfd0cfe9e8b478", "86": "410d15a346bd5894", "87": "6b41d288cfd834ab", "88": "5aa6db96312a8811", "89": "80225792bb62ba04", "90": "fd291228c3311f40", "91": "d4579c5b7aa2742a", "92": "7b9ba7c3bff11361", "93": "71cd9c1fa4a857f0", "94": "34cd980be3c32d21", "95": "970093453f3b7d90", "96": "9e96780a2b7c4bd6", "97": "b7c10b0e09caac0b", "98": "089ea6a55861c693", "99": "cb6165a7dc822d29", "100": "7bdb2c6abfbde762", "101": "3e1a3e8ca5030d57", "102": "6aa00d5a83295f11", "103": "f58668f5b10ccdeb", "104": "4ec787414cc6f50b", "105": "1cf3077fd45d874a", "106": "edcd5d9bb3c4b00f", "107": "37f2f47110fe3eaa", "108": "b7ad5abb1da8cf8d", "109": "cb48a9c4f54efa38", "110": "f6dd36fd7f3edbe5", "111": "2925b620f0b1fd17", "112": "d8b3099917c3b711", "113": "3bb61caa0302c804", "114": "0d4f1d08dd056bb9", "115": "4ac2dcddeec2ff11", "116": "07da9aae9668f573", "117": "196f63e0c4536d30", "118": "ade84262e3cfac12", "119": "ebe4e5e0c42c613f", "120": "16f0543249fafdd8", "121": "5396e68ca6c185ad", "122": "d1eeebf358bcab60", "123": "96618c9a362af46c", "124": "83f104cbb62fd03e", "125": "6db738fb27190349", "126": "6a087a22cbcef972", "127": "85fd71a0cad8d3a5", "128": "22dd4feed96c4229", "129": "c4d2f60f532e6f16", "130": "f6b0aa8a1568926b", "131": "142d0db70bad18fe", "132": "67ea4284cbc02c54", "133": "d899cfc86c7a4a27", "134": "ba9464410a9b4168", "135": "0d496a12149eca27", "136": "521f5c7b9d566b4d", "137": "9e37828bdd2ba8c5", "138": "ec03c97194c56f91", "139": "f4e6d5dfa30034c5", "140": "d9b47fa423cf0748", "141": "1e0330b8757f333e", "142": "f35d75e1194c008d", "143": "4d2ae7190b514a34", "144": "4a98716cabf43f86", "145": "56b2431193739c38", "146": "edc785fd71bb0675", "147": "4c7347f8f86e1fbd", "148": "8ca77cc4fd6fd437", "149": "8968e9e7d55afcbe", "150": "f4f4ce1a1180ddb1", "151": "e426746f6e9ee15f", "152": "4c30720833331d86", "153": "a3593e6f45bafb20", "154": "56747bad6345d66b", "155": "a8e7498fa7e257df", "156": "56e7b2355898077f", "157": "f2fc88f7d8214711", "158": "c966f6f8570c8c29", "159": "5f6094aa385f3bfe", "160": "2834e7d59672e756", "161": "cc6e436d94fd0033", "162": "64f33a2fc8969cd2", "163": "26ac1c0a00eabce1", "164": "cdcea44a6805ae55", "165": "1e217fe7a21a2501", "166": "85c97dab928b9b1b", "167": "7ab5695391e32126"} {"0": "36bdb6f09c457f7c", "1": "8c5bf6244cf710c6", "2": "efbcc9c62a3ee78b", "3": "8599153989b07faa", "4": "b5952a1f7fee9f20", "5": "5b8462a3f82d188c", "6": "80f79e9e2011a3e3", "7": "4654167fd211d027", "8": "50acfa00fe353440", "9": "c6d2f770737823f1", "10": "44f2ca451aea24be", "11": "feaac5ab67a8c17a", "12": "b71bd92e5edbf2e0", "13": "219d65ba6d2689e4", "14": "8e30bb8112fd02d1", "15": "03906ab80b99db85", "16": "5d51c60ba1bc2be0", "17": "a1da914f522dcd21", "18": "fbad840891b90569", "19": "0686ff2d6fe29fb3", "20": "060baa9e1924b465", "21": "a5c8f2c3080b8243", "22": "0d76852f1d29eeb1", "23": "6ff68619f2d72924", "24": "3da11675eee7ec46", "25": "a6696589e9556f97", "26": "6c752999e8a4d4b6", "27": "2d4e13ea2111d750", "28": "4b60cb0ee1ac186a", "29": "f56fbca9bb8235ec", "30": "c7beed940704509f", "31": "38be2d254fb31ae8", "32": "ee5596fcf7e7c0b3", "33": "e4d4e0a440bc599f", "34": "c897e49c001acdae", "35": "3aad272a2cf5d495", "36": "0a197439d306b956", "37": "f43acf5c8b1329af", "38": "6775efafc9b33338", "39": "8176a164778526f9", "40": "66b69189c0acc3ff", "41": "0322ff824966a4d8", "42": "784c9e3d336a7f53", "43": "4b8bb6c3f7b64856", "44": "18c0ff3e6225bcb2", "45": "943c894b7e96a921", "46": "b30963ae66d7e3c9", "47": "85bec2d6e2b742cf", "48": "0237e1e02ee47a27", "49": "0d0f9f015921feef", "50": "8d0c81e5ca23e9a6", "51": "f79963571b9c15ee", "52": "5935824c825606cb", "53": "9685f9cbe158e50b", "54": "3d5ab759f350bc79", "55": "d549f24931a990e9", "56": "3cc031dcb648797c", "57": "a0ab88e6c629251d", "58": "76bd6412e2a22ecd", "59": "54827845564490c9", "60": "0a9736c416c0c6b9", "61": "77358620ac528153", "62": "3b0c585df09df48a", "63": "7e78cd3b28828c20", "64": "1c0c958231735f61", "65": "60b0f81225f62f69", "66": "920754c65cc94b88", "67": "df911472140a9b94", "68": "8e17bc11bcea91b9", "69": "7e905b75e4f28b95", "70": "a28424eca5d36c55", "71": "2cdb53d5b6051ab6", "72": "e42fbd3dc744e730", "73": "7fe2cac980de160c", "74": "2b1343a6a9db1487", "75": "54a1bb232f1d4ceb", "76": "442ba11d31ec0e0a", "77": "852a25b8b95bf8d1", "78": "1810ab370b9cd608", "79": "0fc5dca02a3f02f6", "80": "6ff8a97e63c9a2f3", "81": "a38f84ae3d895236", "82": "dc6ddc157a3b9efb", "83": "a05140495d7a0353", "84": "24ca89fec34df075", "85": "f8bfd0cfe9e8b478", "86": "410d15a346bd5894", "87": "6b41d288cfd834ab", "88": "5aa6db96312a8811", "89": "80225792bb62ba04", "90": "e18a0a239fe528ba", "91": "d4579c5b7aa2742a", "92": "7b9ba7c3bff11361", "93": "71cd9c1fa4a857f0", "94": "34cd980be3c32d21", "95": "970093453f3b7d90", "96": "9e96780a2b7c4bd6", "97": "b7c10b0e09caac0b", "98": "089ea6a55861c693", "99": "cb6165a7dc822d29", "100": "8cb2e59dcf557313", "101": "75f2ab420202693e", "102": "6aa00d5a83295f11", "103": "f58668f5b10ccdeb", "104": "4ec787414cc6f50b", "105": "1cf3077fd45d874a", "106": "edcd5d9bb3c4b00f", "107": "37f2f47110fe3eaa", "108": "b7ad5abb1da8cf8d", "109": "cb48a9c4f54efa38", "110": "f6dd36fd7f3edbe5", "111": "2925b620f0b1fd17", "112": "d8b3099917c3b711", "113": "3bb61caa0302c804", "114": "0d4f1d08dd056bb9", "115": "4ac2dcddeec2ff11", "116": "07da9aae9668f573", "117": "196f63e0c4536d30", "118": "ade84262e3cfac12", "119": "ebe4e5e0c42c613f", "120": "27256931b19a2867", "121": "3d7cd9541766116e", "122": "aa8a1de55696b666", "123": "96618c9a362af46c", "124": "83f104cbb62fd03e", "125": "6db738fb27190349", "126": "6a087a22cbcef972", "127": "85fd71a0cad8d3a5", "128": "22dd4feed96c4229", "129": "c4d2f60f532e6f16", "130": "f6b0aa8a1568926b", "131": "142d0db70bad18fe", "132": "67ea4284cbc02c54", "133": "d899cfc86c7a4a27", "134": "ba9464410a9b4168", "135": "0d496a12149eca27", "136": "521f5c7b9d566b4d", "137": "9e37828bdd2ba8c5", "138": "ec03c97194c56f91", "139": "f4e6d5dfa30034c5", "140": "d9b47fa423cf0748", "141": "1e0330b8757f333e", "142": "f35d75e1194c008d", "143": "4d2ae7190b514a34", "144": "4a98716cabf43f86", "145": "56b2431193739c38", "146": "edc785fd71bb0675", "147": "4c7347f8f86e1fbd", "148": "8ca77cc4fd6fd437", "149": "09850697b717469a", "150": "f4f4ce1a1180ddb1", "151": "e426746f6e9ee15f", "152": "73cf7c102ed797ed", "153": "a3593e6f45bafb20", "154": "56747bad6345d66b", "155": "a8e7498fa7e257df", "156": "56e7b2355898077f", "157": "f2fc88f7d8214711", "158": "c966f6f8570c8c29", "159": "5f6094aa385f3bfe", "160": "2834e7d59672e756", "161": "cc6e436d94fd0033", "162": "64f33a2fc8969cd2", "163": "26ac1c0a00eabce1", "164": "cdcea44a6805ae55", "165": "f3f95b2d8f95c75f", "166": "85c97dab928b9b1b", "167": "7ab5695391e32126", "168": "eafca6a072d4f435", "169": "a68dbc0869da4c5e"}
@@ -99,7 +99,7 @@
"97": "🧠 TextNLPClassifierApp", "97": "🧠 TextNLPClassifierApp",
"98": "Extraction Pipeline Checklist: Article Content Multi-Engine Extractor", "98": "Extraction Pipeline Checklist: Article Content Multi-Engine Extractor",
"99": "parametrize", "99": "parametrize",
"100": "models.py", "100": "main",
"101": "ECPSnapshot", "101": "ECPSnapshot",
"102": "Feature Specification: Multilingual NLP Entity Inherence Classifier (POC)", "102": "Feature Specification: Multilingual NLP Entity Inherence Classifier (POC)",
"103": "4. Requisitos Funcionais (FR)", "103": "4. Requisitos Funcionais (FR)",
@@ -151,7 +151,7 @@
"149": "sample_rss_xml", "149": "sample_rss_xml",
"150": "13. Estratégia de testes", "150": "13. Estratégia de testes",
"151": "6. Contrato de entrada", "151": "6. Contrato de entrada",
"152": "InherenceClassifier", "152": "LLMFallbackAdapter",
"153": "convert_html_to_markdown", "153": "convert_html_to_markdown",
"154": "JSON Schema Contract: Deterministic Article Content Selection", "154": "JSON Schema Contract: Deterministic Article Content Selection",
"155": "5. Escopo", "155": "5. Escopo",
@@ -164,7 +164,7 @@
"162": "test_normalize_date_iso_8601_variants", "162": "test_normalize_date_iso_8601_variants",
"163": "test_metadata_priority_original_url_all_fallbacks", "163": "test_metadata_priority_original_url_all_fallbacks",
"164": "test_normalize_scalar_non_string_types", "164": "test_normalize_scalar_non_string_types",
"165": ".disambiguate", "165": "InherenceClassifier",
"166": "remove_duplicate_initial_h1", "166": "remove_duplicate_initial_h1",
"167": "test_normalize_scalar_whitespace_collapsing" "167": "test_normalize_scalar_whitespace_collapsing"
} }
+37 -37
View File
@@ -1,16 +1,16 @@
# Graph Report - TextNLPClassifierApp (2026-08-21) # Graph Report - TextNLPClassifierApp (2026-08-21)
## Corpus Check ## Corpus Check
- 201 files · ~110,615 words - 202 files · ~112,197 words
- Verdict: corpus is large enough that graph structure adds value. - Verdict: corpus is large enough that graph structure adds value.
## Summary ## Summary
- 1554 nodes · 1970 edges · 168 communities (120 shown, 48 thin omitted) - 1579 nodes · 2030 edges · 168 communities (120 shown, 48 thin omitted)
- Extraction: 97% EXTRACTED · 3% INFERRED · 0% AMBIGUOUS · INFERRED: 59 edges (avg confidence: 0.95) - Extraction: 96% EXTRACTED · 4% INFERRED · 0% AMBIGUOUS · INFERRED: 73 edges (avg confidence: 0.95)
- Token cost: 0 input · 0 output - Token cost: 0 input · 0 output
## Graph Freshness ## Graph Freshness
- Built from commit: `31152d50` - Built from commit: `bae14405`
- Run `git rev-parse HEAD` and compare to check if the graph is stale. - Run `git rev-parse HEAD` and compare to check if the graph is stale.
- Run `graphify update .` after code changes (no API cost). - Run `graphify update .` after code changes (no API cost).
@@ -110,7 +110,7 @@
- 🧠 TextNLPClassifierApp - 🧠 TextNLPClassifierApp
- Extraction Pipeline Checklist: Article Content Multi-Engine Extractor - Extraction Pipeline Checklist: Article Content Multi-Engine Extractor
- parametrize - parametrize
- models.py - main
- ECPSnapshot - ECPSnapshot
- Feature Specification: Multilingual NLP Entity Inherence Classifier (POC) - Feature Specification: Multilingual NLP Entity Inherence Classifier (POC)
- 4. Requisitos Funcionais (FR) - 4. Requisitos Funcionais (FR)
@@ -160,7 +160,7 @@
- sample_rss_xml - sample_rss_xml
- 13. Estratégia de testes - 13. Estratégia de testes
- 6. Contrato de entrada - 6. Contrato de entrada
- InherenceClassifier - LLMFallbackAdapter
- convert_html_to_markdown - convert_html_to_markdown
- JSON Schema Contract: Deterministic Article Content Selection - JSON Schema Contract: Deterministic Article Content Selection
- 5. Escopo - 5. Escopo
@@ -173,16 +173,16 @@
- test_normalize_date_iso_8601_variants - test_normalize_date_iso_8601_variants
- test_metadata_priority_original_url_all_fallbacks - test_metadata_priority_original_url_all_fallbacks
- test_normalize_scalar_non_string_types - test_normalize_scalar_non_string_types
- .disambiguate - InherenceClassifier
- remove_duplicate_initial_h1 - remove_duplicate_initial_h1
- test_normalize_scalar_whitespace_collapsing - test_normalize_scalar_whitespace_collapsing
## God Nodes (most connected - your core abstractions) ## God Nodes (most connected - your core abstractions)
1. `ECPSnapshot` - 40 edges 1. `ECPSnapshot` - 49 edges
2. `InherenceClassifier` - 29 edges 2. `InherenceClassifier` - 36 edges
3. `DecisionCategory` - 28 edges 3. `DecisionCategory` - 35 edges
4. `LLMFallbackAdapter` - 26 edges 4. `LLMFallbackAdapter` - 32 edges
5. `ClassificationResult` - 24 edges 5. `ClassificationResult` - 26 edges
6. `select_article_extractor()` - 23 edges 6. `select_article_extractor()` - 23 edges
7. `ExtractorName` - 21 edges 7. `ExtractorName` - 21 edges
8. `PRD — Conversão de artigo JSON para Markdown` - 16 edges 8. `PRD — Conversão de artigo JSON para Markdown` - 16 edges
@@ -347,12 +347,12 @@ Cohesion: 0.29
Nodes (6): 1.1 Arguments & Options, 1. Command Line Interface, 2.1 Exit Codes, 2.2 Standard Output (`stdout`) / Standard Error (`stderr`), 2. Standard Streams & Exit Codes, CLI Contract & Interface Specification (POC) Nodes (6): 1.1 Arguments & Options, 1. Command Line Interface, 2.1 Exit Codes, 2.2 Standard Output (`stdout`) / Standard Error (`stderr`), 2. Standard Streams & Exit Codes, CLI Contract & Interface Specification (POC)
### Community 45 - "ClassificationResult" ### Community 45 - "ClassificationResult"
Cohesion: 0.11 Cohesion: 0.10
Nodes (17): ABC, BaseNLPAdapter, Base abstract adapter interface for optional Tier 2 / Tier 3 NLP enhancers., Abstract interface for pluggable NLP classification adapters., Return True if the underlying provider or model is installed and configured., Compute semantic similarity score between text and a set of candidate terms., Optionally refine an ambiguous classification result., LocalEmbeddingsAdapter (+9 more) Nodes (20): ABC, BaseNLPAdapter, Base abstract adapter interface for optional Tier 2 / Tier 3 NLP enhancers., Abstract interface for pluggable NLP classification adapters., Return True if the underlying provider or model is installed and configured., Compute semantic similarity score between text and a set of candidate terms., Optionally refine an ambiguous classification result., LocalEmbeddingsAdapter (+12 more)
### Community 46 - "test_adversarial.py" ### Community 46 - "test_adversarial.py"
Cohesion: 0.10 Cohesion: 0.11
Nodes (19): Any, RelatedEntity, Adversarial and robustness test suite for Multilingual NLP Entity Inherence…, Run CLI via subprocess without --output and verify stdout is pure parseable…, Run CLI via subprocess with empty content and verify error code and exit code., Content about city/state governance of São Paulo against ECP for São Paulo FC., Run CLI via subprocess with missing target_name and verify error payload., Run CLI via subprocess with corrupted JSON and verify error payload. (+11 more) Nodes (19): RelatedEntity, Adversarial and robustness test suite for Multilingual NLP Entity Inherence…, Run CLI via subprocess without --output and verify stdout is pure parseable…, Run CLI via subprocess with empty content and verify error code and exit code., Run CLI via subprocess with missing target_name and verify error payload., Run CLI via subprocess with corrupted JSON and verify error payload., Content about apple fruit/culinary recipe against Apple Inc. tech entity., High-weight related entity mentioned in passing without required domain anchors. (+11 more)
### Community 47 - "classifier.py" ### Community 47 - "classifier.py"
Cohesion: 0.16 Cohesion: 0.16
@@ -434,13 +434,13 @@ Nodes (34): 1. Requirement Completeness, 2. Requirement Clarity & Non-Ambiguity,
Cohesion: 0.22 Cohesion: 0.22
Nodes (9): parametrize, Garante aceitação de URLs absolutas com esquema HTTP e HTTPS válidos., Garante rejeição de esquemas não permitidos, URLs relativas e strings vazias., Garante que a ausência de corpo no extrator selecionado NUNCA faça fallback…, Garante que todos os placeholders documentados no PRD sejam descartados…, test_normalize_scalar_placeholders_discarded(), test_resolve_article_body_strict_isolation_all_extractors(), test_validate_url_invalid_schemes() (+1 more) Nodes (9): parametrize, Garante aceitação de URLs absolutas com esquema HTTP e HTTPS válidos., Garante rejeição de esquemas não permitidos, URLs relativas e strings vazias., Garante que a ausência de corpo no extrator selecionado NUNCA faça fallback…, Garante que todos os placeholders documentados no PRD sejam descartados…, test_normalize_scalar_placeholders_discarded(), test_resolve_article_body_strict_isolation_all_extractors(), test_validate_url_invalid_schemes() (+1 more)
### Community 100 - "models.py" ### Community 100 - "main"
Cohesion: 0.15 Cohesion: 0.17
Nodes (17): emit_error(), main(), parse_args(), Namespace, ClassificationError, ErrorCode, MatchedGraphEntity, Enum (+9 more) Nodes (15): emit_error(), main(), parse_args(), Namespace, ClassificationError, ErrorCode, Enum, str (+7 more)
### Community 101 - "ECPSnapshot" ### Community 101 - "ECPSnapshot"
Cohesion: 0.24 Cohesion: 0.18
Nodes (10): ECPSnapshot, classifier(), fixture, parametrize, Controlled 24-case benchmark suite for Multilingual NLP Entity Inherence…, test_benchmark_case(), Unit tests for ECP models, schema validation, and structured error handling., test_ecp_snapshot_defaults() (+2 more) Nodes (12): ECPSnapshot, Any, classifier(), fixture, parametrize, Controlled 24-case benchmark suite for Multilingual NLP Entity Inherence…, test_benchmark_case(), Unit tests for ECP models, schema validation, and structured error handling. (+4 more)
### Community 102 - "Feature Specification: Multilingual NLP Entity Inherence Classifier (POC)" ### Community 102 - "Feature Specification: Multilingual NLP Entity Inherence Classifier (POC)"
Cohesion: 0.14 Cohesion: 0.14
@@ -520,7 +520,7 @@ Nodes (16): detect_language(), extract_words(), normalize_text(), Lightweight mu
### Community 122 - "test_select_article_extractor.py" ### Community 122 - "test_select_article_extractor.py"
Cohesion: 0.18 Cohesion: 0.18
Nodes (16): generate_shingles(), normalize_text(), Executa a normalização determinística para comparação: 1. Decodificar entidades…, Gera conjunto de shingles ordenados de tamanho window_size (padrão 5). - Se…, Suíte de Testes Automatizados para o Seletor Determinístico de Extrator. Cobre…, Garante que marcação de imagem Markdown ![alt](url) seja descartada e link…, E2E: Executa scripts/select_article_extractor.py como subprocesso real na linha…, test_e2e_cli_subprocess_real_execution() (+8 more) Nodes (16): generate_shingles(), normalize_text(), Executa a normalização determinística para comparação: 1. Decodificar entidades…, Gera conjunto de shingles ordenados de tamanho window_size (padrão 5). - Se…, Suíte de Testes Automatizados para o Seletor Determinístico de Extrator. Cobre…, Garante que marcação de imagem Markdown ![alt](url) seja descartada e link…, E2E: Executa CLI com arquivo inexistente e valida código 1 e mensagem no stderr., test_e2e_cli_subprocess_missing_file() (+8 more)
### Community 123 - "Feature Specification: Deterministic Content Selection" ### Community 123 - "Feature Specification: Deterministic Content Selection"
Cohesion: 0.17 Cohesion: 0.17
@@ -634,9 +634,9 @@ Nodes (4): 13.1 Testes unitários, 13.2 Testes de integração do CLI, 13.3 Caso
Cohesion: 0.50 Cohesion: 0.50
Nodes (4): 6.1 Formato, 6.2 Valores aceitos para `selected_extractor`, 6.3 Campos obrigatórios após a resolução, 6. Contrato de entrada Nodes (4): 6.1 Formato, 6.2 Valores aceitos para `selected_extractor`, 6.3 Campos obrigatórios após a resolução, 6. Contrato de entrada
### Community 152 - "InherenceClassifier" ### Community 152 - "LLMFallbackAdapter"
Cohesion: 0.11 Cohesion: 0.08
Nodes (31): LLMFallbackAdapter, Optional adapter for LLM fallback boundary disambiguation., InherenceClassifier, Any, Tier 1 Deterministic NLP Entity Inherence Classifier with optional Tier 2 /…, DecisionCategory, Unit tests for deterministic classification decision logic., test_contextual_inherent() (+23 more) Nodes (26): LLMFallbackAdapter, Executes LLM fallback for ambiguous boundary cases. Returns a refined…, Parses and validates structured JSON response from LLM., Optional adapter for LLM fallback boundary disambiguation., Returns True if an API key or custom provider function is configured., Constructs an expert-engineered prompt for multilingual entity inherence…, Any, Suíte de Testes para o Adaptador de Fallback para LLM (Tier 3) do Classificador… (+18 more)
### Community 153 - "convert_html_to_markdown" ### Community 153 - "convert_html_to_markdown"
Cohesion: 0.33 Cohesion: 0.33
@@ -650,9 +650,9 @@ Nodes (3): 1. Input JSON Schema, 2. Output JSON Schema, JSON Schema Contract: De
Cohesion: 0.67 Cohesion: 0.67
Nodes (3): 5.1 Incluído, 5.2 Fora do escopo, 5. Escopo Nodes (3): 5.1 Incluído, 5.2 Fora do escopo, 5. Escopo
### Community 165 - ".disambiguate" ### Community 165 - "InherenceClassifier"
Cohesion: 0.25 Cohesion: 0.11
Nodes (4): Executes LLM fallback for ambiguous boundary cases. Returns a refined…, Parses and validates structured JSON response from LLM., Returns True if an API key or custom provider function is configured., Constructs an expert-engineered prompt for multilingual entity inherence… Nodes (30): InherenceClassifier, Tier 1 Deterministic NLP Entity Inherence Classifier with optional Tier 2 /…, DecisionCategory, Content about city/state governance of São Paulo against ECP for São Paulo FC., test_adversarial_sao_paulo_city_vs_fc(), Unit tests for deterministic classification decision logic., test_contextual_inherent(), test_direct_inherent() (+22 more)
### Community 166 - "remove_duplicate_initial_h1" ### Community 166 - "remove_duplicate_initial_h1"
Cohesion: 0.50 Cohesion: 0.50
@@ -667,16 +667,16 @@ Nodes (4): Remove o primeiro título H1 do corpo somente quando ele for igual ao
_Questions this graph is uniquely positioned to answer:_ _Questions this graph is uniquely positioned to answer:_
- **Why does `PRD — Conversão de artigo JSON para Markdown` connect `PRD — Conversão de artigo JSON para Markdown` to `8. Regras funcionais`, `12. Critérios de aceite`, `11. Requisitos não funcionais`, `9. Interface CLI`, `13. Estratégia de testes`, `6. Contrato de entrada`, `5. Escopo`?** - **Why does `PRD — Conversão de artigo JSON para Markdown` connect `PRD — Conversão de artigo JSON para Markdown` to `8. Regras funcionais`, `12. Critérios de aceite`, `11. Requisitos não funcionais`, `9. Interface CLI`, `13. Estratégia de testes`, `6. Contrato de entrada`, `5. Escopo`?**
_High betweenness centrality (0.008) - this node is a cross-community bridge._ _High betweenness centrality (0.006) - this node is a cross-community bridge._
- **Why does `Implementation Plan: Convert Article JSON to Markdown` connect `Implementation Plan: Convert Article JSON to Markdown` to `005-convert-json-markdown/plan.md`?**
_High betweenness centrality (0.004) - this node is a cross-community bridge._
- **Why does `Tasks: Convert Article JSON to Markdown` connect `Tasks: Convert Article JSON to Markdown` to `005-convert-json-markdown/plan.md`?** - **Why does `Tasks: Convert Article JSON to Markdown` connect `Tasks: Convert Article JSON to Markdown` to `005-convert-json-markdown/plan.md`?**
_High betweenness centrality (0.004) - this node is a cross-community bridge._ _High betweenness centrality (0.004) - this node is a cross-community bridge._
- **Are the 10 inferred relationships involving `ECPSnapshot` (e.g. with `main()` and `BaseNLPAdapter`) actually correct?** - **Why does `ECPSnapshot` connect `ECPSnapshot` to `main`, `InherenceClassifier`, `ClassificationResult`, `test_adversarial.py`, `classifier.py`, `LLMFallbackAdapter`?**
_`ECPSnapshot` has 10 INFERRED edges - model-reasoned connections that need verification._ _High betweenness centrality (0.004) - this node is a cross-community bridge._
- **Are the 6 inferred relationships involving `InherenceClassifier` (e.g. with `LocalEmbeddingsAdapter` and `LLMFallbackAdapter`) actually correct?** - **Are the 16 inferred relationships involving `ECPSnapshot` (e.g. with `main()` and `BaseNLPAdapter`) actually correct?**
_`InherenceClassifier` has 6 INFERRED edges - model-reasoned connections that need verification._ _`ECPSnapshot` has 16 INFERRED edges - model-reasoned connections that need verification._
- **Are the 18 inferred relationships involving `DecisionCategory` (e.g. with `LLMFallbackAdapter` and `InherenceClassifier`) actually correct?** - **Are the 7 inferred relationships involving `InherenceClassifier` (e.g. with `LocalEmbeddingsAdapter` and `LLMFallbackAdapter`) actually correct?**
_`DecisionCategory` has 18 INFERRED edges - model-reasoned connections that need verification._ _`InherenceClassifier` has 7 INFERRED edges - model-reasoned connections that need verification._
- **Are the 24 inferred relationships involving `DecisionCategory` (e.g. with `LLMFallbackAdapter` and `InherenceClassifier`) actually correct?**
_`DecisionCategory` has 24 INFERRED edges - model-reasoned connections that need verification._
- **Are the 4 inferred relationships involving `LLMFallbackAdapter` (e.g. with `ClassificationResult` and `DecisionCategory`) actually correct?** - **Are the 4 inferred relationships involving `LLMFallbackAdapter` (e.g. with `ClassificationResult` and `DecisionCategory`) actually correct?**
_`LLMFallbackAdapter` has 4 INFERRED edges - model-reasoned connections that need verification._ _`LLMFallbackAdapter` has 4 INFERRED edges - model-reasoned connections that need verification._
File diff suppressed because it is too large Load Diff
+12 -6
View File
@@ -330,9 +330,9 @@
"semantic_hash": "" "semantic_hash": ""
}, },
"src/adapters/llm.py": { "src/adapters/llm.py": {
"mtime": 1787320797.2485664, "mtime": 1787321086.752706,
"seen": 1787320818.2953389, "seen": 1787321205.5144775,
"ast_hash": "a5cd6f66048ee1d443c2c91ae9947a14", "ast_hash": "21ac74a13ac5dfad7db498b17165f8b3",
"semantic_hash": "" "semantic_hash": ""
}, },
"src/classifier.py": { "src/classifier.py": {
@@ -654,9 +654,9 @@
"semantic_hash": "" "semantic_hash": ""
}, },
"README.md": { "README.md": {
"mtime": 1787320219.5852203, "mtime": 1787321187.0812356,
"seen": 1787320239.0933797, "seen": 1787321205.5201268,
"ast_hash": "ce59670fbaebc5e408a30a1009a58d12", "ast_hash": "aedfaf7a245288227952a2e28e7e7b13",
"semantic_hash": "" "semantic_hash": ""
}, },
"scripts/extract_article_contents.py": { "scripts/extract_article_contents.py": {
@@ -916,5 +916,11 @@
"seen": 1787320818.2967606, "seen": 1787320818.2967606,
"ast_hash": "e5d98de814ceeecd8bd601a7c206e92d", "ast_hash": "e5d98de814ceeecd8bd601a7c206e92d",
"semantic_hash": "" "semantic_hash": ""
},
"tests/test_e2e_text_analysis_pipeline.py": {
"mtime": 1787321086.751707,
"seen": 1787321205.5156026,
"ast_hash": "3a2d47f2ffcf8371ffdf90bb797b5346",
"semantic_hash": ""
} }
} }
+69 -59
View File
@@ -1,16 +1,16 @@
# Graph Report - TextNLPClassifierApp (2026-08-21) # Graph Report - TextNLPClassifierApp (2026-08-21)
## Corpus Check ## Corpus Check
- 202 files · ~112,197 words - 203 files · ~114,894 words
- Verdict: corpus is large enough that graph structure adds value. - Verdict: corpus is large enough that graph structure adds value.
## Summary ## Summary
- 1579 nodes · 2030 edges · 168 communities (120 shown, 48 thin omitted) - 1651 nodes · 2220 edges · 170 communities (122 shown, 48 thin omitted)
- Extraction: 96% EXTRACTED · 4% INFERRED · 0% AMBIGUOUS · INFERRED: 73 edges (avg confidence: 0.95) - Extraction: 94% EXTRACTED · 6% INFERRED · 0% AMBIGUOUS · INFERRED: 127 edges (avg confidence: 0.95)
- Token cost: 0 input · 0 output - Token cost: 0 input · 0 output
## Graph Freshness ## Graph Freshness
- Built from commit: `bae14405` - Built from commit: `a874b98d`
- Run `git rev-parse HEAD` and compare to check if the graph is stale. - Run `git rev-parse HEAD` and compare to check if the graph is stale.
- Run `graphify update .` after code changes (no API cost). - Run `graphify update .` after code changes (no API cost).
@@ -58,7 +58,7 @@
- 2. Standard Streams & Exit Codes - 2. Standard Streams & Exit Codes
- ClassificationResult - ClassificationResult
- test_adversarial.py - test_adversarial.py
- classifier.py - LLMFallbackAdapter
- test_convert_article_to_markdown.py - test_convert_article_to_markdown.py
- content_northvolt_de.md - content_northvolt_de.md
- content_presal_pt.md - content_presal_pt.md
@@ -111,7 +111,7 @@
- Extraction Pipeline Checklist: Article Content Multi-Engine Extractor - Extraction Pipeline Checklist: Article Content Multi-Engine Extractor
- parametrize - parametrize
- main - main
- ECPSnapshot - classifier.py
- Feature Specification: Multilingual NLP Entity Inherence Classifier (POC) - Feature Specification: Multilingual NLP Entity Inherence Classifier (POC)
- 4. Requisitos Funcionais (FR) - 4. Requisitos Funcionais (FR)
- Tasks: Article Content Multi-Engine Extractor - Tasks: Article Content Multi-Engine Extractor
@@ -130,7 +130,7 @@
- Tasks: Deterministic Article Content Selection - Tasks: Deterministic Article Content Selection
- select_article_extractor - select_article_extractor
- process_batch - process_batch
- detect_language - test_models.py
- test_select_article_extractor.py - test_select_article_extractor.py
- Feature Specification: Deterministic Content Selection - Feature Specification: Deterministic Content Selection
- 2. Entity Descriptions & Fields - 2. Entity Descriptions & Fields
@@ -157,10 +157,10 @@
- Specification Quality Checklist: Convert Article JSON to Markdown - Specification Quality Checklist: Convert Article JSON to Markdown
- CLI Contract: `convert_article_to_markdown.py` - CLI Contract: `convert_article_to_markdown.py`
- 9. Interface CLI - 9. Interface CLI
- sample_rss_xml - get_hl_gl_ceid
- 13. Estratégia de testes - 13. Estratégia de testes
- 6. Contrato de entrada - 6. Contrato de entrada
- LLMFallbackAdapter - ECPSnapshot
- convert_html_to_markdown - convert_html_to_markdown
- JSON Schema Contract: Deterministic Article Content Selection - JSON Schema Contract: Deterministic Article Content Selection
- 5. Escopo - 5. Escopo
@@ -176,35 +176,37 @@
- InherenceClassifier - InherenceClassifier
- remove_duplicate_initial_h1 - remove_duplicate_initial_h1
- test_normalize_scalar_whitespace_collapsing - test_normalize_scalar_whitespace_collapsing
- .disambiguate
- test_funnel_cli_subprocess_end_to_end
## God Nodes (most connected - your core abstractions) ## God Nodes (most connected - your core abstractions)
1. `ECPSnapshot` - 49 edges 1. `ECPSnapshot` - 78 edges
2. `InherenceClassifier` - 36 edges 2. `InherenceClassifier` - 62 edges
3. `DecisionCategory` - 35 edges 3. `DecisionCategory` - 62 edges
4. `LLMFallbackAdapter` - 32 edges 4. `LLMFallbackAdapter` - 48 edges
5. `ClassificationResult` - 26 edges 5. `ClassificationResult` - 29 edges
6. `select_article_extractor()` - 23 edges 6. `select_article_extractor()` - 23 edges
7. `ExtractorName` - 21 edges 7. `ExtractorName` - 21 edges
8. `PRD — Conversão de artigo JSON para Markdown` - 16 edges 8. `main()` - 20 edges
9. `process_batch()` - 15 edges 9. `PRD — Conversão de artigo JSON para Markdown` - 16 edges
10. `8. Regras funcionais` - 15 edges 10. `process_batch()` - 15 edges
## Surprising Connections (you probably didn't know these) ## Surprising Connections (you probably didn't know these)
- `main()` --uses--> `ECPSnapshot` [INFERRED] - `main()` --uses--> `ECPSnapshot` [INFERRED]
classify.py → src/models.py classify.py → src/models.py
- `test_extract_google_news_orchestration_mocked()` --uses--> `ExtractionResult` [INFERRED] - `main()` --uses--> `ErrorCode` [INFERRED]
classify.py → src/models.py
- `test_e2e_extract_google_news_live_pipeline()` --uses--> `ExtractionResult` [INFERRED]
tests/test_extract_google_news.py → scripts/extract_google_news.py tests/test_extract_google_news.py → scripts/extract_google_news.py
- `test_llm_adapter_interface()` --calls--> `LLMFallbackAdapter` [EXTRACTED] - `test_llm_adapter_interface()` --calls--> `LLMFallbackAdapter` [EXTRACTED]
tests/test_adapters.py → src/adapters/llm.py tests/test_adapters.py → src/adapters/llm.py
- `classifier()` --uses--> `InherenceClassifier` [INFERRED] - `classifier()` --uses--> `InherenceClassifier` [INFERRED]
tests/test_benchmark_24.py → src/classifier.py tests/test_benchmark_24.py → src/classifier.py
- `test_adversarial_apple_fruit_recipe()` --uses--> `DecisionCategory` [INFERRED]
tests/test_adversarial.py → src/models.py
## Import Cycles ## Import Cycles
- None detected. - None detected.
## Communities (168 total, 48 thin omitted) ## Communities (170 total, 48 thin omitted)
### Community 0 - "Task Planning" ### Community 0 - "Task Planning"
Cohesion: 0.07 Cohesion: 0.07
@@ -348,15 +350,15 @@ Nodes (6): 1.1 Arguments & Options, 1. Command Line Interface, 2.1 Exit Codes, 2
### Community 45 - "ClassificationResult" ### Community 45 - "ClassificationResult"
Cohesion: 0.10 Cohesion: 0.10
Nodes (20): ABC, BaseNLPAdapter, Base abstract adapter interface for optional Tier 2 / Tier 3 NLP enhancers., Abstract interface for pluggable NLP classification adapters., Return True if the underlying provider or model is installed and configured., Compute semantic similarity score between text and a set of candidate terms., Optionally refine an ambiguous classification result., LocalEmbeddingsAdapter (+12 more) Nodes (19): ABC, BaseNLPAdapter, Base abstract adapter interface for optional Tier 2 / Tier 3 NLP enhancers., Abstract interface for pluggable NLP classification adapters., Return True if the underlying provider or model is installed and configured., Compute semantic similarity score between text and a set of candidate terms., Optionally refine an ambiguous classification result., LocalEmbeddingsAdapter (+11 more)
### Community 46 - "test_adversarial.py" ### Community 46 - "test_adversarial.py"
Cohesion: 0.11 Cohesion: 0.10
Nodes (19): RelatedEntity, Adversarial and robustness test suite for Multilingual NLP Entity Inherence…, Run CLI via subprocess without --output and verify stdout is pure parseable…, Run CLI via subprocess with empty content and verify error code and exit code., Run CLI via subprocess with missing target_name and verify error payload., Run CLI via subprocess with corrupted JSON and verify error payload., Content about apple fruit/culinary recipe against Apple Inc. tech entity., High-weight related entity mentioned in passing without required domain anchors. (+11 more) Nodes (21): RelatedEntity, Adversarial and robustness test suite for Multilingual NLP Entity Inherence…, Run CLI via subprocess without --output and verify stdout is pure parseable…, Run CLI via subprocess with empty content and verify error code and exit code., Content about city/state governance of São Paulo against ECP for São Paulo FC., Run CLI via subprocess with missing target_name and verify error payload., Run CLI via subprocess with corrupted JSON and verify error payload., High-weight related entity mentioned in passing without required domain anchors. (+13 more)
### Community 47 - "classifier.py" ### Community 47 - "LLMFallbackAdapter"
Cohesion: 0.16 Cohesion: 0.06
Nodes (15): count_phrase_occurrences(), match_phrase_in_text(), Core deterministic classification engine (Tier 1 core)., Check if a normalized phrase appears in normalized text with word boundary…, Count occurrences of a phrase in text., Classify inherence of content against an ECP snapshot., extract_evidence_snippets(), extract_sentences() (+7 more) Nodes (43): LLMFallbackAdapter, Optional adapter for LLM fallback boundary disambiguation., Any, parametrize, Suíte de Testes Exaustiva para o Classificador de Inerência (classify.py e…, Cenário 4.1: Caso ambíguo elevado para DIRECT_INHERENT pelo LLM., Cenário 4.2: Caso ambíguo elevado para CONTEXTUAL_INHERENT pelo LLM., Cenário 4.3: LLM confirma categoricamente que a menção é periférica /… (+35 more)
### Community 48 - "test_convert_article_to_markdown.py" ### Community 48 - "test_convert_article_to_markdown.py"
Cohesion: 0.08 Cohesion: 0.08
@@ -371,16 +373,16 @@ Cohesion: 0.08
Nodes (24): 1. Visão geral (arquitetura), 2.1 DTO de entrada (`googlenews_etl/application/dtos/extract_news_dto.py`), 2.2 Value Object de validação (`googlenews_etl/domain/entities/search_query.py`), 2. Entrada, 3.1 O caso de uso (`googlenews_etl/application/use_cases/extract_news_use_case.py`), 3.2 A porta (`googlenews_etl/domain/ports/news_extractor_port.py`), 3.3.1 Inicialização: sessão HTTP com impersonação de browser, 3.3.2 Mapeamento idioma → parâmetros `hl`/`gl` (`_get_hl_gl`) (+16 more) Nodes (24): 1. Visão geral (arquitetura), 2.1 DTO de entrada (`googlenews_etl/application/dtos/extract_news_dto.py`), 2.2 Value Object de validação (`googlenews_etl/domain/entities/search_query.py`), 2. Entrada, 3.1 O caso de uso (`googlenews_etl/application/use_cases/extract_news_use_case.py`), 3.2 A porta (`googlenews_etl/domain/ports/news_extractor_port.py`), 3.3.1 Inicialização: sessão HTTP com impersonação de browser, 3.3.2 Mapeamento idioma → parâmetros `hl`/`gl` (`_get_hl_gl`) (+16 more)
### Community 82 - "extract_google_news.py" ### Community 82 - "extract_google_news.py"
Cohesion: 0.15 Cohesion: 0.20
Nodes (18): extract_google_news(), _fetch_rss_content(), get_hl_gl_ceid(), NewsArticle, _normalize_text_for_comparison(), parse_google_news_rss(), Mapeia idioma e locale para os parâmetros hl, gl e ceid do Google News., Remove pontuação e espaços extras para comparação de redundância. (+10 more) Nodes (14): extract_google_news(), _fetch_rss_content(), NewsArticle, _normalize_text_for_comparison(), parse_google_news_rss(), Remove pontuação e espaços extras para comparação de redundância., Parseia o XML do RSS do Google News e extrai os itens estruturados., Resolve em paralelo as URLs intermediárias do Google News para os links finais… (+6 more)
### Community 83 - "ExtractionResult" ### Community 83 - "ExtractionResult"
Cohesion: 0.29 Cohesion: 0.29
Nodes (5): ExtractionResult, Any, Resultado consolidado da extração., Valida E2E o fluxo completo de busca, parsing e resolução de URLs reais ao vivo., test_e2e_extract_google_news_live_pipeline() Nodes (5): ExtractionResult, Any, Resultado consolidado da extração., Valida a consolidação do ExtractionResult a partir da busca mockada com URLs…, test_extract_google_news_orchestration_mocked()
### Community 84 - "test_extract_google_news.py" ### Community 84 - "test_extract_google_news.py"
Cohesion: 0.15 Cohesion: 0.16
Nodes (15): Resolve a URL intermediária do Google News para a URL real do veículo., resolve_article_url(), Testes unitários e de integração para o Extrator de Manchetes do Google News.…, Valida fallback gracioso de URL quando não é link do Google News ou em erro., Valida resolução bem-sucedida de URL do Google News para o portal destino., Valida E2E que o decodificador resolve uma URL real do Google News para o…, Valida o mapeamento padrão de idiomas para pares (hl, gl, ceid)., Valida a sobrescrita geográfica quando o argumento locale é especificado. (+7 more) Nodes (14): Resolve a URL intermediária do Google News para a URL real do veículo., resolve_article_url(), fixture, Testes unitários e de integração para o Extrator de Manchetes do Google News.…, Valida o parsing do feed RSS, higienização de tags HTML e deduplicação., Valida fallback gracioso de URL quando não é link do Google News ou em erro., Valida resolução bem-sucedida de URL do Google News para o portal destino., Valida E2E que o decodificador resolve uma URL real do Google News para o… (+6 more)
### Community 85 - "Implementation Tasks: Google News Headlines Extractor" ### Community 85 - "Implementation Tasks: Google News Headlines Extractor"
Cohesion: 0.14 Cohesion: 0.14
@@ -400,7 +402,7 @@ Nodes (7): Architecture & Pipeline, Documentation (this feature), Implementation
### Community 90 - "SearchQuery" ### Community 90 - "SearchQuery"
Cohesion: 0.20 Cohesion: 0.20
Nodes (6): Value Object com parâmetros de busca validados., SearchQuery, Valida a consolidação do ExtractionResult a partir da busca mockada com URLs…, Valida as regras de negócio e limites de SearchQuery., test_extract_google_news_orchestration_mocked(), test_search_query_validation() Nodes (6): Value Object com parâmetros de busca validados., SearchQuery, Valida E2E o fluxo completo de busca, parsing e resolução de URLs reais ao vivo., Valida as regras de negócio e limites de SearchQuery., test_e2e_extract_google_news_live_pipeline(), test_search_query_validation()
### Community 91 - "1. Technical Decisions & Tradeoffs" ### Community 91 - "1. Technical Decisions & Tradeoffs"
Cohesion: 0.25 Cohesion: 0.25
@@ -435,12 +437,12 @@ Cohesion: 0.22
Nodes (9): parametrize, Garante aceitação de URLs absolutas com esquema HTTP e HTTPS válidos., Garante rejeição de esquemas não permitidos, URLs relativas e strings vazias., Garante que a ausência de corpo no extrator selecionado NUNCA faça fallback…, Garante que todos os placeholders documentados no PRD sejam descartados…, test_normalize_scalar_placeholders_discarded(), test_resolve_article_body_strict_isolation_all_extractors(), test_validate_url_invalid_schemes() (+1 more) Nodes (9): parametrize, Garante aceitação de URLs absolutas com esquema HTTP e HTTPS válidos., Garante rejeição de esquemas não permitidos, URLs relativas e strings vazias., Garante que a ausência de corpo no extrator selecionado NUNCA faça fallback…, Garante que todos os placeholders documentados no PRD sejam descartados…, test_normalize_scalar_placeholders_discarded(), test_resolve_article_body_strict_isolation_all_extractors(), test_validate_url_invalid_schemes() (+1 more)
### Community 100 - "main" ### Community 100 - "main"
Cohesion: 0.17 Cohesion: 0.14
Nodes (15): emit_error(), main(), parse_args(), Namespace, ClassificationError, ErrorCode, Enum, str (+7 more) Nodes (20): main(), Path, Cenário 6.1: Caminho de ECP inexistente -> Exit Code 1, error_code:…, Cenário 6.2: Arquivo ECP com sintaxe JSON corrompida., Cenário 6.3: Valida erro para falta de cada um dos campos obrigatórios do ECP., Cenário 6.4: Caminho de arquivo Markdown inexistente., Cenário 6.5: Arquivo Markdown vazio ou contendo apenas espaços em branco., Cenário 6.6: A flag -o / --output cria diretórios aninhados automaticamente. (+12 more)
### Community 101 - "ECPSnapshot" ### Community 101 - "classifier.py"
Cohesion: 0.18 Cohesion: 0.17
Nodes (12): ECPSnapshot, Any, classifier(), fixture, parametrize, Controlled 24-case benchmark suite for Multilingual NLP Entity Inherence…, test_benchmark_case(), Unit tests for ECP models, schema validation, and structured error handling. (+4 more) Nodes (12): emit_error(), parse_args(), Namespace, Core deterministic classification engine (Tier 1 core)., ErrorCode, MatchedGraphEntity, Enum, str (+4 more)
### Community 102 - "Feature Specification: Multilingual NLP Entity Inherence Classifier (POC)" ### Community 102 - "Feature Specification: Multilingual NLP Entity Inherence Classifier (POC)"
Cohesion: 0.14 Cohesion: 0.14
@@ -514,13 +516,13 @@ Nodes (35): CandidateStatus, extract_candidate_data(), ExtractorName, Any, Enum,
Cohesion: 0.11 Cohesion: 0.11
Nodes (24): atomic_save_json(), process_batch(), Path, Salva dados em JSON de forma atômica utilizando arquivo temporário e rename., Lê o JSON de entrada, valida a estrutura, processa todos os artigos e grava o…, Path, CT-012: A entrada já contém selected_extractor -> Recalcular e substituir…, CT-013: articles está vazio -> Gerar saída válida com articles vazio. (+16 more) Nodes (24): atomic_save_json(), process_batch(), Path, Salva dados em JSON de forma atômica utilizando arquivo temporário e rename., Lê o JSON de entrada, valida a estrutura, processa todos os artigos e grava o…, Path, CT-012: A entrada já contém selected_extractor -> Recalcular e substituir…, CT-013: articles está vazio -> Gerar saída válida com articles vazio. (+16 more)
### Community 121 - "detect_language" ### Community 121 - "test_models.py"
Cohesion: 0.19 Cohesion: 0.07
Nodes (16): detect_language(), extract_words(), normalize_text(), Lightweight multilingual language detection and text normalization., Normalize text by converting to lowercase and stripping combining diacritical…, Tokenize text into lowercase alphanumeric words., Detect the ISO-639-1 language code of text among supported languages (pt, en,…, Unit tests for language detection and text normalization. (+8 more) Nodes (37): count_phrase_occurrences(), match_phrase_in_text(), Check if a normalized phrase appears in normalized text with word boundary…, Count occurrences of a phrase in text., Classify inherence of content against an ECP snapshot., detect_language(), extract_words(), normalize_text() (+29 more)
### Community 122 - "test_select_article_extractor.py" ### Community 122 - "test_select_article_extractor.py"
Cohesion: 0.18 Cohesion: 0.18
Nodes (16): generate_shingles(), normalize_text(), Executa a normalização determinística para comparação: 1. Decodificar entidades…, Gera conjunto de shingles ordenados de tamanho window_size (padrão 5). - Se…, Suíte de Testes Automatizados para o Seletor Determinístico de Extrator. Cobre…, Garante que marcação de imagem Markdown ![alt](url) seja descartada e link…, E2E: Executa CLI com arquivo inexistente e valida código 1 e mensagem no stderr., test_e2e_cli_subprocess_missing_file() (+8 more) Nodes (16): generate_shingles(), normalize_text(), Executa a normalização determinística para comparação: 1. Decodificar entidades…, Gera conjunto de shingles ordenados de tamanho window_size (padrão 5). - Se…, Suíte de Testes Automatizados para o Seletor Determinístico de Extrator. Cobre…, Garante que marcação de imagem Markdown ![alt](url) seja descartada e link…, E2E: Executa scripts/select_article_extractor.py como subprocesso real na linha…, test_e2e_cli_subprocess_real_execution() (+8 more)
### Community 123 - "Feature Specification: Deterministic Content Selection" ### Community 123 - "Feature Specification: Deterministic Content Selection"
Cohesion: 0.17 Cohesion: 0.17
@@ -622,9 +624,9 @@ Nodes (5): 1. Script Signature, 2. Command-Line Arguments, 3. Exit Codes, 4. Sta
Cohesion: 0.40 Cohesion: 0.40
Nodes (5): 9.1 Script, 9.2 Argumentos, 9.3 Exemplos, 9.4 Saída do processo, 9. Interface CLI Nodes (5): 9.1 Script, 9.2 Argumentos, 9.3 Exemplos, 9.4 Saída do processo, 9. Interface CLI
### Community 149 - "sample_rss_xml" ### Community 149 - "get_hl_gl_ceid"
Cohesion: 0.67 Cohesion: 0.25
Nodes (3): fixture, Fixture que fornece o conteúdo do XML de exemplo para testes offline., sample_rss_xml() Nodes (8): get_hl_gl_ceid(), Mapeia idioma e locale para os parâmetros hl, gl e ceid do Google News., Valida o mapeamento padrão de idiomas para pares (hl, gl, ceid)., Valida a sobrescrita geográfica quando o argumento locale é especificado., Valida fallback dinâmico para idiomas regionais não listados explicitamente., test_get_hl_gl_ceid_default_mappings(), test_get_hl_gl_ceid_dynamic_fallback(), test_get_hl_gl_ceid_with_custom_locale()
### Community 150 - "13. Estratégia de testes" ### Community 150 - "13. Estratégia de testes"
Cohesion: 0.50 Cohesion: 0.50
@@ -634,9 +636,9 @@ Nodes (4): 13.1 Testes unitários, 13.2 Testes de integração do CLI, 13.3 Caso
Cohesion: 0.50 Cohesion: 0.50
Nodes (4): 6.1 Formato, 6.2 Valores aceitos para `selected_extractor`, 6.3 Campos obrigatórios após a resolução, 6. Contrato de entrada Nodes (4): 6.1 Formato, 6.2 Valores aceitos para `selected_extractor`, 6.3 Campos obrigatórios após a resolução, 6. Contrato de entrada
### Community 152 - "LLMFallbackAdapter" ### Community 152 - "ECPSnapshot"
Cohesion: 0.08 Cohesion: 0.13
Nodes (26): LLMFallbackAdapter, Executes LLM fallback for ambiguous boundary cases. Returns a refined…, Parses and validates structured JSON response from LLM., Optional adapter for LLM fallback boundary disambiguation., Returns True if an API key or custom provider function is configured., Constructs an expert-engineered prompt for multilingual entity inherence…, Any, Suíte de Testes para o Adaptador de Fallback para LLM (Tier 3) do Classificador… (+18 more) Nodes (20): ECPSnapshot, parametrize, test_benchmark_case(), Suíte de Testes para o Adaptador de Fallback para LLM (Tier 3) do Classificador…, Valida extração de JSON quando a resposta do LLM vem formatada em bloco…, Valida que respostas corrompidas ou JSONs sem campos obrigatórios retornem None…, Garante que o classificador dispare o Tier 3 LLM para casos ambíguos…, Garante que casos claros (alta confiança e alta densidade de âncoras) NÃO… (+12 more)
### Community 153 - "convert_html_to_markdown" ### Community 153 - "convert_html_to_markdown"
Cohesion: 0.33 Cohesion: 0.33
@@ -651,13 +653,21 @@ Cohesion: 0.67
Nodes (3): 5.1 Incluído, 5.2 Fora do escopo, 5. Escopo Nodes (3): 5.1 Incluído, 5.2 Fora do escopo, 5. Escopo
### Community 165 - "InherenceClassifier" ### Community 165 - "InherenceClassifier"
Cohesion: 0.11 Cohesion: 0.07
Nodes (30): InherenceClassifier, Tier 1 Deterministic NLP Entity Inherence Classifier with optional Tier 2 /…, DecisionCategory, Content about city/state governance of São Paulo against ECP for São Paulo FC., test_adversarial_sao_paulo_city_vs_fc(), Unit tests for deterministic classification decision logic., test_contextual_inherent(), test_direct_inherent() (+22 more) Nodes (45): InherenceClassifier, Tier 1 Deterministic NLP Entity Inherence Classifier with optional Tier 2 /…, DecisionCategory, Content about apple fruit/culinary recipe against Apple Inc. tech entity., test_adversarial_apple_fruit_recipe(), Unit tests for deterministic classification decision logic., test_contextual_inherent(), test_direct_inherent() (+37 more)
### Community 166 - "remove_duplicate_initial_h1" ### Community 166 - "remove_duplicate_initial_h1"
Cohesion: 0.50 Cohesion: 0.50
Nodes (4): Remove o primeiro título H1 do corpo somente quando ele for igual ao título…, remove_duplicate_initial_h1(), Testa remoção de H1 inicial coincidente com título com variações de espaços e…, test_remove_duplicate_initial_h1_exact_and_variations() Nodes (4): Remove o primeiro título H1 do corpo somente quando ele for igual ao título…, remove_duplicate_initial_h1(), Testa remoção de H1 inicial coincidente com título com variações de espaços e…, test_remove_duplicate_initial_h1_exact_and_variations()
### Community 168 - ".disambiguate"
Cohesion: 0.25
Nodes (4): Executes LLM fallback for ambiguous boundary cases. Returns a refined…, Parses and validates structured JSON response from LLM., Returns True if an API key or custom provider function is configured., Constructs an expert-engineered prompt for multilingual entity inherence…
### Community 169 - "test_funnel_cli_subprocess_end_to_end"
Cohesion: 0.67
Nodes (3): Path, Valida o contrato CLI completo classify.py com saída em arquivo JSON e flags…, test_funnel_cli_subprocess_end_to_end()
## Knowledge Gaps ## Knowledge Gaps
- **696 isolated node(s):** `text-nlp-classifier`, `MatchedGraphEntity`, `graphify`, `Usage`, `What graphify is for` (+691 more) - **696 isolated node(s):** `text-nlp-classifier`, `MatchedGraphEntity`, `graphify`, `Usage`, `What graphify is for` (+691 more)
These have ≤1 connection - possible missing edges or undocumented components. These have ≤1 connection - possible missing edges or undocumented components.
@@ -666,17 +676,17 @@ Nodes (4): Remove o primeiro título H1 do corpo somente quando ele for igual ao
## Suggested Questions ## Suggested Questions
_Questions this graph is uniquely positioned to answer:_ _Questions this graph is uniquely positioned to answer:_
- **Why does `ECPSnapshot` connect `ECPSnapshot` to `main`, `classifier.py`, `InherenceClassifier`, `.disambiguate`, `ClassificationResult`, `test_adversarial.py`, `LLMFallbackAdapter`, `test_models.py`?**
_High betweenness centrality (0.009) - this node is a cross-community bridge._
- **Why does `PRD — Conversão de artigo JSON para Markdown` connect `PRD — Conversão de artigo JSON para Markdown` to `8. Regras funcionais`, `12. Critérios de aceite`, `11. Requisitos não funcionais`, `9. Interface CLI`, `13. Estratégia de testes`, `6. Contrato de entrada`, `5. Escopo`?** - **Why does `PRD — Conversão de artigo JSON para Markdown` connect `PRD — Conversão de artigo JSON para Markdown` to `8. Regras funcionais`, `12. Critérios de aceite`, `11. Requisitos não funcionais`, `9. Interface CLI`, `13. Estratégia de testes`, `6. Contrato de entrada`, `5. Escopo`?**
_High betweenness centrality (0.006) - this node is a cross-community bridge._
- **Why does `Tasks: Convert Article JSON to Markdown` connect `Tasks: Convert Article JSON to Markdown` to `005-convert-json-markdown/plan.md`?**
_High betweenness centrality (0.004) - this node is a cross-community bridge._ _High betweenness centrality (0.004) - this node is a cross-community bridge._
- **Why does `ECPSnapshot` connect `ECPSnapshot` to `main`, `InherenceClassifier`, `ClassificationResult`, `test_adversarial.py`, `classifier.py`, `LLMFallbackAdapter`?** - **Why does `InherenceClassifier` connect `InherenceClassifier` to `main`, `classifier.py`, `ClassificationResult`, `test_adversarial.py`, `LLMFallbackAdapter`, `ECPSnapshot`, `test_models.py`?**
_High betweenness centrality (0.004) - this node is a cross-community bridge._ _High betweenness centrality (0.004) - this node is a cross-community bridge._
- **Are the 16 inferred relationships involving `ECPSnapshot` (e.g. with `main()` and `BaseNLPAdapter`) actually correct?** - **Are the 42 inferred relationships involving `ECPSnapshot` (e.g. with `main()` and `BaseNLPAdapter`) actually correct?**
_`ECPSnapshot` has 16 INFERRED edges - model-reasoned connections that need verification._ _`ECPSnapshot` has 42 INFERRED edges - model-reasoned connections that need verification._
- **Are the 7 inferred relationships involving `InherenceClassifier` (e.g. with `LocalEmbeddingsAdapter` and `LLMFallbackAdapter`) actually correct?** - **Are the 8 inferred relationships involving `InherenceClassifier` (e.g. with `LocalEmbeddingsAdapter` and `LLMFallbackAdapter`) actually correct?**
_`InherenceClassifier` has 7 INFERRED edges - model-reasoned connections that need verification._ _`InherenceClassifier` has 8 INFERRED edges - model-reasoned connections that need verification._
- **Are the 24 inferred relationships involving `DecisionCategory` (e.g. with `LLMFallbackAdapter` and `InherenceClassifier`) actually correct?** - **Are the 50 inferred relationships involving `DecisionCategory` (e.g. with `LLMFallbackAdapter` and `InherenceClassifier`) actually correct?**
_`DecisionCategory` has 24 INFERRED edges - model-reasoned connections that need verification._ _`DecisionCategory` has 50 INFERRED edges - model-reasoned connections that need verification._
- **Are the 4 inferred relationships involving `LLMFallbackAdapter` (e.g. with `ClassificationResult` and `DecisionCategory`) actually correct?** - **Are the 4 inferred relationships involving `LLMFallbackAdapter` (e.g. with `ClassificationResult` and `DecisionCategory`) actually correct?**
_`LLMFallbackAdapter` has 4 INFERRED edges - model-reasoned connections that need verification._ _`LLMFallbackAdapter` has 4 INFERRED edges - model-reasoned connections that need verification._
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
+1 -1
View File
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
+3815 -804
View File
File diff suppressed because it is too large Load Diff
+12 -6
View File
@@ -336,9 +336,9 @@
"semantic_hash": "" "semantic_hash": ""
}, },
"src/classifier.py": { "src/classifier.py": {
"mtime": 1787320047.0362887, "mtime": 1787321309.8981817,
"seen": 1787320239.025866, "seen": 1787321481.1505442,
"ast_hash": "d6cc674d407a99ab52f6d5156f9b3d8e", "ast_hash": "a4e5dafed12aa4eaf096988b2c6a8ae0",
"semantic_hash": "" "semantic_hash": ""
}, },
"src/language.py": { "src/language.py": {
@@ -654,9 +654,9 @@
"semantic_hash": "" "semantic_hash": ""
}, },
"README.md": { "README.md": {
"mtime": 1787321187.0812356, "mtime": 1787321467.0542295,
"seen": 1787321205.5201268, "seen": 1787321481.1561577,
"ast_hash": "aedfaf7a245288227952a2e28e7e7b13", "ast_hash": "0cde8e800125cbba61a1d7de9d9d2c9e",
"semantic_hash": "" "semantic_hash": ""
}, },
"scripts/extract_article_contents.py": { "scripts/extract_article_contents.py": {
@@ -922,5 +922,11 @@
"seen": 1787321205.5156026, "seen": 1787321205.5156026,
"ast_hash": "3a2d47f2ffcf8371ffdf90bb797b5346", "ast_hash": "3a2d47f2ffcf8371ffdf90bb797b5346",
"semantic_hash": "" "semantic_hash": ""
},
"tests/test_classify_exhaustive_suite.py": {
"mtime": 1787321371.2658408,
"seen": 1787321481.15137,
"ast_hash": "08e3c680d8669b2849a19d73b27b869a",
"semantic_hash": ""
} }
} }
+1 -1
View File
@@ -48,7 +48,7 @@ class InherenceClassifier:
llm_adapter: Optional[Any] = None, llm_adapter: Optional[Any] = None,
) -> None: ) -> None:
self.enable_embeddings = enable_embeddings self.enable_embeddings = enable_embeddings
self.enable_llm = enable_llm or (llm_adapter is not None) self.enable_llm = enable_llm
self._embeddings_adapter = None self._embeddings_adapter = None
self._llm_adapter = llm_adapter self._llm_adapter = llm_adapter
+806
View File
@@ -0,0 +1,806 @@
"""
Suíte de Testes Exaustiva para o Classificador de Inerência (classify.py e src/).
Cobre 100% dos caminhos felizes, infelizes, limiares, de ambiguidade,
fallback de LLM (OpenAI e Gemini), resiliência de API, erros de contrato CLI
e suporte aos 6 idiomas conforme a metodologia da skill-suite-tests.
"""
from __future__ import annotations
import json
from pathlib import Path
from unittest.mock import MagicMock, patch
import pytest
from classify import main
from src.adapters.llm import LLMFallbackAdapter
from src.classifier import InherenceClassifier
from src.models import (
ClassificationResult,
DecisionCategory,
ECPSnapshot,
RelatedEntity,
)
CLASSIFY_CLI = Path(__file__).parent.parent / "classify.py"
# ==============================================================================
# Fixtures Universais
# ==============================================================================
@pytest.fixture
def ecp_tech_corp() -> ECPSnapshot:
return ECPSnapshot(
target_entity_id="ent_tech_corp",
target_name="TechCorp Global",
aliases=["TechCorp", "TechCorp Global", "TCG"],
domain="Tecnologia e Cloud",
anchors=[
"cloud",
"computação em nuvem",
"software",
"inteligência artificial",
"datacenter",
],
negative_anchors=["TechCorp Calçados", "TechCorp Imóveis", "homônimo"],
related_entities=[
RelatedEntity(
entity_id="ent_cloud_subsidiary",
name="CloudPlatform Solutions",
relation_type="SUBSIDIARY_OF",
weight=0.90,
aliases=["CloudPlatform"],
scope="cloud_services",
),
RelatedEntity(
entity_id="ent_ceo_tech",
name="Alan Turing Silva",
relation_type="CEO_OF",
weight=0.80,
aliases=["Alan Turing"],
scope="executive",
),
],
)
# ==============================================================================
# 1. Casos Felizes (Happy Paths) - NLP Determinístico (Tier 1)
# ==============================================================================
def test_happy_path_direct_inherent_with_canonical_and_anchors(ecp_tech_corp: ECPSnapshot):
"""Cenário 1.1: Nome canônico + múltiplas âncoras temáticas -> DIRECT_INHERENT com alta confiança."""
classifier = InherenceClassifier()
content = (
"# TechCorp Global anuncia novo datacenter de computação em nuvem\n\n"
"A TechCorp Global investiu 500 milhões para expandir sua infraestrutura de software "
"e inteligência artificial na América Latina."
)
res = classifier.classify(ecp_tech_corp, content)
assert res.decision == DecisionCategory.DIRECT_INHERENT
assert res.is_inherent is True
assert res.confidence >= 0.90
assert "TechCorp Global" in res.matched_anchors or "TechCorp" in res.matched_anchors
assert len(res.evidence) >= 1
def test_happy_path_direct_inherent_via_alias_and_acronym(ecp_tech_corp: ECPSnapshot):
"""Cenário 1.2: Apenas o alias / sigla 'TCG' é mencionado, com âncoras do domínio."""
classifier = InherenceClassifier()
content = (
"# Inovação em Cloud\n\n"
"A TCG lançou hoje uma nova plataforma de software baseada em computação em nuvem."
)
res = classifier.classify(ecp_tech_corp, content)
assert res.decision == DecisionCategory.DIRECT_INHERENT
assert res.is_inherent is True
assert res.confidence >= 0.85
def test_happy_path_direct_inherent_by_repetition_without_heavy_anchors(ecp_tech_corp: ECPSnapshot):
"""Cenário 1.3: O nome 'TechCorp' aparece 3 vezes no texto, satisfazendo a regra de menção múltipla."""
classifier = InherenceClassifier()
content = (
"# Relatório Corporativo Trimestral\n\n"
"A TechCorp divulgou seus resultados. A TechCorp superou as estimativas de analistas. "
"O conselho da TechCorp aprovou dividendos extraordinários."
)
res = classifier.classify(ecp_tech_corp, content)
assert res.decision == DecisionCategory.DIRECT_INHERENT
assert res.is_inherent is True
assert res.confidence >= 0.85
def test_happy_path_contextual_inherent_via_subsidiary_graph_entity(ecp_tech_corp: ECPSnapshot):
"""Cenário 1.4: Menção da subsidiária 'CloudPlatform Solutions' com âncoras de cloud."""
classifier = InherenceClassifier()
content = (
"# Expansão de Infraestrutura de Nuvem\n\n"
"A CloudPlatform Solutions ativou novos servidores em seu datacenter de computação em nuvem."
)
res = classifier.classify(ecp_tech_corp, content)
assert res.decision == DecisionCategory.CONTEXTUAL_INHERENT
assert res.is_inherent is True
assert res.confidence >= 0.75
assert len(res.graph_matches) >= 1
assert res.graph_matches[0]["name"] == "CloudPlatform Solutions"
def test_happy_path_contextual_inherent_via_executive_graph_entity(ecp_tech_corp: ECPSnapshot):
"""Cenário 1.5: Menção ao CEO no grafo + âncoras de tecnologia."""
classifier = InherenceClassifier()
content = (
"# Discurso na Conferência de Tecnologia\n\n"
"O executivo Alan Turing Silva discursou sobre o futuro da inteligência artificial e software."
)
res = classifier.classify(ecp_tech_corp, content)
assert res.decision == DecisionCategory.CONTEXTUAL_INHERENT
assert res.is_inherent is True
assert any(g["name"] == "Alan Turing Silva" for g in res.graph_matches)
# ==============================================================================
# 2. Casos Infelizes e Rejeições (Sad Paths) - NLP Determinístico (Tier 1)
# ==============================================================================
def test_sad_path_not_related_completely_off_topic(ecp_tech_corp: ECPSnapshot):
"""Cenário 2.1: Conteúdo totalmente desvinculado (culinária/jardinagem)."""
classifier = InherenceClassifier()
content = (
"# Receita de Pão Caseiro Fácil\n\n"
"Misture a farinha, o fermento biológico seco e a água morna. "
"Deixe a massa descansar por 40 minutos em local aquecido."
)
res = classifier.classify(ecp_tech_corp, content)
assert res.decision == DecisionCategory.NOT_RELATED
assert res.is_inherent is False
assert res.confidence >= 0.90
assert len(res.matched_anchors) == 0
def test_sad_path_not_related_generic_domain_without_target_or_graph(ecp_tech_corp: ECPSnapshot):
"""Cenário 2.2: Artigo cita muitas âncoras ('cloud', 'software'), mas NÃO cita a TechCorp nem o grafo."""
classifier = InherenceClassifier()
content = (
"# O Mercado Global de Computação em Nuvem\n\n"
"O setor de computação em nuvem, datacenter e inteligência artificial cresceu 25% este ano."
)
res = classifier.classify(ecp_tech_corp, content)
assert res.decision == DecisionCategory.NOT_RELATED
assert res.is_inherent is False
assert (
"General domain topics mentioned, but target entity or related entities are absent."
in res.rationale
)
def test_sad_path_not_related_negative_anchor_dominance(ecp_tech_corp: ECPSnapshot):
"""Cenário 2.3: Homônimo 'TechCorp Calçados' dispara âncora negativa dominante."""
classifier = InherenceClassifier()
content = (
"# Feira de Moda e Varejo\n\n"
"A TechCorp Calçados apresentou sua nova linha de sandálias de couro para o verão."
)
res = classifier.classify(ecp_tech_corp, content)
assert res.decision == DecisionCategory.NOT_RELATED
assert res.is_inherent is False
assert "TechCorp Calçados" in res.negative_matches
def test_sad_path_not_related_negative_anchor_ties_with_positive_anchor(ecp_tech_corp: ECPSnapshot):
"""Cenário 2.4: 1 âncora negativa e 1 positiva -> prioridade de segurança rejeita para NOT_RELATED."""
classifier = InherenceClassifier()
content = "A TechCorp Calçados adotou um novo software interno de gestão."
res = classifier.classify(ecp_tech_corp, content)
assert res.decision == DecisionCategory.NOT_RELATED
assert res.is_inherent is False
# ==============================================================================
# 3. Casos Limiares e Ambiguidades (Borderline / Tangential)
# ==============================================================================
def test_borderline_tangential_single_passing_mention(ecp_tech_corp: ECPSnapshot):
"""Cenário 3.1: Menção única isolada sem âncoras temáticas -> TANGENTIAL com baixa confiança."""
classifier = InherenceClassifier()
content = "Estávamos caminhando pela avenida e vimos a placa da TechCorp ao longe na esquina."
res = classifier.classify(ecp_tech_corp, content)
assert res.decision == DecisionCategory.TANGENTIAL
assert res.is_inherent is False
assert res.confidence == 0.40
assert any("Low contextual density" in w for w in res.warnings)
def test_borderline_tangential_graph_entity_in_isolation(ecp_tech_corp: ECPSnapshot):
"""Cenário 3.2: Entidade do grafo mencionada sem contexto de domínio -> TANGENTIAL."""
classifier = InherenceClassifier()
content = "Alan Turing Silva participou de uma corrida beneficente no parque no domingo."
res = classifier.classify(ecp_tech_corp, content)
assert res.decision == DecisionCategory.TANGENTIAL
assert res.is_inherent is False
assert res.confidence == 0.45
# ==============================================================================
# 4. Suíte Abrangente de Fallback para LLM (Tier 3)
# ==============================================================================
def test_llm_happy_path_upgrade_tangential_to_direct_inherent(ecp_tech_corp: ECPSnapshot):
"""Cenário 4.1: Caso ambíguo elevado para DIRECT_INHERENT pelo LLM."""
mock_resp = json.dumps(
{
"analysis_summary": "Artigo detalha o projeto estratégico secreto da TechCorp.",
"decision": "DIRECT_INHERENT",
"confidence": 0.95,
"rationale": "Embora a redação use linguagem coloquial, o artigo foca inteiramente na estratégia da TechCorp.",
}
)
adapter = LLMFallbackAdapter(provider_fn=lambda p: mock_resp)
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
content = "A diretoria da TechCorp finalizou as negociações confidenciais da rodada."
res = classifier.classify(ecp_tech_corp, content)
assert res.decision == DecisionCategory.DIRECT_INHERENT
assert res.is_inherent is True
assert res.confidence == 0.95
assert "[Tier 3 LLM]" in res.rationale
assert "[Tier 3 LLM Override applied]" in res.warnings
def test_llm_happy_path_upgrade_tangential_to_contextual_inherent(ecp_tech_corp: ECPSnapshot):
"""Cenário 4.2: Caso ambíguo elevado para CONTEXTUAL_INHERENT pelo LLM."""
mock_resp = json.dumps(
{
"analysis_summary": "Matéria sobre fusão de fornecedores onde a TechCorp é impactada diretamente.",
"decision": "CONTEXTUAL_INHERENT",
"confidence": 0.88,
"rationale": "A TechCorp é parte material do ecossistema afetado pela fusão anunciada.",
}
)
adapter = LLMFallbackAdapter(provider_fn=lambda p: mock_resp)
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
content = "O consórcio fornecedor foi reestruturado e envolverá contratos com a TechCorp."
res = classifier.classify(ecp_tech_corp, content)
assert res.decision == DecisionCategory.CONTEXTUAL_INHERENT
assert res.is_inherent is True
assert res.confidence == 0.88
def test_llm_happy_path_confirmation_of_tangential(ecp_tech_corp: ECPSnapshot):
"""Cenário 4.3: LLM confirma categoricamente que a menção é periférica / irrelevante."""
mock_resp = json.dumps(
{
"analysis_summary": "Crônica sobre trânsito urbano com citação lateral a um outdoor da TechCorp.",
"decision": "TANGENTIAL",
"confidence": 0.97,
"rationale": "A empresa é apenas uma referência visual casual sem relação com a narrativa de trânsito.",
}
)
adapter = LLMFallbackAdapter(provider_fn=lambda p: mock_resp)
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
content = "O tráfego estava parado bem em frente ao painel da TechCorp na autoestrada."
res = classifier.classify(ecp_tech_corp, content)
assert res.decision == DecisionCategory.TANGENTIAL
assert res.is_inherent is False
assert res.confidence == 0.97
def test_llm_happy_path_rejection_to_not_related(ecp_tech_corp: ECPSnapshot):
"""Cenário 4.4: LLM identifica homônimo não mapeado nas regras determinísticas e rebaixa para NOT_RELATED."""
mock_resp = json.dumps(
{
"analysis_summary": "Artigo sobre uma banda de rock indie com nome idêntico.",
"decision": "NOT_RELATED",
"confidence": 0.99,
"rationale": "O texto refere-se a um grupo musical e não à empresa de tecnologia.",
}
)
adapter = LLMFallbackAdapter(provider_fn=lambda p: mock_resp)
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
content = "A banda TechCorp tocou seus novos acordes no festival de música independente."
res = classifier.classify(ecp_tech_corp, content)
assert res.decision == DecisionCategory.NOT_RELATED
assert res.is_inherent is False
assert res.confidence == 0.99
def test_llm_sad_path_llm_disabled_by_default_never_invokes_adapter(ecp_tech_corp: ECPSnapshot):
"""Cenário 4.5: Quando enable_llm=False (padrão), o LLM NUNCA é chamado mesmo em caso limiar."""
called = {"status": False}
def tracking_fn(p: str) -> str:
called["status"] = True
return "{}"
adapter = LLMFallbackAdapter(provider_fn=tracking_fn)
classifier = InherenceClassifier(enable_llm=False, llm_adapter=adapter)
content = "Menção isolada da TechCorp sem contexto algum."
res = classifier.classify(ecp_tech_corp, content)
assert res.decision == DecisionCategory.TANGENTIAL
assert called["status"] is False
def test_llm_sad_path_flag_enabled_without_api_key_or_provider(ecp_tech_corp: ECPSnapshot):
"""Cenário 4.6: enable_llm=True mas sem chaves no ambiente -> degrada sem quebrar, retém Tier 1."""
with patch.dict("os.environ", {}, clear=True):
adapter = LLMFallbackAdapter(api_key="", provider_fn=None)
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
content = "Menção isolada da TechCorp em relatório breve."
res = classifier.classify(ecp_tech_corp, content)
assert res.decision == DecisionCategory.TANGENTIAL
assert res.is_inherent is False
def test_llm_sad_path_network_timeout_graceful_degradation(ecp_tech_corp: ECPSnapshot):
"""Cenário 4.7: API do LLM sofre TimeoutError -> retém Tier 1 e registra aviso em warnings."""
def timeout_fn(p: str) -> str:
raise TimeoutError("Conexão com gateway do LLM excedeu tempo limite de 30s.")
adapter = LLMFallbackAdapter(provider_fn=timeout_fn)
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
content = "A TechCorp esteve presente no evento de premiação."
res = classifier.classify(ecp_tech_corp, content)
assert res.decision == DecisionCategory.TANGENTIAL
assert any("LLM fallback failed" in w for w in res.warnings)
def test_llm_sad_path_http_500_server_error_graceful_degradation(ecp_tech_corp: ECPSnapshot):
"""Cenário 4.8: API do LLM retorna erro 500 / ConnectionError -> retém Tier 1 com aviso."""
def error_500_fn(p: str) -> str:
raise ConnectionError("HTTP 500: Internal Server Error do provedor de IA.")
adapter = LLMFallbackAdapter(provider_fn=error_500_fn)
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
content = "Menção da TechCorp em comunicado à imprensa."
res = classifier.classify(ecp_tech_corp, content)
assert res.decision == DecisionCategory.TANGENTIAL
assert any("LLM fallback failed" in w for w in res.warnings)
def test_llm_sad_path_malformed_json_and_non_json_strings(ecp_tech_corp: ECPSnapshot):
"""Cenário 4.9: LLM retorna texto livre ou JSON quebrado -> parser ignora com segurança."""
def make_bad_provider(resp_text: str):
def _prov(prompt: str) -> str:
return resp_text
return _prov
for bad_resp in [
"Não tenho certeza sobre este documento.",
"{json_quebrado_sem_aspas: true",
"```json\n{invalido: 123}\n```",
]:
adapter = LLMFallbackAdapter(provider_fn=make_bad_provider(bad_resp))
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
content = "Menção isolada da TechCorp."
res = classifier.classify(ecp_tech_corp, content)
assert res.decision == DecisionCategory.TANGENTIAL
def test_llm_sad_path_missing_decision_key_in_json(ecp_tech_corp: ECPSnapshot):
"""Cenário 4.10: LLM retorna JSON válido mas sem o campo obrigatório 'decision'."""
adapter = LLMFallbackAdapter(
provider_fn=lambda p: json.dumps({"confidence": 0.90, "rationale": "Faltou a decisao"})
)
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
content = "Menção isolada da TechCorp."
res = classifier.classify(ecp_tech_corp, content)
assert res.decision == DecisionCategory.TANGENTIAL
def test_llm_sad_path_unknown_hallucinated_decision_enum(ecp_tech_corp: ECPSnapshot):
"""Cenário 4.11: LLM alucina uma categoria inexistente (ex: 'SUPER_INHERENT')."""
adapter = LLMFallbackAdapter(
provider_fn=lambda p: json.dumps({"decision": "SUPER_INHERENT", "confidence": 0.99})
)
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
content = "Menção isolada da TechCorp."
res = classifier.classify(ecp_tech_corp, content)
assert res.decision == DecisionCategory.TANGENTIAL
def test_llm_resilience_confidence_clipping(ecp_tech_corp: ECPSnapshot):
"""Cenário 4.12: LLM retorna confidence fora do intervalo [0.0, 1.0] -> clippa com segurança."""
def make_clipping_provider(c_val: float):
def _prov(prompt: str) -> str:
return json.dumps(
{
"decision": "DIRECT_INHERENT",
"confidence": c_val,
"rationale": "Teste de clipping.",
}
)
return _prov
for raw_conf, expected_conf in [(1.5, 1.0), (-0.5, 0.0), (0.85432, 0.8543)]:
adapter = LLMFallbackAdapter(provider_fn=make_clipping_provider(raw_conf))
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
res = classifier.classify(ecp_tech_corp, "Menção da TechCorp.")
assert res.confidence == expected_conf
def test_llm_optimization_clear_case_bypasses_llm(ecp_tech_corp: ECPSnapshot):
"""Cenário 4.13: Caso claro de alta densidade NÃO chama LLM mesmo com enable_llm=True."""
called = {"status": False}
def tracking_fn(p: str) -> str:
called["status"] = True
return json.dumps({"decision": "DIRECT_INHERENT"})
adapter = LLMFallbackAdapter(provider_fn=tracking_fn)
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
content = (
"# TechCorp Global anuncia nova inteligência artificial para computação em nuvem\n\n"
"A TechCorp Global ativou hoje novos clusters de datacenter com software avançado."
)
res = classifier.classify(ecp_tech_corp, content)
assert res.decision == DecisionCategory.DIRECT_INHERENT
assert called["status"] is False # LLM NÃO foi acionado
# ==============================================================================
# 5. Provedores Reais de LLM (OpenAI Mock e Gemini REST Mock)
# ==============================================================================
def test_llm_provider_openai_client_execution(ecp_tech_corp: ECPSnapshot):
"""Cenário 5.1: Simula execução bem-sucedida via cliente OpenAI SDK."""
mock_chat_completion = MagicMock()
mock_choice = MagicMock()
mock_choice.message.content = json.dumps(
{
"decision": "DIRECT_INHERENT",
"confidence": 0.96,
"rationale": "OpenAI validou o contexto corporativo com precisão.",
}
)
mock_chat_completion.choices = [mock_choice]
mock_openai_instance = MagicMock()
mock_openai_instance.chat.completions.create.return_value = mock_chat_completion
with patch("openai.OpenAI", return_value=mock_openai_instance):
adapter = LLMFallbackAdapter(api_key="sk-mock-openai-key")
initial_res = ClassificationResult(
decision=DecisionCategory.TANGENTIAL,
is_inherent=False,
confidence=0.40,
detected_language="pt",
matched_anchors=[],
negative_matches=[],
graph_matches=[],
evidence=[],
rationale="Passing.",
warnings=[],
)
res = adapter.disambiguate(ecp_tech_corp, "Artigo sobre TechCorp.", initial_res)
assert res is not None
assert res.decision == DecisionCategory.DIRECT_INHERENT
assert res.confidence == 0.96
def test_llm_provider_gemini_rest_execution(ecp_tech_corp: ECPSnapshot):
"""Cenário 5.2: Simula execução bem-sucedida via API REST do Google Gemini."""
gemini_payload = {
"candidates": [
{
"content": {
"parts": [
{
"text": json.dumps(
{
"decision": "DIRECT_INHERENT",
"confidence": 0.98,
"rationale": "Gemini 2.5 Flash confirmou aderência direta ao tópico.",
}
)
}
]
}
}
]
}
mock_response = MagicMock()
mock_response.read.return_value = json.dumps(gemini_payload).encode("utf-8")
mock_response.__enter__.return_value = mock_response
with patch("urllib.request.urlopen", return_value=mock_response):
with patch.dict("os.environ", {"GEMINI_API_KEY": "mock-gemini-key"}):
adapter = LLMFallbackAdapter(api_key="")
initial_res = ClassificationResult(
decision=DecisionCategory.TANGENTIAL,
is_inherent=False,
confidence=0.40,
detected_language="pt",
matched_anchors=[],
negative_matches=[],
graph_matches=[],
evidence=[],
rationale="Passing.",
warnings=[],
)
res = adapter.disambiguate(ecp_tech_corp, "Artigo sobre TechCorp.", initial_res)
assert res is not None
assert res.decision == DecisionCategory.DIRECT_INHERENT
assert res.confidence == 0.98
# ==============================================================================
# 6. Suíte de Contrato e Erros da CLI classify.py
# ==============================================================================
def test_cli_error_ecp_file_does_not_exist(tmp_path: Path, capsys):
"""Cenário 6.1: Caminho de ECP inexistente -> Exit Code 1, error_code: invalid_ecp_json."""
content_file = tmp_path / "valid.md"
content_file.write_text("# Conteúdo válido", encoding="utf-8")
exit_code = main(["--ecp", str(tmp_path / "nao_existe.json"), "--content", str(content_file)])
assert exit_code == 1
captured = capsys.readouterr()
err_json = json.loads(captured.err)
assert err_json["error_code"] == "invalid_ecp_json"
def test_cli_error_ecp_corrupted_json_syntax(tmp_path: Path, capsys):
"""Cenário 6.2: Arquivo ECP com sintaxe JSON corrompida."""
bad_ecp = tmp_path / "corrupt.json"
bad_ecp.write_text("{ target_name: 'sem_aspas' ", encoding="utf-8")
content_file = tmp_path / "valid.md"
content_file.write_text("# Conteúdo válido", encoding="utf-8")
exit_code = main(["--ecp", str(bad_ecp), "--content", str(content_file)])
assert exit_code == 1
captured = capsys.readouterr()
err_json = json.loads(captured.err)
assert err_json["error_code"] == "invalid_ecp_json"
def test_cli_error_ecp_missing_each_required_field(tmp_path: Path, capsys):
"""Cenário 6.3: Valida erro para falta de cada um dos campos obrigatórios do ECP."""
required_fields = ["target_entity_id", "target_name", "aliases", "domain", "anchors"]
base_ecp = {
"target_entity_id": "ent_1",
"target_name": "Nome",
"aliases": ["Alias"],
"domain": "Domínio",
"anchors": ["Âncora"],
}
content_file = tmp_path / "valid.md"
content_file.write_text("# Conteúdo válido", encoding="utf-8")
for field in required_fields:
bad_data = base_ecp.copy()
del bad_data[field]
bad_file = tmp_path / f"missing_{field}.json"
bad_file.write_text(json.dumps(bad_data), encoding="utf-8")
exit_code = main(["--ecp", str(bad_file), "--content", str(content_file)])
assert exit_code == 1
captured = capsys.readouterr()
err_json = json.loads(captured.err)
assert err_json["error_code"] == "missing_required_field"
assert field in err_json["message"]
def test_cli_error_content_file_does_not_exist(tmp_path: Path, capsys):
"""Cenário 6.4: Caminho de arquivo Markdown inexistente."""
ecp_file = tmp_path / "ecp.json"
ecp_file.write_text(
json.dumps(
{
"target_entity_id": "ent_1",
"target_name": "TechCorp",
"aliases": ["TechCorp"],
"domain": "Tech",
"anchors": ["cloud"],
}
),
encoding="utf-8",
)
exit_code = main(["--ecp", str(ecp_file), "--content", str(tmp_path / "doc_fantasma.md")])
assert exit_code == 1
captured = capsys.readouterr()
err_json = json.loads(captured.err)
assert err_json["error_code"] == "invalid_markdown"
def test_cli_error_empty_and_whitespace_content(tmp_path: Path, capsys):
"""Cenário 6.5: Arquivo Markdown vazio ou contendo apenas espaços em branco."""
ecp_file = tmp_path / "ecp.json"
ecp_file.write_text(
json.dumps(
{
"target_entity_id": "ent_1",
"target_name": "TechCorp",
"aliases": ["TechCorp"],
"domain": "Tech",
"anchors": ["cloud"],
}
),
encoding="utf-8",
)
for empty_text in ["", " \n\n\t \n "]:
empty_file = tmp_path / "empty.md"
empty_file.write_text(empty_text, encoding="utf-8")
exit_code = main(["--ecp", str(ecp_file), "--content", str(empty_file)])
assert exit_code == 1
captured = capsys.readouterr()
err_json = json.loads(captured.err)
assert err_json["error_code"] == "empty_content"
def test_cli_output_file_creates_nested_directories(tmp_path: Path):
"""Cenário 6.6: A flag -o / --output cria diretórios aninhados automaticamente."""
ecp_file = tmp_path / "ecp.json"
ecp_file.write_text(
json.dumps(
{
"target_entity_id": "ent_1",
"target_name": "TechCorp",
"aliases": ["TechCorp"],
"domain": "Tech",
"anchors": ["cloud"],
}
),
encoding="utf-8",
)
content_file = tmp_path / "content.md"
content_file.write_text("# TechCorp\n\nTechCorp cloud computing.", encoding="utf-8")
nested_out = tmp_path / "deep" / "nested" / "folder" / "resultado.json"
exit_code = main(
["--ecp", str(ecp_file), "--content", str(content_file), "-o", str(nested_out)]
)
assert exit_code == 0
assert nested_out.exists()
payload = json.loads(nested_out.read_text(encoding="utf-8"))
assert payload["decision"] == "DIRECT_INHERENT"
# ==============================================================================
# 7. Matriz Multilíngue Completa (6 Idiomas)
# ==============================================================================
@pytest.mark.parametrize(
"lang,target,aliases,domain,anchors,content,expected_decision,expected_lang",
[
# Português
(
"pt",
"Petrobras",
["Petrobras"],
"Energia",
["pré-sal", "petróleo", "refinaria"],
"# Petrobras bate recorde de produção no pré-sal com novas plataformas.",
DecisionCategory.DIRECT_INHERENT,
"pt",
),
# Inglês
(
"en",
"Apple Inc.",
["Apple", "Apple Inc."],
"Technology",
["iPhone", "MacBook", "iOS", "silicon"],
"# Apple unveils new MacBook Pro with M4 silicon and advanced iOS features.",
DecisionCategory.DIRECT_INHERENT,
"en",
),
# Espanhol
(
"es",
"River Plate",
["River Plate", "River"],
"Fútbol",
["Monumental", "Libertadores", "Sudamericana"],
"# River Plate se prepara para disputar el torneo continental en el Estadio Monumental.",
DecisionCategory.DIRECT_INHERENT,
"es",
),
# Alemão (Compostos e Diacríticos)
(
"de",
"Volkswagen AG",
["Volkswagen", "VW"],
"Automobilindustrie",
["Elektroauto", "Batteriefabrik", "Produktion"],
"# Volkswagen investiert Milliarden in eine neue Batteriefabrik für Elektroautos in Deutschland.",
DecisionCategory.DIRECT_INHERENT,
"de",
),
# Italiano
(
"it",
"Scuderia Ferrari",
["Ferrari", "Scuderia Ferrari"],
"Automobilismo",
["Monza", "Gran Premio", "motore", "pole position"],
"# La Ferrari conquista una straordinaria pole position nel Gran Premio di Monza.",
DecisionCategory.DIRECT_INHERENT,
"it",
),
# Francês (Elisão e Apóstrofos)
(
"fr",
"TotalEnergies",
["TotalEnergies", "Total"],
"Énergie",
["énergie solaire", "pétrole", "renouvelable", "électricité"],
"# L'entreprise TotalEnergies accélère ses investissements dans l'énergie solaire et l'électricité en France.",
DecisionCategory.DIRECT_INHERENT,
"fr",
),
],
)
def test_multilingual_matrix_6_languages(
lang: str,
target: str,
aliases: list[str],
domain: str,
anchors: list[str],
content: str,
expected_decision: DecisionCategory,
expected_lang: str,
):
"""Garante a precisão e robustez do classificador nos 6 idiomas suportados pela POC."""
ecp = ECPSnapshot(
target_entity_id=f"ent_{lang}",
target_name=target,
aliases=aliases,
domain=domain,
anchors=anchors,
)
classifier = InherenceClassifier()
res = classifier.classify(ecp, content)
assert res.decision == expected_decision
assert res.is_inherent is True
assert res.detected_language == expected_lang
assert res.confidence >= 0.85