feat(classifier): implement Tier 3 LLM fallback adapter and boundary disambiguation test suite

This commit is contained in:
2026-08-21 10:50:57 -03:00
parent cb33dafac1
commit 31152d5031
25 changed files with 2997 additions and 1404 deletions
+9 -5
View File
@@ -118,7 +118,7 @@ flowchart TD
* **Tier 1 (Determinístico / NLP Leve)**: Análise de frequência de termos, detecção de âncoras temáticas no primeiro terço do documento, contagem de aliases e penalização por âncoras negativas. * **Tier 1 (Determinístico / NLP Leve)**: Análise de frequência de termos, detecção de âncoras temáticas no primeiro terço do documento, contagem de aliases e penalização por âncoras negativas.
* **Tier 2 (Vetorial / Embeddings)** *(Opcional: `--enable-embeddings`)*: Projeção vetorial e cálculo de cosseno entre o perfil da entidade e os parágrafos do documento. * **Tier 2 (Vetorial / Embeddings)** *(Opcional: `--enable-embeddings`)*: Projeção vetorial e cálculo de cosseno entre o perfil da entidade e os parágrafos do documento.
* **Tier 3 (LLM Fallback)** *(Opcional: `--enable-llm`)*: Consulta a modelo de linguagem para desambiguação de casos limiares e sutilezas semânticas. * **Tier 3 (LLM Fallback)** *(Opcional: `--enable-llm`)*: Adaptador de desambiguação inteligente (`src/adapters/llm.py`) acionado exclusivamente para casos limiares e ambíguos (ex: menção isolada `TANGENTIAL` ou confiança `< 0.60`). Casos claros não chamam o LLM para economizar custos e latência; em caso de falha de conexão com a API, degrada graciosamente mantendo o resultado do Tier 1 com aviso registrado em `warnings`.
### Categorias de Decisão ### Categorias de Decisão
@@ -499,7 +499,8 @@ TextNLPClassifierApp/
│ ├── test_extract_google_news.py │ ├── test_extract_google_news.py
│ ├── test_extract_article_contents.py │ ├── test_extract_article_contents.py
│ ├── test_select_article_extractor.py │ ├── test_select_article_extractor.py
│ └── test_convert_article_to_markdown.py # Testes da conversão para Markdown │ ├── test_convert_article_to_markdown.py # Testes da conversão para Markdown
│ └── test_llm_fallback.py # Testes do Tier 3 LLM Fallback
├── requirements.txt # Dependências do projeto ├── requirements.txt # Dependências do projeto
├── pyproject.toml # Configurações de ferramentas (pytest, ruff, mypy) ├── pyproject.toml # Configurações de ferramentas (pytest, ruff, mypy)
└── README.md # Documentação principal └── README.md # Documentação principal
@@ -509,12 +510,15 @@ TextNLPClassifierApp/
## 🧪 Testes e Qualidade de Código ## 🧪 Testes e Qualidade de Código
O repositório possui **187 testes automatizados** com 100% de aprovação cobrindo testes unitários, de regressão, de integração, Golden Fixtures exatas, testes de sensibilidade de mutação e testes End-to-End (E2E) via CLI subprocess: O repositório possui **196 testes automatizados** com 100% de aprovação cobrindo testes unitários, de regressão, de integração, Golden Fixtures exatas, testes de sensibilidade de mutação, testes de fallback para LLM (Tier 3) e testes End-to-End (E2E) via CLI subprocess:
```bash ```bash
# Executar toda a suíte de testes do projeto (187 testes) # Executar toda a suíte de testes do projeto (196 testes)
pytest -v pytest -v
# Executar os testes do Fallback para LLM (Tier 3)
pytest tests/test_llm_fallback.py -v
# Executar os testes de Conversão de Artigo para Markdown (67 testes) # Executar os testes de Conversão de Artigo para Markdown (67 testes)
pytest tests/test_convert_article_to_markdown.py -v pytest tests/test_convert_article_to_markdown.py -v
@@ -531,7 +535,7 @@ pytest tests/test_extract_google_news.py -v
ruff check . ruff check .
# Verificação estática de tipos com Mypy # Verificação estática de tipos com Mypy
mypy scripts/convert_article_to_markdown.py mypy src/ scripts/ tests/
``` ```
--- ---
+1 -1
View File
@@ -98,7 +98,7 @@ def main(argv: list[str] | None = None) -> int:
try: try:
args = parse_args(argv) args = parse_args(argv)
except SystemExit as e: except SystemExit as e:
return int(e.code) return int(e.code) if isinstance(e.code, int) else 2
ecp_path = Path(args.ecp) ecp_path = Path(args.ecp)
content_path = Path(args.content) content_path = Path(args.content)
+6 -3
View File
@@ -45,7 +45,7 @@
"43": "2. Basic CLI Usage Examples", "43": "2. Basic CLI Usage Examples",
"44": "2. Standard Streams & Exit Codes", "44": "2. Standard Streams & Exit Codes",
"45": "ClassificationResult", "45": "ClassificationResult",
"46": "InherenceClassifier", "46": "test_adversarial.py",
"47": "classifier.py", "47": "classifier.py",
"48": "test_convert_article_to_markdown.py", "48": "test_convert_article_to_markdown.py",
"49": "content_northvolt_de.md", "49": "content_northvolt_de.md",
@@ -151,7 +151,7 @@
"149": "get_hl_gl_ceid", "149": "get_hl_gl_ceid",
"150": "13. Estratégia de testes", "150": "13. Estratégia de testes",
"151": "6. Contrato de entrada", "151": "6. Contrato de entrada",
"152": "assemble_markdown_document", "152": "InherenceClassifier",
"153": "convert_html_to_markdown", "153": "convert_html_to_markdown",
"154": "JSON Schema Contract: Deterministic Article Content Selection", "154": "JSON Schema Contract: Deterministic Article Content Selection",
"155": "5. Escopo", "155": "5. Escopo",
@@ -163,5 +163,8 @@
"161": "test_normalize_list_deduplication_preserves_case_and_order", "161": "test_normalize_list_deduplication_preserves_case_and_order",
"162": "test_normalize_date_iso_8601_variants", "162": "test_normalize_date_iso_8601_variants",
"163": "test_metadata_priority_original_url_all_fallbacks", "163": "test_metadata_priority_original_url_all_fallbacks",
"164": "test_normalize_scalar_non_string_types" "164": "test_normalize_scalar_non_string_types",
"165": "LLMFallbackAdapter",
"166": "remove_duplicate_initial_h1",
"167": "test_normalize_scalar_whitespace_collapsing"
} }
+1 -1
View File
@@ -1 +1 @@
{"0": "36bdb6f09c457f7c", "1": "8c5bf6244cf710c6", "2": "efbcc9c62a3ee78b", "3": "8599153989b07faa", "4": "b5952a1f7fee9f20", "5": "5b8462a3f82d188c", "6": "80f79e9e2011a3e3", "7": "4654167fd211d027", "8": "50acfa00fe353440", "9": "c6d2f770737823f1", "10": "44f2ca451aea24be", "11": "feaac5ab67a8c17a", "12": "b71bd92e5edbf2e0", "13": "219d65ba6d2689e4", "14": "8e30bb8112fd02d1", "15": "03906ab80b99db85", "16": "5d51c60ba1bc2be0", "17": "a1da914f522dcd21", "18": "fbad840891b90569", "19": "0686ff2d6fe29fb3", "20": "060baa9e1924b465", "21": "a5c8f2c3080b8243", "22": "0d76852f1d29eeb1", "23": "6ff68619f2d72924", "24": "3da11675eee7ec46", "25": "a6696589e9556f97", "26": "6c752999e8a4d4b6", "27": "2d4e13ea2111d750", "28": "4b60cb0ee1ac186a", "29": "f56fbca9bb8235ec", "30": "c7beed940704509f", "31": "38be2d254fb31ae8", "32": "ee5596fcf7e7c0b3", "33": "e4d4e0a440bc599f", "34": "c897e49c001acdae", "35": "3aad272a2cf5d495", "36": "0a197439d306b956", "37": "f43acf5c8b1329af", "38": "6775efafc9b33338", "39": "8176a164778526f9", "40": "66b69189c0acc3ff", "41": "0322ff824966a4d8", "42": "784c9e3d336a7f53", "43": "4b8bb6c3f7b64856", "44": "18c0ff3e6225bcb2", "45": "0e7fcc21c22f118e", "46": "58f3596265f5f902", "47": "4de4f30d96344790", "48": "41da3f4214d41862", "49": "0d0f9f015921feef", "50": "8d0c81e5ca23e9a6", "51": "f79963571b9c15ee", "52": "5935824c825606cb", "53": "9685f9cbe158e50b", "54": "3d5ab759f350bc79", "55": "d549f24931a990e9", "56": "3cc031dcb648797c", "57": "a0ab88e6c629251d", "58": "76bd6412e2a22ecd", "59": "54827845564490c9", "60": "0a9736c416c0c6b9", "61": "77358620ac528153", "62": "3b0c585df09df48a", "63": "7e78cd3b28828c20", "64": "1c0c958231735f61", "65": "60b0f81225f62f69", "66": "920754c65cc94b88", "67": "df911472140a9b94", "68": "8e17bc11bcea91b9", "69": "7e905b75e4f28b95", "70": "a28424eca5d36c55", "71": "2cdb53d5b6051ab6", "72": "e42fbd3dc744e730", "73": "7fe2cac980de160c", "74": "2b1343a6a9db1487", "75": "54a1bb232f1d4ceb", "76": "442ba11d31ec0e0a", "77": "852a25b8b95bf8d1", "78": "1810ab370b9cd608", "79": "0fc5dca02a3f02f6", "80": "6ff8a97e63c9a2f3", "81": "a38f84ae3d895236", "82": "dc6ddc157a3b9efb", "83": "a05140495d7a0353", "84": "24ca89fec34df075", "85": "f8bfd0cfe9e8b478", "86": "410d15a346bd5894", "87": "6b41d288cfd834ab", "88": "5aa6db96312a8811", "89": "80225792bb62ba04", "90": "e18a0a239fe528ba", "91": "d4579c5b7aa2742a", "92": "7b9ba7c3bff11361", "93": "71cd9c1fa4a857f0", "94": "34cd980be3c32d21", "95": "970093453f3b7d90", "96": "9e96780a2b7c4bd6", "97": "b7c10b0e09caac0b", "98": "089ea6a55861c693", "99": "b4fe3d520c1fee9a", "100": "8c97d8c400895e15", "101": "55faed4f78d00dd4", "102": "6aa00d5a83295f11", "103": "f58668f5b10ccdeb", "104": "4ec787414cc6f50b", "105": "1cf3077fd45d874a", "106": "edcd5d9bb3c4b00f", "107": "37f2f47110fe3eaa", "108": "b7ad5abb1da8cf8d", "109": "cb48a9c4f54efa38", "110": "f6dd36fd7f3edbe5", "111": "2925b620f0b1fd17", "112": "d8b3099917c3b711", "113": "3bb61caa0302c804", "114": "0d4f1d08dd056bb9", "115": "4ac2dcddeec2ff11", "116": "07da9aae9668f573", "117": "196f63e0c4536d30", "118": "ade84262e3cfac12", "119": "ebe4e5e0c42c613f", "120": "27256931b19a2867", "121": "5396e68ca6c185ad", "122": "aa8a1de55696b666", "123": "96618c9a362af46c", "124": "83f104cbb62fd03e", "125": "6db738fb27190349", "126": "6a087a22cbcef972", "127": "85fd71a0cad8d3a5", "128": "22dd4feed96c4229", "129": "c4d2f60f532e6f16", "130": "f6b0aa8a1568926b", "131": "633a2029dd2fc8e8", "132": "67ea4284cbc02c54", "133": "d899cfc86c7a4a27", "134": "ba9464410a9b4168", "135": "0b1a09562254038e", "136": "521f5c7b9d566b4d", "137": "9e37828bdd2ba8c5", "138": "ec03c97194c56f91", "139": "f4e6d5dfa30034c5", "140": "94280631e0d38796", "141": "1e0330b8757f333e", "142": "f35d75e1194c008d", "143": "4d2ae7190b514a34", "144": "4a98716cabf43f86", "145": "0270e11190087cea", "146": "edc785fd71bb0675", "147": "4c7347f8f86e1fbd", "148": "8ca77cc4fd6fd437", "149": "09850697b717469a", "150": "f4f4ce1a1180ddb1", "151": "e426746f6e9ee15f", "152": "3f9db1282dfb2a7e", "153": "fc1af7f79936134a", "154": "56747bad6345d66b", "155": "a8e7498fa7e257df", "156": "56e7b2355898077f", "157": "f2fc88f7d8214711", "158": "c966f6f8570c8c29", "159": "dd0e66a92d9d67a8", "160": "bf44eb423b3a9ac2", "161": "c8ca209168c5125d", "162": "ec7a7324fb6c3680", "163": "6c6e3486c0ffb834", "164": "4bc6f88cad2f1d07"} {"0": "36bdb6f09c457f7c", "1": "8c5bf6244cf710c6", "2": "efbcc9c62a3ee78b", "3": "8599153989b07faa", "4": "b5952a1f7fee9f20", "5": "5b8462a3f82d188c", "6": "80f79e9e2011a3e3", "7": "4654167fd211d027", "8": "50acfa00fe353440", "9": "c6d2f770737823f1", "10": "44f2ca451aea24be", "11": "feaac5ab67a8c17a", "12": "b71bd92e5edbf2e0", "13": "219d65ba6d2689e4", "14": "8e30bb8112fd02d1", "15": "03906ab80b99db85", "16": "5d51c60ba1bc2be0", "17": "a1da914f522dcd21", "18": "fbad840891b90569", "19": "0686ff2d6fe29fb3", "20": "060baa9e1924b465", "21": "a5c8f2c3080b8243", "22": "0d76852f1d29eeb1", "23": "6ff68619f2d72924", "24": "3da11675eee7ec46", "25": "a6696589e9556f97", "26": "6c752999e8a4d4b6", "27": "2d4e13ea2111d750", "28": "4b60cb0ee1ac186a", "29": "f56fbca9bb8235ec", "30": "c7beed940704509f", "31": "38be2d254fb31ae8", "32": "ee5596fcf7e7c0b3", "33": "e4d4e0a440bc599f", "34": "c897e49c001acdae", "35": "3aad272a2cf5d495", "36": "0a197439d306b956", "37": "f43acf5c8b1329af", "38": "6775efafc9b33338", "39": "8176a164778526f9", "40": "66b69189c0acc3ff", "41": "0322ff824966a4d8", "42": "784c9e3d336a7f53", "43": "4b8bb6c3f7b64856", "44": "18c0ff3e6225bcb2", "45": "b18defdea7d53e37", "46": "54c0fceb01591230", "47": "0fad42a4989aa7e3", "48": "0237e1e02ee47a27", "49": "0d0f9f015921feef", "50": "8d0c81e5ca23e9a6", "51": "f79963571b9c15ee", "52": "5935824c825606cb", "53": "9685f9cbe158e50b", "54": "3d5ab759f350bc79", "55": "d549f24931a990e9", "56": "3cc031dcb648797c", "57": "a0ab88e6c629251d", "58": "76bd6412e2a22ecd", "59": "54827845564490c9", "60": "0a9736c416c0c6b9", "61": "77358620ac528153", "62": "3b0c585df09df48a", "63": "7e78cd3b28828c20", "64": "1c0c958231735f61", "65": "60b0f81225f62f69", "66": "920754c65cc94b88", "67": "df911472140a9b94", "68": "8e17bc11bcea91b9", "69": "7e905b75e4f28b95", "70": "a28424eca5d36c55", "71": "2cdb53d5b6051ab6", "72": "e42fbd3dc744e730", "73": "7fe2cac980de160c", "74": "2b1343a6a9db1487", "75": "54a1bb232f1d4ceb", "76": "442ba11d31ec0e0a", "77": "852a25b8b95bf8d1", "78": "1810ab370b9cd608", "79": "0fc5dca02a3f02f6", "80": "6ff8a97e63c9a2f3", "81": "a38f84ae3d895236", "82": "dc6ddc157a3b9efb", "83": "a05140495d7a0353", "84": "24ca89fec34df075", "85": "f8bfd0cfe9e8b478", "86": "410d15a346bd5894", "87": "6b41d288cfd834ab", "88": "5aa6db96312a8811", "89": "80225792bb62ba04", "90": "e18a0a239fe528ba", "91": "d4579c5b7aa2742a", "92": "7b9ba7c3bff11361", "93": "71cd9c1fa4a857f0", "94": "34cd980be3c32d21", "95": "970093453f3b7d90", "96": "9e96780a2b7c4bd6", "97": "b7c10b0e09caac0b", "98": "089ea6a55861c693", "99": "cb6165a7dc822d29", "100": "8c97d8c400895e15", "101": "bd22248d0516e5d4", "102": "6aa00d5a83295f11", "103": "f58668f5b10ccdeb", "104": "4ec787414cc6f50b", "105": "1cf3077fd45d874a", "106": "edcd5d9bb3c4b00f", "107": "37f2f47110fe3eaa", "108": "b7ad5abb1da8cf8d", "109": "cb48a9c4f54efa38", "110": "f6dd36fd7f3edbe5", "111": "2925b620f0b1fd17", "112": "d8b3099917c3b711", "113": "3bb61caa0302c804", "114": "0d4f1d08dd056bb9", "115": "4ac2dcddeec2ff11", "116": "07da9aae9668f573", "117": "196f63e0c4536d30", "118": "ade84262e3cfac12", "119": "ebe4e5e0c42c613f", "120": "27256931b19a2867", "121": "5396e68ca6c185ad", "122": "aa8a1de55696b666", "123": "96618c9a362af46c", "124": "83f104cbb62fd03e", "125": "6db738fb27190349", "126": "6a087a22cbcef972", "127": "85fd71a0cad8d3a5", "128": "22dd4feed96c4229", "129": "c4d2f60f532e6f16", "130": "f6b0aa8a1568926b", "131": "142d0db70bad18fe", "132": "67ea4284cbc02c54", "133": "d899cfc86c7a4a27", "134": "ba9464410a9b4168", "135": "0d496a12149eca27", "136": "521f5c7b9d566b4d", "137": "9e37828bdd2ba8c5", "138": "ec03c97194c56f91", "139": "f4e6d5dfa30034c5", "140": "d9b47fa423cf0748", "141": "1e0330b8757f333e", "142": "f35d75e1194c008d", "143": "4d2ae7190b514a34", "144": "4a98716cabf43f86", "145": "56b2431193739c38", "146": "edc785fd71bb0675", "147": "4c7347f8f86e1fbd", "148": "8ca77cc4fd6fd437", "149": "09850697b717469a", "150": "f4f4ce1a1180ddb1", "151": "e426746f6e9ee15f", "152": "756d0fd1d69866c1", "153": "a3593e6f45bafb20", "154": "56747bad6345d66b", "155": "a8e7498fa7e257df", "156": "56e7b2355898077f", "157": "f2fc88f7d8214711", "158": "c966f6f8570c8c29", "159": "5f6094aa385f3bfe", "160": "2834e7d59672e756", "161": "cc6e436d94fd0033", "162": "64f33a2fc8969cd2", "163": "26ac1c0a00eabce1", "164": "cdcea44a6805ae55", "165": "5faccba309eb478d", "166": "85c97dab928b9b1b", "167": "7ab5695391e32126"}
@@ -158,10 +158,10 @@
"156": "Los puntajes de River vs. Independiente Santa Fe, por la Copa Sudamericana - TyC Sports", "156": "Los puntajes de River vs. Independiente Santa Fe, por la Copa Sudamericana - TyC Sports",
"157": "valid_newspaper4k.md", "157": "valid_newspaper4k.md",
"158": "valid_readability.md", "158": "valid_readability.md",
"159": "test_normalize_date_rfc_2822_variants", "159": "test_normalize_list_author_url_filtering",
"160": "test_normalize_date_invalid_and_placeholders", "160": "test_normalize_date_invalid_and_placeholders",
"161": "test_metadata_priority_title_all_fallbacks", "161": "test_normalize_list_deduplication_preserves_case_and_order",
"162": "test_metadata_priority_subtitle_omitted_when_equal_to_title", "162": "test_normalize_date_iso_8601_variants",
"163": "test_metadata_priority_first_valid_source_no_cross_merging", "163": "test_metadata_priority_original_url_all_fallbacks",
"164": "test_normalize_scalar_non_string_types" "164": "test_normalize_scalar_non_string_types"
} }
+7 -7
View File
@@ -1,7 +1,7 @@
# Graph Report - TextNLPClassifierApp (2026-08-21) # Graph Report - TextNLPClassifierApp (2026-08-21)
## Corpus Check ## Corpus Check
- 199 files · ~108,365 words - 200 files · ~108,772 words
- Verdict: corpus is large enough that graph structure adds value. - Verdict: corpus is large enough that graph structure adds value.
## Summary ## Summary
@@ -10,7 +10,7 @@
- Token cost: 0 input · 0 output - Token cost: 0 input · 0 output
## Graph Freshness ## Graph Freshness
- Built from commit: `64dfd842` - Built from commit: `926a6b8c`
- Run `git rev-parse HEAD` and compare to check if the graph is stale. - Run `git rev-parse HEAD` and compare to check if the graph is stale.
- Run `graphify update .` after code changes (no API cost). - Run `graphify update .` after code changes (no API cost).
@@ -167,11 +167,11 @@
- Los puntajes de River vs. Independiente Santa Fe, por la Copa Sudamericana - TyC Sports - Los puntajes de River vs. Independiente Santa Fe, por la Copa Sudamericana - TyC Sports
- valid_newspaper4k.md - valid_newspaper4k.md
- valid_readability.md - valid_readability.md
- test_normalize_date_rfc_2822_variants - test_normalize_list_author_url_filtering
- test_normalize_date_invalid_and_placeholders - test_normalize_date_invalid_and_placeholders
- test_metadata_priority_title_all_fallbacks - test_normalize_list_deduplication_preserves_case_and_order
- test_metadata_priority_subtitle_omitted_when_equal_to_title - test_normalize_date_iso_8601_variants
- test_metadata_priority_first_valid_source_no_cross_merging - test_metadata_priority_original_url_all_fallbacks
- test_normalize_scalar_non_string_types - test_normalize_scalar_non_string_types
## God Nodes (most connected - your core abstractions) ## God Nodes (most connected - your core abstractions)
@@ -357,7 +357,7 @@ Nodes (16): count_phrase_occurrences(), match_phrase_in_text(), Core determinist
### Community 48 - "test_convert_article_to_markdown.py" ### Community 48 - "test_convert_article_to_markdown.py"
Cohesion: 0.07 Cohesion: 0.07
Nodes (27): Suíte de Testes Automatizados para Conversão de Artigo JSON para Markdown.…, Testa deduplicação case-insensitive preservando a grafia e ordem da primeira…, Valida parsing de datas ISO 8601 em múltiplos formatos e fusos., Valida a cadeia de fallback completa para a URL ORIGINAL (5 níveis)., Valida decodificação de entidades HTML nomeadas e numéricas., Valida colapso de tabs, quebras de linha e espaços múltiplos em um único espaço., Garante correspondência exata byte a byte para Trafilatura, Newspaper4k e…, Garante que múltiplas execuções no mesmo arquivo produzam hashes SHA-256… (+19 more) Nodes (27): Suíte de Testes Automatizados para Conversão de Artigo JSON para Markdown.…, Testa comportamento com listas vazias, nulas ou contendo apenas placeholders., Valida parsing de datas no formato RFC 2822 (usado em feeds RSS e cabeçalhos…, Valida a cadeia de fallback completa para o campo TÍTULO (6 níveis)., Garante que subtítulo idêntico ao título seja automaticamente omitido (None)., Valida decodificação de entidades HTML nomeadas e numéricas., Garante que listas de autores/tags usem apenas a primeira fonte válida, sem…, Garante correspondência exata byte a byte para Trafilatura, Newspaper4k e… (+19 more)
### Community 80 - "test_extract_article_contents.py" ### Community 80 - "test_extract_article_contents.py"
Cohesion: 0.06 Cohesion: 0.06
File diff suppressed because it is too large Load Diff
+9 -9
View File
@@ -654,9 +654,9 @@
"semantic_hash": "" "semantic_hash": ""
}, },
"README.md": { "README.md": {
"mtime": 1787318406.7167659, "mtime": 1787319775.4204426,
"seen": 1787318418.3361757, "seen": 1787319817.9012172,
"ast_hash": "d007bb3f3e6ad04e5e63981899f662d6", "ast_hash": "f809a191cb40e8a0367c748b9c8d3e84",
"semantic_hash": "" "semantic_hash": ""
}, },
"scripts/extract_article_contents.py": { "scripts/extract_article_contents.py": {
@@ -816,15 +816,15 @@
"semantic_hash": "" "semantic_hash": ""
}, },
"scripts/convert_article_to_markdown.py": { "scripts/convert_article_to_markdown.py": {
"mtime": 1787318308.4358957, "mtime": 1787318905.723034,
"seen": 1787318418.3304515, "seen": 1787319817.895658,
"ast_hash": "ae22787219cc382a19894270753d83d1", "ast_hash": "b58fbd8b426de464df2c269c32831583",
"semantic_hash": "" "semantic_hash": ""
}, },
"tests/test_convert_article_to_markdown.py": { "tests/test_convert_article_to_markdown.py": {
"mtime": 1787318233.6574264, "mtime": 1787318905.723034,
"seen": 1787318418.3321846, "seen": 1787319817.8974621,
"ast_hash": "ebc2d3e6b40728ac95f2330c692b9276", "ast_hash": "95c041594afec14ed25bc237b7ff8b89",
"semantic_hash": "" "semantic_hash": ""
}, },
"docs/prd_convert_json_markdown.md": { "docs/prd_convert_json_markdown.md": {
+56 -45
View File
@@ -1,16 +1,16 @@
# Graph Report - TextNLPClassifierApp (2026-08-21) # Graph Report - TextNLPClassifierApp (2026-08-21)
## Corpus Check ## Corpus Check
- 200 files · ~108,772 words - 201 files · ~110,162 words
- Verdict: corpus is large enough that graph structure adds value. - Verdict: corpus is large enough that graph structure adds value.
## Summary ## Summary
- 1527 nodes · 1899 edges · 165 communities (118 shown, 47 thin omitted) - 1554 nodes · 1970 edges · 168 communities (120 shown, 48 thin omitted)
- Extraction: 97% EXTRACTED · 3% INFERRED · 0% AMBIGUOUS · INFERRED: 51 edges (avg confidence: 0.95) - Extraction: 97% EXTRACTED · 3% INFERRED · 0% AMBIGUOUS · INFERRED: 59 edges (avg confidence: 0.95)
- Token cost: 0 input · 0 output - Token cost: 0 input · 0 output
## Graph Freshness ## Graph Freshness
- Built from commit: `926a6b8c` - Built from commit: `cb33dafa`
- Run `git rev-parse HEAD` and compare to check if the graph is stale. - Run `git rev-parse HEAD` and compare to check if the graph is stale.
- Run `graphify update .` after code changes (no API cost). - Run `graphify update .` after code changes (no API cost).
@@ -57,7 +57,7 @@
- 2. Basic CLI Usage Examples - 2. Basic CLI Usage Examples
- 2. Standard Streams & Exit Codes - 2. Standard Streams & Exit Codes
- ClassificationResult - ClassificationResult
- InherenceClassifier - test_adversarial.py
- classifier.py - classifier.py
- test_convert_article_to_markdown.py - test_convert_article_to_markdown.py
- content_northvolt_de.md - content_northvolt_de.md
@@ -160,7 +160,7 @@
- get_hl_gl_ceid - get_hl_gl_ceid
- 13. Estratégia de testes - 13. Estratégia de testes
- 6. Contrato de entrada - 6. Contrato de entrada
- assemble_markdown_document - InherenceClassifier
- convert_html_to_markdown - convert_html_to_markdown
- JSON Schema Contract: Deterministic Article Content Selection - JSON Schema Contract: Deterministic Article Content Selection
- 5. Escopo - 5. Escopo
@@ -173,35 +173,38 @@
- test_normalize_date_iso_8601_variants - test_normalize_date_iso_8601_variants
- test_metadata_priority_original_url_all_fallbacks - test_metadata_priority_original_url_all_fallbacks
- test_normalize_scalar_non_string_types - test_normalize_scalar_non_string_types
- LLMFallbackAdapter
- remove_duplicate_initial_h1
- test_normalize_scalar_whitespace_collapsing
## God Nodes (most connected - your core abstractions) ## God Nodes (most connected - your core abstractions)
1. `ECPSnapshot` - 31 edges 1. `ECPSnapshot` - 40 edges
2. `InherenceClassifier` - 25 edges 2. `InherenceClassifier` - 29 edges
3. `select_article_extractor()` - 23 edges 3. `DecisionCategory` - 28 edges
4. `ExtractorName` - 21 edges 4. `LLMFallbackAdapter` - 26 edges
5. `DecisionCategory` - 17 edges 5. `ClassificationResult` - 24 edges
6. `ClassificationResult` - 17 edges 6. `select_article_extractor()` - 23 edges
7. `PRD — Conversão de artigo JSON para Markdown` - 16 edges 7. `ExtractorName` - 21 edges
8. `process_batch()` - 15 edges 8. `PRD — Conversão de artigo JSON para Markdown` - 16 edges
9. `8. Regras funcionais` - 15 edges 9. `process_batch()` - 15 edges
10. `resolve_article_metadata()` - 14 edges 10. `8. Regras funcionais` - 15 edges
## Surprising Connections (you probably didn't know these) ## Surprising Connections (you probably didn't know these)
- `main()` --uses--> `ECPSnapshot` [INFERRED] - `main()` --uses--> `ECPSnapshot` [INFERRED]
classify.py → src/models.py classify.py → src/models.py
- `test_e2e_extract_google_news_live_pipeline()` --uses--> `ExtractionResult` [INFERRED] - `test_e2e_extract_google_news_live_pipeline()` --uses--> `ExtractionResult` [INFERRED]
tests/test_extract_google_news.py → scripts/extract_google_news.py tests/test_extract_google_news.py → scripts/extract_google_news.py
- `test_llm_adapter_interface()` --calls--> `LLMFallbackAdapter` [EXTRACTED]
tests/test_adapters.py → src/adapters/llm.py
- `classifier()` --uses--> `InherenceClassifier` [INFERRED] - `classifier()` --uses--> `InherenceClassifier` [INFERRED]
tests/test_benchmark_24.py → src/classifier.py tests/test_benchmark_24.py → src/classifier.py
- `test_classification_result_serialization()` --uses--> `DecisionCategory` [INFERRED] - `test_adversarial_apple_fruit_recipe()` --uses--> `DecisionCategory` [INFERRED]
tests/test_models.py → src/models.py tests/test_adversarial.py → src/models.py
- `petrobras_ecp()` --uses--> `ECPSnapshot` [INFERRED]
tests/test_classifier.py → src/models.py
## Import Cycles ## Import Cycles
- None detected. - None detected.
## Communities (165 total, 47 thin omitted) ## Communities (168 total, 48 thin omitted)
### Community 0 - "Task Planning" ### Community 0 - "Task Planning"
Cohesion: 0.07 Cohesion: 0.07
@@ -344,20 +347,20 @@ Cohesion: 0.29
Nodes (6): 1.1 Arguments & Options, 1. Command Line Interface, 2.1 Exit Codes, 2.2 Standard Output (`stdout`) / Standard Error (`stderr`), 2. Standard Streams & Exit Codes, CLI Contract & Interface Specification (POC) Nodes (6): 1.1 Arguments & Options, 1. Command Line Interface, 2.1 Exit Codes, 2.2 Standard Output (`stdout`) / Standard Error (`stderr`), 2. Standard Streams & Exit Codes, CLI Contract & Interface Specification (POC)
### Community 45 - "ClassificationResult" ### Community 45 - "ClassificationResult"
Cohesion: 0.09 Cohesion: 0.11
Nodes (19): ABC, BaseNLPAdapter, Base abstract adapter interface for optional Tier 2 / Tier 3 NLP enhancers., Abstract interface for pluggable NLP classification adapters., Return True if the underlying provider or model is installed and configured., Compute semantic similarity score between text and a set of candidate terms., Optionally refine an ambiguous classification result., LocalEmbeddingsAdapter (+11 more) Nodes (17): ABC, BaseNLPAdapter, Base abstract adapter interface for optional Tier 2 / Tier 3 NLP enhancers., Abstract interface for pluggable NLP classification adapters., Return True if the underlying provider or model is installed and configured., Compute semantic similarity score between text and a set of candidate terms., Optionally refine an ambiguous classification result., LocalEmbeddingsAdapter (+9 more)
### Community 46 - "InherenceClassifier" ### Community 46 - "test_adversarial.py"
Cohesion: 0.12 Cohesion: 0.10
Nodes (27): InherenceClassifier, Tier 1 Deterministic NLP Entity Inherence Classifier., DecisionCategory, RelatedEntity, Adversarial and robustness test suite for Multilingual NLP Entity Inherence…, Run CLI via subprocess without --output and verify stdout is pure parseable…, Run CLI via subprocess with empty content and verify error code and exit code., Content about city/state governance of São Paulo against ECP for São Paulo FC. (+19 more) Nodes (19): Any, RelatedEntity, Adversarial and robustness test suite for Multilingual NLP Entity Inherence…, Run CLI via subprocess without --output and verify stdout is pure parseable…, Run CLI via subprocess with empty content and verify error code and exit code., Content about city/state governance of São Paulo against ECP for São Paulo FC., Run CLI via subprocess with missing target_name and verify error payload., Run CLI via subprocess with corrupted JSON and verify error payload. (+11 more)
### Community 47 - "classifier.py" ### Community 47 - "classifier.py"
Cohesion: 0.16 Cohesion: 0.16
Nodes (16): count_phrase_occurrences(), match_phrase_in_text(), Core deterministic classification engine (Tier 1 core)., Check if a normalized phrase appears in normalized text with word boundary…, Count occurrences of a phrase in text., Classify inherence of content against an ECP snapshot., extract_evidence_snippets(), extract_sentences() (+8 more) Nodes (15): count_phrase_occurrences(), match_phrase_in_text(), Core deterministic classification engine (Tier 1 core)., Check if a normalized phrase appears in normalized text with word boundary…, Count occurrences of a phrase in text., Classify inherence of content against an ECP snapshot., extract_evidence_snippets(), extract_sentences() (+7 more)
### Community 48 - "test_convert_article_to_markdown.py" ### Community 48 - "test_convert_article_to_markdown.py"
Cohesion: 0.07 Cohesion: 0.08
Nodes (27): Suíte de Testes Automatizados para Conversão de Artigo JSON para Markdown.…, Testa comportamento com listas vazias, nulas ou contendo apenas placeholders., Valida parsing de datas no formato RFC 2822 (usado em feeds RSS e cabeçalhos…, Valida a cadeia de fallback completa para o campo TÍTULO (6 níveis)., Garante que subtítulo idêntico ao título seja automaticamente omitido (None)., Valida decodificação de entidades HTML nomeadas e numéricas., Garante que listas de autores/tags usem apenas a primeira fonte válida, sem…, Garante correspondência exata byte a byte para Trafilatura, Newspaper4k e… (+19 more) Nodes (25): Suíte de Testes Automatizados para Conversão de Artigo JSON para Markdown.…, Testa divisão por ponto e vírgula na string e vírgulas em elementos de lista…, Valida parsing de datas no formato RFC 2822 (usado em feeds RSS e cabeçalhos…, Valida a cadeia de fallback completa para o campo TÍTULO (6 níveis)., Garante que subtítulo idêntico ao título seja automaticamente omitido (None)., Valida decodificação de entidades HTML nomeadas e numéricas., Garante correspondência exata byte a byte para Trafilatura, Newspaper4k e…, Garante que múltiplas execuções no mesmo arquivo produzam hashes SHA-256… (+17 more)
### Community 80 - "test_extract_article_contents.py" ### Community 80 - "test_extract_article_contents.py"
Cohesion: 0.06 Cohesion: 0.06
@@ -436,8 +439,8 @@ Cohesion: 0.15
Nodes (17): emit_error(), main(), parse_args(), Namespace, ClassificationError, ErrorCode, MatchedGraphEntity, Enum (+9 more) Nodes (17): emit_error(), main(), parse_args(), Namespace, ClassificationError, ErrorCode, MatchedGraphEntity, Enum (+9 more)
### Community 101 - "ECPSnapshot" ### Community 101 - "ECPSnapshot"
Cohesion: 0.20 Cohesion: 0.24
Nodes (10): ECPSnapshot, Any, classifier(), fixture, parametrize, Controlled 24-case benchmark suite for Multilingual NLP Entity Inherence…, test_benchmark_case(), test_ecp_snapshot_defaults() (+2 more) Nodes (10): ECPSnapshot, classifier(), fixture, parametrize, Controlled 24-case benchmark suite for Multilingual NLP Entity Inherence…, test_benchmark_case(), Unit tests for ECP models, schema validation, and structured error handling., test_ecp_snapshot_defaults() (+2 more)
### Community 102 - "Feature Specification: Multilingual NLP Entity Inherence Classifier (POC)" ### Community 102 - "Feature Specification: Multilingual NLP Entity Inherence Classifier (POC)"
Cohesion: 0.14 Cohesion: 0.14
@@ -585,7 +588,7 @@ Nodes (11): 1. Validação de Entrada, Tipagem & Isolamento de Lotes, 2. Isolame
### Community 140 - "convert_article" ### Community 140 - "convert_article"
Cohesion: 0.15 Cohesion: 0.15
Nodes (13): clean_body_images(), convert_article(), Path, Remove o primeiro título H1 do corpo somente quando ele for igual ao título…, Preserva imagens com URL absoluta http/https, remove relativas/data:/vazias e…, Executa a leitura do JSON, validação, conversão e escrita atômica do arquivo…, remove_duplicate_initial_h1(), Testa remoção de H1 inicial coincidente com título com variações de espaços e… (+5 more) Nodes (13): assemble_markdown_document(), clean_body_images(), convert_article(), Path, Preserva imagens com URL absoluta http/https, remove relativas/data:/vazias e…, Monta a estrutura final do documento Markdown respeitando a ordem estrita do…, Executa a leitura do JSON, validação, conversão e escrita atômica do arquivo…, Valida descarte de data:, relativos e deduplicação mantendo a primeira… (+5 more)
### Community 141 - "005-convert-json-markdown/plan.md" ### Community 141 - "005-convert-json-markdown/plan.md"
Cohesion: 0.33 Cohesion: 0.33
@@ -631,9 +634,9 @@ Nodes (4): 13.1 Testes unitários, 13.2 Testes de integração do CLI, 13.3 Caso
Cohesion: 0.50 Cohesion: 0.50
Nodes (4): 6.1 Formato, 6.2 Valores aceitos para `selected_extractor`, 6.3 Campos obrigatórios após a resolução, 6. Contrato de entrada Nodes (4): 6.1 Formato, 6.2 Valores aceitos para `selected_extractor`, 6.3 Campos obrigatórios após a resolução, 6. Contrato de entrada
### Community 152 - "assemble_markdown_document" ### Community 152 - "InherenceClassifier"
Cohesion: 0.50 Cohesion: 0.14
Nodes (4): assemble_markdown_document(), Monta a estrutura final do documento Markdown respeitando a ordem estrita do…, Testa montagem com todos os campos e apenas com campos obrigatórios., test_assemble_markdown_document_full_and_minimal() Nodes (24): InherenceClassifier, Tier 1 Deterministic NLP Entity Inherence Classifier with optional Tier 2 /…, DecisionCategory, Unit tests for deterministic classification decision logic., test_contextual_inherent(), test_direct_inherent(), test_negative_anchor_suppression(), test_not_related() (+16 more)
### Community 153 - "convert_html_to_markdown" ### Community 153 - "convert_html_to_markdown"
Cohesion: 0.33 Cohesion: 0.33
@@ -647,25 +650,33 @@ Nodes (3): 1. Input JSON Schema, 2. Output JSON Schema, JSON Schema Contract: De
Cohesion: 0.67 Cohesion: 0.67
Nodes (3): 5.1 Incluído, 5.2 Fora do escopo, 5. Escopo Nodes (3): 5.1 Incluído, 5.2 Fora do escopo, 5. Escopo
### Community 165 - "LLMFallbackAdapter"
Cohesion: 0.13
Nodes (11): LLMFallbackAdapter, Optional adapter for LLM fallback boundary disambiguation., Returns True if an API key or custom provider function is configured., Constructs a structured disambiguation prompt for the LLM., Executes LLM fallback for ambiguous boundary cases. Returns a refined…, Parses and validates structured JSON response from LLM., Any, Valida detecção de disponibilidade por chave de API ou provider customizado. (+3 more)
### Community 166 - "remove_duplicate_initial_h1"
Cohesion: 0.50
Nodes (4): Remove o primeiro título H1 do corpo somente quando ele for igual ao título…, remove_duplicate_initial_h1(), Testa remoção de H1 inicial coincidente com título com variações de espaços e…, test_remove_duplicate_initial_h1_exact_and_variations()
## Knowledge Gaps ## Knowledge Gaps
- **696 isolated node(s):** `text-nlp-classifier`, `MatchedGraphEntity`, `graphify`, `Usage`, `What graphify is for` (+691 more) - **696 isolated node(s):** `text-nlp-classifier`, `MatchedGraphEntity`, `graphify`, `Usage`, `What graphify is for` (+691 more)
These have ≤1 connection - possible missing edges or undocumented components. These have ≤1 connection - possible missing edges or undocumented components.
- **47 thin communities (<3 nodes) omitted from report** — run `graphify query` to explore isolated nodes. - **48 thin communities (<3 nodes) omitted from report** — run `graphify query` to explore isolated nodes.
## Suggested Questions ## Suggested Questions
_Questions this graph is uniquely positioned to answer:_ _Questions this graph is uniquely positioned to answer:_
- **Why does `PRD — Conversão de artigo JSON para Markdown` connect `PRD — Conversão de artigo JSON para Markdown` to `8. Regras funcionais`, `12. Critérios de aceite`, `11. Requisitos não funcionais`, `9. Interface CLI`, `13. Estratégia de testes`, `6. Contrato de entrada`, `5. Escopo`?** - **Why does `PRD — Conversão de artigo JSON para Markdown` connect `PRD — Conversão de artigo JSON para Markdown` to `8. Regras funcionais`, `12. Critérios de aceite`, `11. Requisitos não funcionais`, `9. Interface CLI`, `13. Estratégia de testes`, `6. Contrato de entrada`, `5. Escopo`?**
_High betweenness centrality (0.007) - this node is a cross-community bridge._ _High betweenness centrality (0.008) - this node is a cross-community bridge._
- **Why does `Implementation Plan: Convert Article JSON to Markdown` connect `Implementation Plan: Convert Article JSON to Markdown` to `005-convert-json-markdown/plan.md`?**
_High betweenness centrality (0.004) - this node is a cross-community bridge._
- **Why does `Tasks: Convert Article JSON to Markdown` connect `Tasks: Convert Article JSON to Markdown` to `005-convert-json-markdown/plan.md`?**
_High betweenness centrality (0.004) - this node is a cross-community bridge._
- **Are the 10 inferred relationships involving `ECPSnapshot` (e.g. with `main()` and `BaseNLPAdapter`) actually correct?** - **Are the 10 inferred relationships involving `ECPSnapshot` (e.g. with `main()` and `BaseNLPAdapter`) actually correct?**
_`ECPSnapshot` has 10 INFERRED edges - model-reasoned connections that need verification._ _`ECPSnapshot` has 10 INFERRED edges - model-reasoned connections that need verification._
- **Are the 6 inferred relationships involving `InherenceClassifier` (e.g. with `LocalEmbeddingsAdapter` and `LLMFallbackAdapter`) actually correct?** - **Are the 6 inferred relationships involving `InherenceClassifier` (e.g. with `LocalEmbeddingsAdapter` and `LLMFallbackAdapter`) actually correct?**
_`InherenceClassifier` has 6 INFERRED edges - model-reasoned connections that need verification._ _`InherenceClassifier` has 6 INFERRED edges - model-reasoned connections that need verification._
- **Are the 12 inferred relationships involving `ExtractorName` (e.g. with `test_article_1_regression_technical_tie_markdown_images()` and `test_ct_001_three_candidates_clear_winner()`) actually correct?** - **Are the 18 inferred relationships involving `DecisionCategory` (e.g. with `LLMFallbackAdapter` and `InherenceClassifier`) actually correct?**
_`ExtractorName` has 12 INFERRED edges - model-reasoned connections that need verification._ _`DecisionCategory` has 18 INFERRED edges - model-reasoned connections that need verification._
- **Are the 10 inferred relationships involving `DecisionCategory` (e.g. with `InherenceClassifier` and `test_adversarial_apple_fruit_recipe()`) actually correct?** - **Are the 4 inferred relationships involving `LLMFallbackAdapter` (e.g. with `ClassificationResult` and `DecisionCategory`) actually correct?**
_`DecisionCategory` has 10 INFERRED edges - model-reasoned connections that need verification._ _`LLMFallbackAdapter` has 4 INFERRED edges - model-reasoned connections that need verification._
- **What connects `text-nlp-classifier`, `MatchedGraphEntity`, `graphify` to the rest of the system?**
_696 weakly-connected nodes found - possible documentation gaps or missing edges._
- **Should `Task Planning` be split into smaller, more focused modules?**
_Cohesion score 0.07407407407407407 - nodes in this community are weakly interconnected._
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
+1 -1
View File
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
+1826 -699
View File
File diff suppressed because it is too large Load Diff
+24 -18
View File
@@ -294,9 +294,9 @@
"semantic_hash": "" "semantic_hash": ""
}, },
"classify.py": { "classify.py": {
"mtime": 1787264412.2457643, "mtime": 1787320064.0177462,
"seen": 1787264519.0539427, "seen": 1787320239.022448,
"ast_hash": "e89fc4b64bd7606fc466d5338108705b", "ast_hash": "3679326c88a59f796f587e0c61c31417",
"semantic_hash": "" "semantic_hash": ""
}, },
"pyproject.toml": { "pyproject.toml": {
@@ -330,15 +330,15 @@
"semantic_hash": "" "semantic_hash": ""
}, },
"src/adapters/llm.py": { "src/adapters/llm.py": {
"mtime": 1787264412.2437606, "mtime": 1787320047.0352886,
"seen": 1787264519.0542035, "seen": 1787320239.0258572,
"ast_hash": "ba2990328f5b25ac6cdfc3b57acc9d39", "ast_hash": "ffcf26cb4e89395aa8771ddb4f5de785",
"semantic_hash": "" "semantic_hash": ""
}, },
"src/classifier.py": { "src/classifier.py": {
"mtime": 1787264459.052089, "mtime": 1787320047.0362887,
"seen": 1787264519.0542045, "seen": 1787320239.025866,
"ast_hash": "b8fb440374ecd66bb8bf67c70c19ed1f", "ast_hash": "d6cc674d407a99ab52f6d5156f9b3d8e",
"semantic_hash": "" "semantic_hash": ""
}, },
"src/language.py": { "src/language.py": {
@@ -654,9 +654,9 @@
"semantic_hash": "" "semantic_hash": ""
}, },
"README.md": { "README.md": {
"mtime": 1787319775.4204426, "mtime": 1787320219.5852203,
"seen": 1787319817.9012172, "seen": 1787320239.0933797,
"ast_hash": "f809a191cb40e8a0367c748b9c8d3e84", "ast_hash": "ce59670fbaebc5e408a30a1009a58d12",
"semantic_hash": "" "semantic_hash": ""
}, },
"scripts/extract_article_contents.py": { "scripts/extract_article_contents.py": {
@@ -816,15 +816,15 @@
"semantic_hash": "" "semantic_hash": ""
}, },
"scripts/convert_article_to_markdown.py": { "scripts/convert_article_to_markdown.py": {
"mtime": 1787318905.723034, "mtime": 1787320071.2809863,
"seen": 1787319817.895658, "seen": 1787320239.0233297,
"ast_hash": "b58fbd8b426de464df2c269c32831583", "ast_hash": "7939a71acd9b264f9d00eab5c652cd36",
"semantic_hash": "" "semantic_hash": ""
}, },
"tests/test_convert_article_to_markdown.py": { "tests/test_convert_article_to_markdown.py": {
"mtime": 1787318905.723034, "mtime": 1787320111.811304,
"seen": 1787319817.8974621, "seen": 1787320239.0292969,
"ast_hash": "95c041594afec14ed25bc237b7ff8b89", "ast_hash": "5223f35131b8e952e05c44710f3988b2",
"semantic_hash": "" "semantic_hash": ""
}, },
"docs/prd_convert_json_markdown.md": { "docs/prd_convert_json_markdown.md": {
@@ -910,5 +910,11 @@
"seen": 1787317712.8280091, "seen": 1787317712.8280091,
"ast_hash": "c63c1c39e34e08a239aa8bea3c264756", "ast_hash": "c63c1c39e34e08a239aa8bea3c264756",
"semantic_hash": "" "semantic_hash": ""
},
"tests/test_llm_fallback.py": {
"mtime": 1787320089.2120655,
"seen": 1787320239.0310187,
"ast_hash": "74bc17c4668551e77214d6581baf206c",
"semantic_hash": ""
} }
} }
+1 -1
View File
@@ -146,7 +146,7 @@ def validate_url(value: Any) -> Optional[str]:
return None return None
def convert_html_to_markdown(html_content: str) -> str: def convert_html_to_markdown(html_content: Optional[str]) -> str:
"""Converte HTML para Markdown usando títulos ATX, removendo scripts e estilos.""" """Converte HTML para Markdown usando títulos ATX, removendo scripts e estilos."""
if not html_content or not isinstance(html_content, str) or not html_content.strip(): if not html_content or not isinstance(html_content, str) or not html_content.strip():
return "" return ""
+90 -6
View File
@@ -1,38 +1,122 @@
"""Optional LLM fallback adapter (Tier 3). """Optional LLM fallback adapter (Tier 3).
Disabled by default. Provides fallback interface for boundary disambiguation Disabled by default. Provides fallback interface for boundary disambiguation
without requiring OpenAI/Anthropic/Gemini API keys for core POC execution. without requiring external API keys for core POC execution.
""" """
from __future__ import annotations from __future__ import annotations
import json
import os import os
from typing import Callable, Optional
from src.adapters.base import BaseNLPAdapter from src.adapters.base import BaseNLPAdapter
from src.models import ClassificationResult, ECPSnapshot from src.models import ClassificationResult, DecisionCategory, ECPSnapshot
class LLMFallbackAdapter(BaseNLPAdapter): class LLMFallbackAdapter(BaseNLPAdapter):
"""Optional adapter for LLM fallback boundary disambiguation.""" """Optional adapter for LLM fallback boundary disambiguation."""
def __init__(self, model_name: str = "gpt-4o-mini", api_key: str | None = None) -> None: def __init__(
self,
model_name: str = "gpt-4o-mini",
api_key: str | None = None,
provider_fn: Optional[Callable[[str], str]] = None,
) -> None:
self.model_name = model_name self.model_name = model_name
self.api_key = api_key or os.environ.get("OPENAI_API_KEY") self.api_key = api_key or os.environ.get("OPENAI_API_KEY")
self.provider_fn = provider_fn
def is_available(self) -> bool: def is_available(self) -> bool:
return bool(self.api_key) """Returns True if an API key or custom provider function is configured."""
return bool(self.api_key or self.provider_fn)
def evaluate_similarity(self, text: str, terms: list[str]) -> float: def evaluate_similarity(self, text: str, terms: list[str]) -> float:
return 0.0 return 0.0
def build_prompt(
self, ecp: ECPSnapshot, content_md: str, initial_result: ClassificationResult
) -> str:
"""Constructs a structured disambiguation prompt for the LLM."""
return (
f"You are an NLP Entity Inherence Evaluator.\n"
f"Target Entity: {ecp.target_name} (Aliases: {', '.join(ecp.aliases)})\n"
f"Domain: {ecp.domain}\n"
f"Initial Tier-1 Decision: {initial_result.decision.value} (Confidence: {initial_result.confidence})\n\n"
f"Document Content:\n```markdown\n{content_md[:2000]}\n```\n\n"
f"Evaluate if the document is substantively inherent to the target entity.\n"
f'Respond with JSON: {{"decision": "DIRECT_INHERENT"|"CONTEXTUAL_INHERENT"|"TANGENTIAL"|"NOT_RELATED", '
f'"confidence": 0.0-1.0, "rationale": "explanation"}}'
)
def disambiguate( def disambiguate(
self, self,
ecp: ECPSnapshot, ecp: ECPSnapshot,
content_md: str, content_md: str,
initial_result: ClassificationResult, initial_result: ClassificationResult,
) -> ClassificationResult | None: ) -> ClassificationResult | None:
# If API key is not configured or case is already clear, skip """
Executes LLM fallback for ambiguous boundary cases.
Returns a refined ClassificationResult or None if skipped.
"""
if not self.is_available(): if not self.is_available():
return None return None
# In POC, Tier 1 is definitive; LLM fallback stub is available for extension
prompt = self.build_prompt(ecp, content_md, initial_result)
# If custom provider function is provided (e.g. for testing or custom runtime)
if self.provider_fn is not None:
raw_response = self.provider_fn(prompt)
return self._parse_llm_response(raw_response, initial_result)
# Stub default for POC when only API key string is present without active SDK
return None return None
def _parse_llm_response(
self,
raw_response: str,
initial_result: ClassificationResult,
) -> ClassificationResult | None:
"""Parses and validates structured JSON response from LLM."""
try:
# Extract JSON block if surrounded by markdown code fences
clean_str = raw_response.strip()
if clean_str.startswith("```json"):
clean_str = clean_str[7:]
if clean_str.startswith("```"):
clean_str = clean_str[3:]
if clean_str.endswith("```"):
clean_str = clean_str[:-3]
clean_str = clean_str.strip()
parsed = json.loads(clean_str)
if not isinstance(parsed, dict):
return None
raw_decision = parsed.get("decision")
if not raw_decision:
return None
decision = DecisionCategory(raw_decision)
confidence = float(parsed.get("confidence", 0.90))
confidence = max(0.0, min(1.0, confidence))
rationale = str(parsed.get("rationale", "LLM boundary disambiguation."))
is_inherent = decision in (
DecisionCategory.DIRECT_INHERENT,
DecisionCategory.CONTEXTUAL_INHERENT,
)
return ClassificationResult(
decision=decision,
is_inherent=is_inherent,
confidence=round(confidence, 4),
detected_language=initial_result.detected_language,
matched_anchors=initial_result.matched_anchors,
negative_matches=initial_result.negative_matches,
graph_matches=initial_result.graph_matches,
evidence=initial_result.evidence,
rationale=f"[Tier 3 LLM] {rationale}",
warnings=initial_result.warnings + ["[Tier 3 LLM Override applied]"],
)
except Exception:
return None
+30 -7
View File
@@ -3,7 +3,7 @@
from __future__ import annotations from __future__ import annotations
import re import re
from typing import Any from typing import Any, Optional
from src.language import detect_language, normalize_text from src.language import detect_language, normalize_text
from src.models import ( from src.models import (
@@ -39,20 +39,25 @@ def count_phrase_occurrences(phrase: str, normalized_text: str) -> int:
class InherenceClassifier: class InherenceClassifier:
"""Tier 1 Deterministic NLP Entity Inherence Classifier.""" """Tier 1 Deterministic NLP Entity Inherence Classifier with optional Tier 2 / Tier 3 adapters."""
def __init__(self, enable_embeddings: bool = False, enable_llm: bool = False) -> None: def __init__(
self,
enable_embeddings: bool = False,
enable_llm: bool = False,
llm_adapter: Optional[Any] = None,
) -> None:
self.enable_embeddings = enable_embeddings self.enable_embeddings = enable_embeddings
self.enable_llm = enable_llm self.enable_llm = enable_llm or (llm_adapter is not None)
self._embeddings_adapter = None self._embeddings_adapter = None
self._llm_adapter = None self._llm_adapter = llm_adapter
if enable_embeddings: if enable_embeddings:
from src.adapters.embeddings import LocalEmbeddingsAdapter from src.adapters.embeddings import LocalEmbeddingsAdapter
self._embeddings_adapter = LocalEmbeddingsAdapter() self._embeddings_adapter = LocalEmbeddingsAdapter()
if enable_llm: if self.enable_llm and self._llm_adapter is None:
from src.adapters.llm import LLMFallbackAdapter from src.adapters.llm import LLMFallbackAdapter
self._llm_adapter = LLMFallbackAdapter() self._llm_adapter = LLMFallbackAdapter()
@@ -224,7 +229,7 @@ class InherenceClassifier:
if not evidence and has_negative_match: if not evidence and has_negative_match:
evidence = extract_evidence_snippets(content_md, matched_negative_anchors) evidence = extract_evidence_snippets(content_md, matched_negative_anchors)
return ClassificationResult( tier1_result = ClassificationResult(
decision=decision, decision=decision,
is_inherent=is_inherent, is_inherent=is_inherent,
confidence=round(confidence, 4), confidence=round(confidence, 4),
@@ -236,3 +241,21 @@ class InherenceClassifier:
rationale=rationale, rationale=rationale,
warnings=warnings, warnings=warnings,
) )
# Tier 3 (LLM Fallback): Disambiguation for ambiguous boundary cases when enabled
if self.enable_llm and self._llm_adapter and self._llm_adapter.is_available():
# Invoke LLM only for low-confidence or boundary/tangential decisions
if (
tier1_result.confidence < 0.60
or tier1_result.decision == DecisionCategory.TANGENTIAL
):
try:
llm_result = self._llm_adapter.disambiguate(ecp, content_md, tier1_result)
if llm_result is not None:
return llm_result
except Exception as e:
tier1_result.warnings.append(
f"LLM fallback failed, retained Tier 1 decision: {e}"
)
return tier1_result
+2 -1
View File
@@ -14,6 +14,7 @@ import json
import subprocess import subprocess
import sys import sys
from pathlib import Path from pathlib import Path
from typing import Any
import pytest import pytest
@@ -507,7 +508,7 @@ def test_clean_body_images_removes_invalid_and_deduplicates():
def test_assemble_markdown_document_full_and_minimal(): def test_assemble_markdown_document_full_and_minimal():
"""Testa montagem com todos os campos e apenas com campos obrigatórios.""" """Testa montagem com todos os campos e apenas com campos obrigatórios."""
# Artigo Mínimo (apenas Título e URL Original) # Artigo Mínimo (apenas Título e URL Original)
min_meta = { min_meta: dict[str, Any] = {
"title": "Título Mínimo", "title": "Título Mínimo",
"original_url": "https://example.com/minimo", "original_url": "https://example.com/minimo",
"subtitle": None, "subtitle": None,
+327
View File
@@ -0,0 +1,327 @@
"""
Suíte de Testes para o Adaptador de Fallback para LLM (Tier 3) do Classificador de Inerência.
Cobre cenários unitários, de integração de pipeline, de parsing estruturado, de desambiguação
de casos limiares e de degradação graciosa em falhas de API conforme o requisito FR-004.
"""
from __future__ import annotations
import json
import subprocess
import sys
from pathlib import Path
from src.adapters.llm import LLMFallbackAdapter
from src.classifier import InherenceClassifier
from src.models import (
ClassificationResult,
DecisionCategory,
ECPSnapshot,
)
SCRIPT_PATH = Path(__file__).parent.parent / "classify.py"
# ==============================================================================
# 1. Testes Unitários do LLMFallbackAdapter
# ==============================================================================
def test_llm_adapter_availability_detection():
"""Valida detecção de disponibilidade por chave de API ou provider customizado."""
# Sem chave e sem provider
adapter_empty = LLMFallbackAdapter(api_key="")
assert adapter_empty.is_available() is False
# Com chave de API
adapter_with_key = LLMFallbackAdapter(api_key="sk-test-key-12345")
assert adapter_with_key.is_available() is True
# Com provider function
adapter_with_fn = LLMFallbackAdapter(
api_key="", provider_fn=lambda p: '{"decision": "DIRECT_INHERENT"}'
)
assert adapter_with_fn.is_available() is True
def test_llm_adapter_build_prompt_structure():
"""Valida a montagem do prompt de desambiguação com metadados do ECP e documento."""
adapter = LLMFallbackAdapter(api_key="test")
ecp = ECPSnapshot(
target_entity_id="ecp_river",
target_name="River Plate",
aliases=["Club Atlético River Plate", "CARP"],
domain="Futebol",
anchors=["Monumental", "Libertadores"],
)
initial_res = ClassificationResult(
decision=DecisionCategory.TANGENTIAL,
is_inherent=False,
confidence=0.40,
detected_language="es",
matched_anchors=["River"],
negative_matches=[],
graph_matches=[],
evidence=["River"],
rationale="Passing mention.",
warnings=[],
)
prompt = adapter.build_prompt(ecp, "# Título do Artigo\n\nConteúdo sobre o jogo.", initial_res)
assert "Target Entity: River Plate" in prompt
assert "Futebol" in prompt
assert "TANGENTIAL" in prompt
assert "Título do Artigo" in prompt
def test_llm_adapter_parsing_valid_json_response():
"""Valida o parsing e instanciação correta do ClassificationResult a partir da resposta do LLM."""
adapter = LLMFallbackAdapter(
provider_fn=lambda p: json.dumps(
{
"decision": "DIRECT_INHERENT",
"confidence": 0.95,
"rationale": "Artigo detalha o desempenho da equipe no torneio.",
}
)
)
ecp = ECPSnapshot(
target_entity_id="ecp_test",
target_name="Test Entity",
aliases=["Test"],
domain="Tech",
anchors=["cloud"],
)
initial = ClassificationResult(
decision=DecisionCategory.TANGENTIAL,
is_inherent=False,
confidence=0.40,
detected_language="pt",
matched_anchors=["Test"],
negative_matches=[],
graph_matches=[],
evidence=["Test"],
rationale="Weak match.",
warnings=["Low contextual density."],
)
refined = adapter.disambiguate(ecp, "Document content...", initial)
assert refined is not None
assert refined.decision == DecisionCategory.DIRECT_INHERENT
assert refined.is_inherent is True
assert refined.confidence == 0.95
assert "[Tier 3 LLM]" in refined.rationale
assert "[Tier 3 LLM Override applied]" in refined.warnings
def test_llm_adapter_parsing_json_wrapped_in_markdown_codeblock():
"""Valida extração de JSON quando a resposta do LLM vem formatada em bloco markdown ```json ... ```."""
raw_md_json = '```json\n{\n "decision": "CONTEXTUAL_INHERENT",\n "confidence": 0.88,\n "rationale": "Conexão contextual forte através da subsidiária."\n}\n```'
adapter = LLMFallbackAdapter(provider_fn=lambda p: raw_md_json)
ecp = ECPSnapshot(
target_entity_id="ecp_test",
target_name="Test Entity",
aliases=["Test"],
domain="Tech",
anchors=["cloud"],
)
initial = ClassificationResult(
decision=DecisionCategory.TANGENTIAL,
is_inherent=False,
confidence=0.40,
detected_language="pt",
matched_anchors=[],
negative_matches=[],
graph_matches=[],
evidence=[],
rationale="Weak match.",
warnings=[],
)
refined = adapter.disambiguate(ecp, "Content...", initial)
assert refined is not None
assert refined.decision == DecisionCategory.CONTEXTUAL_INHERENT
assert refined.is_inherent is True
assert refined.confidence == 0.88
def test_llm_adapter_handling_invalid_and_corrupt_responses():
"""Valida que respostas corrompidas ou JSONs sem campos obrigatórios retornem None com segurança."""
def make_bad_provider(resp_str: str):
def _prov(prompt: str) -> str:
return resp_str
return _prov
for bad_response in [
"Desculpe, não consegui avaliar o texto.",
"{json_invalido_sem_fechamento",
json.dumps({"campo_desconhecido": "valor"}),
json.dumps({"decision": "DECISAO_INEXISTENTE"}),
]:
adapter = LLMFallbackAdapter(provider_fn=make_bad_provider(bad_response))
ecp = ECPSnapshot(
target_entity_id="ecp_test",
target_name="Test Entity",
aliases=["Test"],
domain="Tech",
anchors=["cloud"],
)
initial = ClassificationResult(
decision=DecisionCategory.TANGENTIAL,
is_inherent=False,
confidence=0.40,
detected_language="pt",
matched_anchors=[],
negative_matches=[],
graph_matches=[],
evidence=[],
rationale="Initial.",
warnings=[],
)
assert adapter.disambiguate(ecp, "Content...", initial) is None
# ==============================================================================
# 2. Testes de Integração de Pipeline (InherenceClassifier com Tier 3)
# ==============================================================================
def test_classifier_triggers_tier3_on_ambiguous_tangential_case():
"""
Garante que o classificador dispare o Tier 3 LLM para casos ambíguos (TANGENTIAL)
e adote o refinamento retornado.
"""
mock_adapter = LLMFallbackAdapter(
provider_fn=lambda prompt: json.dumps(
{
"decision": "DIRECT_INHERENT",
"confidence": 0.92,
"rationale": "Análise profunda revelou que o texto é focado na entidade alvo.",
}
)
)
classifier = InherenceClassifier(enable_llm=True, llm_adapter=mock_adapter)
ecp = ECPSnapshot(
target_entity_id="ecp_empresa",
target_name="EmpresaAlfa",
aliases=["EmpresaAlfa"],
domain="Tecnologia",
anchors=["software", "computação em nuvem"],
)
# Texto com menção única sem âncoras temáticas (Tier 1 produziria TANGENTIAL)
ambiguous_content = "A EmpresaAlfa esteve presente no evento de encerramento anual da cidade."
result = classifier.classify(ecp, ambiguous_content)
# Como enable_llm=True e o caso era TANGENTIAL, o Tier 3 substitui a decisão
assert result.decision == DecisionCategory.DIRECT_INHERENT
assert result.is_inherent is True
assert result.confidence == 0.92
assert "[Tier 3 LLM]" in result.rationale
def test_classifier_skips_tier3_on_clear_direct_inherent_case():
"""
Garante que casos claros (alta confiança e alta densidade de âncoras) NÃO chamem
o LLM, economizando chamadas desnecessárias conforme FR-004.
"""
call_tracker = {"called": False}
def tracking_provider(prompt: str) -> str:
call_tracker["called"] = True
return json.dumps({"decision": "DIRECT_INHERENT", "confidence": 0.99})
mock_adapter = LLMFallbackAdapter(provider_fn=tracking_provider)
classifier = InherenceClassifier(enable_llm=True, llm_adapter=mock_adapter)
ecp = ECPSnapshot(
target_entity_id="ecp_empresa",
target_name="EmpresaAlfa",
aliases=["EmpresaAlfa"],
domain="Tecnologia",
anchors=["software", "computação em nuvem", "inteligência artificial"],
)
# Caso claro com alta densidade de âncoras
clear_content = "A EmpresaAlfa desenvolveu uma nova plataforma de software baseada em computação em nuvem e inteligência artificial."
result = classifier.classify(ecp, clear_content)
assert result.decision == DecisionCategory.DIRECT_INHERENT
assert result.confidence >= 0.85
# O LLM NÃO deve ter sido chamado
assert call_tracker["called"] is False
def test_classifier_graceful_degradation_when_llm_raises_exception():
"""
Garante que se o LLM falhar por erro de rede ou timeout, o classificador mantenha
o resultado do Tier 1 com degradação graciosa e registre o aviso em warnings.
"""
def failing_provider(prompt: str) -> str:
raise ConnectionError("Timeout ao conectar com a API do modelo de linguagem.")
mock_adapter = LLMFallbackAdapter(provider_fn=failing_provider)
classifier = InherenceClassifier(enable_llm=True, llm_adapter=mock_adapter)
ecp = ECPSnapshot(
target_entity_id="ecp_empresa",
target_name="EmpresaAlfa",
aliases=["EmpresaAlfa"],
domain="Tecnologia",
anchors=["software"],
)
ambiguous_content = "A EmpresaAlfa participou da conferência."
result = classifier.classify(ecp, ambiguous_content)
# Retém a decisão original do Tier 1
assert result.decision == DecisionCategory.TANGENTIAL
assert result.is_inherent is False
# Contém aviso sobre a falha do LLM sem quebrar a execução
assert any("LLM fallback failed" in w for w in result.warnings)
# ==============================================================================
# 3. Teste de Integração CLI com a Flag --enable-llm
# ==============================================================================
def test_cli_execution_with_enable_llm_flag(tmp_path):
"""Garante que a CLI classify.py aceite e processe a flag --enable-llm sem erros."""
ecp_file = tmp_path / "test_ecp.json"
ecp_file.write_text(
json.dumps(
{
"target_entity_id": "ecp_test",
"target_name": "TestCorp",
"aliases": ["TestCorp"],
"domain": "Tech",
"anchors": ["software", "cloud"],
}
),
encoding="utf-8",
)
content_file = tmp_path / "test_doc.md"
content_file.write_text("# TestCorp\n\nTestCorp builds cloud software.", encoding="utf-8")
output_file = tmp_path / "out.json"
res = subprocess.run(
[
sys.executable,
str(SCRIPT_PATH),
"--ecp",
str(ecp_file),
"--content",
str(content_file),
"--output",
str(output_file),
"--enable-llm",
],
capture_output=True,
text=True,
)
assert res.returncode == 0, f"Erro na CLI: {res.stderr}"
assert output_file.exists()
data = json.loads(output_file.read_text(encoding="utf-8"))
assert data["decision"] == "DIRECT_INHERENT"
assert data["is_inherent"] is True