refactor(llm): elevate prompt engineering with context grounding and contrastive taxonomy
This commit is contained in:
+81
-10
@@ -37,17 +37,88 @@ class LLMFallbackAdapter(BaseNLPAdapter):
|
||||
def build_prompt(
|
||||
self, ecp: ECPSnapshot, content_md: str, initial_result: ClassificationResult
|
||||
) -> str:
|
||||
"""Constructs a structured disambiguation prompt for the LLM."""
|
||||
return (
|
||||
f"You are an NLP Entity Inherence Evaluator.\n"
|
||||
f"Target Entity: {ecp.target_name} (Aliases: {', '.join(ecp.aliases)})\n"
|
||||
f"Domain: {ecp.domain}\n"
|
||||
f"Initial Tier-1 Decision: {initial_result.decision.value} (Confidence: {initial_result.confidence})\n\n"
|
||||
f"Document Content:\n```markdown\n{content_md[:2000]}\n```\n\n"
|
||||
f"Evaluate if the document is substantively inherent to the target entity.\n"
|
||||
f'Respond with JSON: {{"decision": "DIRECT_INHERENT"|"CONTEXTUAL_INHERENT"|"TANGENTIAL"|"NOT_RELATED", '
|
||||
f'"confidence": 0.0-1.0, "rationale": "explanation"}}'
|
||||
"""Constructs an expert-engineered prompt for multilingual entity inherence disambiguation."""
|
||||
matched_pos = (
|
||||
", ".join(initial_result.matched_anchors) if initial_result.matched_anchors else "None"
|
||||
)
|
||||
matched_neg = (
|
||||
", ".join(initial_result.negative_matches)
|
||||
if initial_result.negative_matches
|
||||
else "None"
|
||||
)
|
||||
matched_graph = (
|
||||
", ".join([f"{g['name']} ({g['relation_type']})" for g in initial_result.graph_matches])
|
||||
if initial_result.graph_matches
|
||||
else "None"
|
||||
)
|
||||
warnings_str = "; ".join(initial_result.warnings) if initial_result.warnings else "None"
|
||||
top_anchors = ", ".join(ecp.anchors[:20]) if ecp.anchors else "None"
|
||||
top_negatives = ", ".join(ecp.negative_anchors[:15]) if ecp.negative_anchors else "None"
|
||||
related_entities_summary = (
|
||||
", ".join(
|
||||
[
|
||||
f"{r.name} [{r.relation_type}, weight: {r.weight}]"
|
||||
for r in ecp.related_entities[:10]
|
||||
]
|
||||
)
|
||||
if ecp.related_entities
|
||||
else "None"
|
||||
)
|
||||
|
||||
return f"""You are a Principal Knowledge Graph & Multilingual NLP Entity Inherence Specialist.
|
||||
|
||||
### OBJECTIVE
|
||||
Your task is to resolve an AMBIGUOUS boundary classification case flagged by the deterministic Tier-1 NLP pipeline for the target entity: **{ecp.target_name}**.
|
||||
|
||||
### 1. TARGET ENTITY CONTEXT PROFILE (ECP)
|
||||
- **Target Entity ID**: `{ecp.target_entity_id}`
|
||||
- **Canonical Name**: {ecp.target_name}
|
||||
- **Domain / Industry**: {ecp.domain}
|
||||
- **Known Valid Aliases**: {", ".join(ecp.aliases)}
|
||||
- **Expected Thematic Anchors (Positive Signals)**: {top_anchors}
|
||||
- **Disambiguation Negative Anchors (Known Homonyms / Distractors)**: {top_negatives}
|
||||
- **Knowledge Graph Connected Entities**: {related_entities_summary}
|
||||
|
||||
### 2. TIER-1 NLP DIAGNOSIS (WHY IT WAS FLAGGED AS AMBIGUOUS)
|
||||
- **Initial Tier-1 Decision**: `{initial_result.decision.value}` (Confidence: {initial_result.confidence})
|
||||
- **Tier-1 Rationale**: {initial_result.rationale}
|
||||
- **Telemetry Warnings**: {warnings_str}
|
||||
- **Positive Term Matches**: {matched_pos}
|
||||
- **Negative / Distractor Matches**: {matched_neg}
|
||||
- **Graph Node Matches**: {matched_graph}
|
||||
|
||||
### 3. CONTRASTIVE TAXONOMY & DECISION DEFINITIONS
|
||||
1. **DIRECT_INHERENT** (`is_inherent = true`):
|
||||
- The document is primarily, directly, or substantively about `{ecp.target_name}`.
|
||||
- The narrative explores the entity's direct actions, performances, strategy, status, or key personnel.
|
||||
2. **CONTEXTUAL_INHERENT** (`is_inherent = true`):
|
||||
- The document is not exclusively about the target entity, but the entity is an active, material participant in the discussed ecosystem (e.g. key rival in an active match, crucial partner in a corporate deal, subsidiary with material parent impact, or direct regulatory subject).
|
||||
3. **TANGENTIAL** (`is_inherent = false`):
|
||||
- The target entity is mentioned only in passing, as a figure of speech / metaphor, in an incidental illustrative list, or as mere background trivia without playing an active role in the article's core narrative.
|
||||
4. **NOT_RELATED** (`is_inherent = false`):
|
||||
- The document is completely unrelated, or the mention refers to a homonym/distractor (matching negative anchors or an entirely different entity with a similar name).
|
||||
|
||||
### 4. DISAMBIGUATION EVALUATION PROTOCOL
|
||||
1. **Language & Intent**: Read the document in its native language (`{initial_result.detected_language}`). Determine the primary subject matter.
|
||||
2. **Homonym Filtering**: Verify whether mentions of `{ecp.target_name}` refer to the intended entity in domain `{ecp.domain}` or to an unrelated namesake.
|
||||
3. **Substantive Role vs. Passing Footnote**: Evaluate whether the mention is central (DIRECT), systemic/contextual (CONTEXTUAL), or merely incidental (TANGENTIAL).
|
||||
4. **Final Decision**: Provide your definitive calibrated resolution in strict JSON format.
|
||||
|
||||
### 5. DOCUMENT CONTENT (MARKDOWN)
|
||||
```markdown
|
||||
{content_md[:3500]}
|
||||
```
|
||||
|
||||
### 6. OUTPUT FORMAT
|
||||
Respond ONLY with a valid JSON object matching this schema:
|
||||
```json
|
||||
{{
|
||||
"analysis_summary": "Brief 1-sentence synthesis of the document's main focus and entity relation.",
|
||||
"decision": "DIRECT_INHERENT" | "CONTEXTUAL_INHERENT" | "TANGENTIAL" | "NOT_RELATED",
|
||||
"confidence": 0.80 to 0.99,
|
||||
"rationale": "Clear, concise justification explaining why this decision resolves the Tier-1 NLP ambiguity."
|
||||
}}
|
||||
```"""
|
||||
|
||||
def disambiguate(
|
||||
self,
|
||||
|
||||
Reference in New Issue
Block a user