35 lines
1.0 KiB
YAML
35 lines
1.0 KiB
YAML
# Promptfoo Test Environment & Suite Configuration
|
|
# Contract: Doc 04, Doc 07, FR-057, FR-070
|
|
# Constraints: Zero regex, no semantic contains/not-contains, no LLM-as-a-judge for grounding, no powerful models.
|
|
|
|
description: "Article Consolidation Runtime - Hygiene & Enrichment Offline Evaluation"
|
|
|
|
prompts:
|
|
- "file://prompts/article_content_hygiene.v1.txt"
|
|
- "file://prompts/article_sentiment_tags.v1.txt"
|
|
|
|
providers:
|
|
- id: "groq:llama-3.1-8b-instant"
|
|
config:
|
|
temperature: 0.0
|
|
response_format:
|
|
type: "json_object"
|
|
- id: "deepseek:deepseek-chat"
|
|
config:
|
|
temperature: 0.0
|
|
response_format:
|
|
type: "json_object"
|
|
|
|
defaultTest:
|
|
options:
|
|
provider: "groq:llama-3.1-8b-instant"
|
|
|
|
tests:
|
|
- description: "Reference 20 Regression - Schema and ID Grounding"
|
|
vars:
|
|
article_json: "file://evals/reference_20/article_01.json"
|
|
assert:
|
|
- type: "is-json"
|
|
- type: "javascript"
|
|
value: "JSON.parse(output).title_candidate_id !== undefined && Array.isArray(JSON.parse(output).kept_block_ids)"
|