- introduce Gold Standard evaluation corpus - document decision taxonomy - define prompt-engineering methodology - add regression workflow - establish Prompt Version 2 baseline - validate decision_simple, decision_deferred and decision_none
28 lines
766 B
JSON
28 lines
766 B
JSON
{
|
|
"facts": [
|
|
{
|
|
"fact": "The vendor invoice came in this morning.",
|
|
"evidence": "Kai: The vendor invoice came in this morning."
|
|
},
|
|
{
|
|
"fact": "The invoice line item says support package.",
|
|
"evidence": "Kai: I don't know. The line item just says support package."
|
|
}
|
|
],
|
|
"decisions": [
|
|
{
|
|
"decision": "The support-hours invoice issue will remain an open question for now.",
|
|
"evidence": "Ruth: Fine. Let's leave it as an open question for now."
|
|
}
|
|
],
|
|
"todos": [],
|
|
"questions": [
|
|
{
|
|
"question": "Does the vendor invoice include the extra support hours from March?",
|
|
"evidence": "Ruth: Does it include the extra support hours from March?"
|
|
}
|
|
],
|
|
"positions": [],
|
|
"technical": []
|
|
}
|