- introduce Gold Standard evaluation corpus - document decision taxonomy - define prompt-engineering methodology - add regression workflow - establish Prompt Version 2 baseline - validate decision_simple, decision_deferred and decision_none
33 lines
1.1 KiB
JSON
33 lines
1.1 KiB
JSON
{
|
|
"facts": [
|
|
{
|
|
"fact": "The old scanner gateway is still running in aisle three.",
|
|
"evidence": "Tom: The old scanner gateway is still running in aisle three."
|
|
},
|
|
{
|
|
"fact": "Aisles one and two moved to the new gateway last week.",
|
|
"evidence": "Tom: Yes. Aisles one and two moved to the new gateway last week."
|
|
},
|
|
{
|
|
"fact": "The new gateway is handling live scans for receiving.",
|
|
"evidence": "Iris: The new gateway is already handling live scans for receiving."
|
|
}
|
|
],
|
|
"decisions": [],
|
|
"todos": [],
|
|
"questions": [
|
|
{
|
|
"question": "Is aisle three the only scanner gateway still left on the old gateway?",
|
|
"evidence": "Elena: Is that the only one left?"
|
|
}
|
|
],
|
|
"positions": [],
|
|
"technical": [
|
|
{
|
|
"subject": "Scanner gateway rollout",
|
|
"statement": "Aisle three remains on the old scanner gateway while aisles one and two use the new gateway.",
|
|
"evidence": "The old scanner gateway is still running in aisle three. Aisles one and two moved to the new gateway last week."
|
|
}
|
|
]
|
|
}
|