- introduce Gold Standard evaluation corpus - document decision taxonomy - define prompt-engineering methodology - add regression workflow - establish Prompt Version 2 baseline - validate decision_simple, decision_deferred and decision_none
51 lines
1.6 KiB
JSON
51 lines
1.6 KiB
JSON
{
|
|
"facts": [
|
|
{
|
|
"fact": "The staging import handled 12,000 rows last night.",
|
|
"evidence": "Priya: The staging import handled 12,000 rows last night."
|
|
},
|
|
{
|
|
"fact": "The staging import took 48 minutes.",
|
|
"evidence": "Mateo: It finished, but it took 48 minutes."
|
|
},
|
|
{
|
|
"fact": "The partner demo target is 30 minutes.",
|
|
"evidence": "Lena: Yes, that is still the demo target."
|
|
}
|
|
],
|
|
"decisions": [
|
|
{
|
|
"decision": "Address normalization will be disabled for the demo import only.",
|
|
"evidence": "Can we agree to disable address normalization for the demo import only? Lena: Yes, for the demo import only. Mateo: Agreed."
|
|
}
|
|
],
|
|
"todos": [
|
|
{
|
|
"task": "Update the demo import configuration.",
|
|
"responsible": "Mateo",
|
|
"deadline": "Friday noon",
|
|
"evidence": "Mateo: I will do that before Friday noon."
|
|
}
|
|
],
|
|
"questions": [
|
|
{
|
|
"question": "Whether a faster normalizer is needed after the demo.",
|
|
"evidence": "Lena: And the open question is whether we need a faster normalizer after the demo."
|
|
}
|
|
],
|
|
"positions": [
|
|
{
|
|
"speaker": "Mateo",
|
|
"position": "Mateo thinks address normalization is the slow part.",
|
|
"evidence": "Mateo: I think the slow part is address normalization."
|
|
}
|
|
],
|
|
"technical": [
|
|
{
|
|
"subject": "Demo import configuration",
|
|
"statement": "Address normalization is disabled only for the demo import; production imports keep full normalization.",
|
|
"evidence": "for the demo import only. Production imports keep the full normalization."
|
|
}
|
|
]
|
|
}
|