- introduce Gold Standard evaluation corpus - document decision taxonomy - define prompt-engineering methodology - add regression workflow - establish Prompt Version 2 baseline - validate decision_simple, decision_deferred and decision_none
34 lines
1.1 KiB
JSON
34 lines
1.1 KiB
JSON
{
|
|
"facts": [
|
|
{
|
|
"fact": "The mobile build failed on the staging runner.",
|
|
"evidence": "Sofia: The mobile build failed again on the staging runner."
|
|
},
|
|
{
|
|
"fact": "The certificate is valid until October.",
|
|
"evidence": "Nils: The certificate itself is valid until October."
|
|
}
|
|
],
|
|
"decisions": [],
|
|
"todos": [],
|
|
"questions": [
|
|
{
|
|
"question": "Is the mobile build failing with the same error as yesterday?",
|
|
"evidence": "Nils: Same error as yesterday?"
|
|
}
|
|
],
|
|
"positions": [],
|
|
"technical": [
|
|
{
|
|
"subject": "iOS staging build",
|
|
"statement": "The iOS job fails during code signing because the runner uses the old keychain path.",
|
|
"evidence": "The iOS job now fails during code signing. The problem is that the runner uses the old keychain path."
|
|
},
|
|
{
|
|
"subject": "Failure classification",
|
|
"statement": "The failure is a runner configuration issue, not a certificate expiry issue.",
|
|
"evidence": "So it is a runner configuration issue, not a certificate expiry issue. Sofia: Exactly."
|
|
}
|
|
]
|
|
}
|