mirror of
https://github.com/priyanshujain/sanderling.git
synced 2026-10-02 19:17:10 +00:00
fix(analyze): divide by actions that ran
Defects per thousand actions counted every step, including steps that chose nothing and steps whose action was never dispatched. The inflation is policy-dependent, so it does not cancel between arms: on the fixture campaign the model arm's yield was reported at 60.3 per thousand against a true 120.7, because half its steps did nothing. A runs.jsonl without the count is refused by name and line rather than read as zero actions, which would report every per-action rate wrongly. The report also carries steps beside actions now, so the gap is visible rather than folded into a denominator. Claude-Session: https://claude.ai/code/session_01A5KmftdEJ49A9z5mF5ESrX
This commit is contained in:
1 parent
f6d562e3dc
commit
2cbe03c3fb
6 files changed
+162
-36
No files matched your search
@@ -32,11 +32,13 @@ func writeReport(result analysis, out io.Writer) {
|
||||
|
||||
fmt.Fprintln(out)
|
||||
fmt.Fprintln(out, "a detection is one distinct property violated in one run; run hours sum the per-run wall clock")
|
||||
writeTable(out, []string{"arm", "actions", "run hours", "detections", "defects/1k actions", "defects/hour", "distinct defects", "found in one run"},
|
||||
fmt.Fprintln(out, "actions count the steps that dispatched one; the rest chose nothing or had the choice thrown away")
|
||||
writeTable(out, []string{"arm", "steps", "actions", "run hours", "detections", "defects/1k actions", "defects/hour", "distinct defects", "found in one run"},
|
||||
func(add func(...string)) {
|
||||
for _, summary := range result.Arms {
|
||||
add(
|
||||
summary.Arm,
|
||||
strconv.Itoa(summary.TotalSteps),
|
||||
strconv.Itoa(summary.TotalActions),
|
||||
fmt.Sprintf("%.2f", summary.TotalRunHours),
|
||||
strconv.Itoa(summary.Detections),
|
||||
|
||||
Reference in new issue
Block a user