| 105 | const pct1 = (x) => `${x.toFixed(1)}%`; |
| 106 | |
| 107 | export function formatComparison(arms) { |
| 108 | const W = 36; const C = 24; |
| 109 | const out = []; |
| 110 | // The leading space is a separator, not padding: a `median [min–max]` cell can |
| 111 | // fill its column, and two of those with only padStart between them run |
| 112 | // together into one unreadable number. |
| 113 | const row = (label, cells) => out.push(' ' + label.padEnd(W) + cells.map((c) => ' ' + String(c).padStart(C - 1)).join('')); |
| 114 | const rule = (title) => out.push(` ${title}`); |
| 115 | |
| 116 | row('', arms.map((a) => a.label)); |
| 117 | row('runs', arms.map((a) => a.runs.length)); |
| 118 | const anyFailed = arms.some((a) => a.runs.some((r) => !r.ok)); |
| 119 | if (anyFailed) row(' of which non-success', arms.map((a) => a.runs.filter((r) => !r.ok).length)); |
| 120 | if (arms.some((a) => a.runs.some((r) => r.raced))) { |
| 121 | row(' MCP cold-start race', arms.map((a) => a.runs.filter((r) => r.raced).length)); |
| 122 | } |
| 123 | out.push(''); |
| 124 | |
| 125 | rule('behavior'); |
| 126 | row(' duration (s)', arms.map((a) => span(a.runs, (r) => r.dur))); |
| 127 | row(' tool calls', arms.map((a) => span(a.runs, (r) => r.tools))); |
| 128 | row(' Read', arms.map((a) => span(a.runs, (r) => r.reads))); |
| 129 | row(' Grep/Glob', arms.map((a) => span(a.runs, (r) => r.grep))); |
| 130 | row(' Bash', arms.map((a) => span(a.runs, (r) => r.bash))); |
| 131 | row(' codegraph calls', arms.map((a) => span(a.runs, (r) => r.cg))); |
| 132 | out.push(''); |
| 133 | |
| 134 | rule('residual context occupancy (CG-7) — tokens still resident at end of run'); |
| 135 | row(' final context (tok)', arms.map((a) => span(a.runs, (r) => r.ctx, int))); |
| 136 | row(' codegraph residual (tok)', arms.map((a) => span(a.runs, (r) => r.occCg, int))); |
| 137 | row(' file-access residual (tok)', arms.map((a) => span(a.runs, (r) => r.occFile, int))); |
| 138 | row(' → retrieval residual (tok)', arms.map((a) => span(a.runs, (r) => r.occSelf, int))); |
| 139 | row(' → share of final context', arms.map((a) => span(a.runs, (r) => r.occShare, pct1))); |
| 140 | out.push(''); |
| 141 | |
| 142 | rule('explore sufficiency (CG-8) — pooled over every answered explore call'); |
| 143 | row(' answered explore calls', arms.map((a) => a.runs.reduce((s, r) => s + r.suffAnswered, 0))); |
| 144 | for (const [key, label] of SUFFICIENCY) { |
| 145 | row(` ${label}`, arms.map((a) => { |
| 146 | const n = a.runs.reduce((s, r) => s + r.suffCounts[key], 0); |
| 147 | const tot = a.runs.reduce((s, r) => s + r.suffAnswered, 0); |
| 148 | return tot ? `${n} ${((n / tot) * 100).toFixed(0)}%` : '—'; |
| 149 | })); |
| 150 | } |
| 151 | if (arms.some((a) => a.runs.some((r) => r.suffErrors))) { |
| 152 | row(' errored/unanswered (not bucketed)', arms.map((a) => a.runs.reduce((s, r) => s + r.suffErrors, 0))); |
| 153 | } |
| 154 | out.push(''); |
| 155 | |
| 156 | rule('explore allocation efficiency (CG-9) — share of returned bytes the answer cited'); |
| 157 | row(' explore calls with source', arms.map((a) => a.runs.reduce((s, r) => s + r.allocCalls, 0))); |
| 158 | row(' pooled efficiency', arms.map((a) => { |
| 159 | const env = a.runs.reduce((s, r) => s + (r.allocEnvelope || 0), 0); |
| 160 | const used = a.runs.reduce((s, r) => s + (r.allocUsed || 0), 0); |
| 161 | return env ? pct1((used / env) * 100) : '—'; |
| 162 | })); |
| 163 | row(' per-run efficiency', arms.map((a) => |
| 164 | span(a.runs, (r) => (r.allocEnvelope ? (r.allocUsed / r.allocEnvelope) * 100 : null), pct1))); |