Repository navigation
Expand file tree
/
Copy pathscore.ts
More file actions
90 lines (82 loc) · 2.59 KB
/
Copy pathscore.ts
File metadata and controls
90 lines (82 loc) · 2.59 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
import type { Investigation } from "@/lib/schema";
/**
* Pure scoring for the investigator eval, no I/O, unit-tested.
*
* Two views, because "accuracy" alone hides the failure that matters in AP:
* • overall accuracy, did the recommendation match the expected label?
* • overcharge precision / recall, treating "likely_overcharge" as the
* positive class. Recall = of the invoices that SHOULD be pushed back on, how
* many did the agent catch; precision = of the ones it flagged, how many were
* real. Missing a real overcharge (low recall) is the expensive error.
*/
export type Recommendation = Investigation["recommendation"];
export type CaseScore = {
id: string;
stresses: string;
expected: Recommendation;
/** undefined when the agent produced nothing / errored (a hard failure). */
got: Recommendation | undefined;
correct: boolean;
failed?: string;
};
export type Confusion = {
truePositives: number;
falsePositives: number;
falseNegatives: number;
precision: number;
recall: number;
f1: number;
};
const POSITIVE: Recommendation = "likely_overcharge";
/** Score one case: did `got` match `expected`? */
export const scoreCase = (
id: string,
stresses: string,
expected: Recommendation,
got: Recommendation | undefined,
failed?: string
): CaseScore => {
return {
id,
stresses,
expected,
got,
correct: got !== undefined && got === expected,
failed,
};
};
/** Overall accuracy across the scored cases (failed cases count as incorrect). */
export const accuracy = (scores: CaseScore[]): number => {
if (scores.length === 0) return 0;
const correct = scores.filter((s) => s.correct).length;
return correct / scores.length;
};
/**
* Precision / recall / F1 for the "likely_overcharge" positive class.
* TP, expected overcharge, got overcharge
* FP, got overcharge, expected something else
* FN, expected overcharge, got something else (or failed)
*/
export const overchargeConfusion = (scores: CaseScore[]): Confusion => {
let tp = 0;
let fp = 0;
let fn = 0;
for (const s of scores) {
const isGotPos = s.got === POSITIVE;
const isExpPos = s.expected === POSITIVE;
if (isExpPos && isGotPos) tp++;
else if (!isExpPos && isGotPos) fp++;
else if (isExpPos && !isGotPos) fn++;
}
const precision = tp + fp === 0 ? 1 : tp / (tp + fp);
const recall = tp + fn === 0 ? 1 : tp / (tp + fn);
const f1 = precision + recall === 0 ? 0 : (2 * precision * recall) / (precision + recall);
return {
truePositives: tp,
falsePositives: fp,
falseNegatives: fn,
precision,
recall,
f1,
};
};