Repository navigation
Expand file tree
/
Copy pathutils.ts
More file actions
233 lines (214 loc) · 6.86 KB
/
Copy pathutils.ts
File metadata and controls
233 lines (214 loc) · 6.86 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
/**
* This file provides utility functions and classes to assist with evaluation tasks.
*
* Key functionalities:
* - String normalization and fuzzy comparison utility functions to compare output strings
* against expected results in a flexible and robust way.
* - Generation of unique experiment names based on the current timestamp, environment,
* and eval name or category.
*/
import fs from "fs";
import { LogLine } from "stagehand-v3";
import stringComparison from "string-comparison";
import type { AgentModelEntry } from "./types/evals.js";
import { inferDefaultStagehandAgentMode } from "./framework/agentModelModes.js";
const { jaroWinkler } = stringComparison;
/**
* normalizeString:
* Prepares a string for comparison by:
* - Converting to lowercase
* - Collapsing multiple spaces to a single space
* - Removing punctuation and special characters that are not alphabetic or numeric
* - Normalizing spacing around commas
* - Trimming leading and trailing whitespace
*
* This helps create a stable string representation to compare against expected outputs,
* even if the actual output contains minor formatting differences.
*/
export function normalizeString(str: string): string {
return str
.toLowerCase()
.replace(/\s+/g, " ")
.replace(/[;/#!$%^&*:{}=\-_`~()]/g, "")
.replace(/\s*,\s*/g, ", ")
.trim();
}
/**
* compareStrings:
* Compares two strings (actual vs. expected) using a similarity metric (Jaro-Winkler).
*
* Arguments:
* - actual: The actual output string to be checked.
* - expected: The expected string we want to match against.
* - similarityThreshold: A number between 0 and 1. Default is 0.85.
* If the computed similarity is greater than or equal to this threshold,
* we consider the strings sufficiently similar.
*
* Returns:
* - similarity: A number indicating how similar the two strings are.
* - meetsThreshold: A boolean indicating if the similarity meets or exceeds the threshold.
*
* This function is useful for tasks where exact string matching is too strict,
* allowing for fuzzy matching that tolerates minor differences in formatting or spelling.
*/
export function compareStrings(
actual: string,
expected: string,
similarityThreshold: number = 0.85,
): { similarity: number; meetsThreshold: boolean } {
const similarity = jaroWinkler.similarity(normalizeString(actual), normalizeString(expected));
return {
similarity,
meetsThreshold: similarity >= similarityThreshold,
};
}
/**
* generateTimestamp:
* Generates a timestamp string formatted as "YYYYMMDDHHMMSS".
* Used to create unique experiment names, ensuring that results can be
* distinguished by the time they were generated.
*/
export function generateTimestamp(): string {
const now = new Date();
return now
.toISOString()
.replace(/[-:TZ]/g, "")
.slice(0, 14);
}
/**
* generateExperimentName:
* Returns just the target label. Braintrust handles uniqueness via IDs.
* All context (env, tool, startup) goes into experiment metadata instead.
*/
export function generateExperimentName({
evalName,
category,
}: {
evalName?: string;
category?: string;
environment?: string;
toolSurface?: string;
startupProfile?: string;
}): string {
if (evalName) return evalName;
if (category) return category;
return "all";
}
function clipLogLine(line: string): string {
const terminalWidth = process.stdout.columns;
const maxWidth = typeof terminalWidth === "number" && terminalWidth > 8 ? terminalWidth - 1 : 119;
if (line.length <= maxWidth) {
return line;
}
return `${line.slice(0, maxWidth - 1)}…`;
}
function clipLogOutput(output: string): string {
return output
.split("\n")
.map((line) => clipLogLine(line))
.join("\n");
}
export function logLineToString(logLine: LogLine): string {
try {
const timestamp = logLine.timestamp || new Date().toISOString();
if (logLine.auxiliary?.error) {
const errorValue = logLine.auxiliary.error?.value ?? "";
const traceValue = logLine.auxiliary.trace?.value ?? "";
const traceSuffix = traceValue ? `\n ${traceValue}` : "";
return clipLogOutput(
`${timestamp}::[stagehand:${logLine.category}] ${logLine.message}\n ${errorValue}${traceSuffix}`,
);
}
return clipLogOutput(
`${timestamp}::[stagehand:${logLine.category}] ${logLine.message} ${
logLine.auxiliary ? JSON.stringify(logLine.auxiliary) : ""
}`,
);
} catch (error) {
console.error(`Error logging line:`, error);
return "error logging line";
}
}
export function dedent(strings: TemplateStringsArray, ...values: unknown[]): string {
// Interleave raw strings with substitution values
const raw = strings.raw;
let result = "";
for (let i = 0; i < raw.length; i++) {
result += raw[i]
// replace newline + any mix of spaces/tabs with “\n”
.replace(/\n[ \t]+/g, "\n")
.replace(/^\n/, ""); // remove leading newline
if (i < values.length) result += values[i];
}
// trim trailing/leading blank lines
return result.trimEnd();
}
// Dataset helpers shared by suites
export function sampleUniform<T>(arr: T[], k: number): T[] {
const n = arr.length;
if (k >= n) return arr.slice();
const copy = arr.slice();
for (let i = n - 1; i > 0; i--) {
const j = Math.floor(Math.random() * (i + 1));
const tmp = copy[i];
copy[i] = copy[j];
copy[j] = tmp;
}
return copy.slice(0, k);
}
export function readJsonlFile(filePath: string): string[] {
let lines: string[];
try {
const content = fs.readFileSync(filePath, "utf-8");
lines = content.split(/\r?\n/).filter((l) => l.trim().length > 0);
} catch (e) {
console.warn(
`Could not read file at ${filePath}. Error: ${e instanceof Error ? e.message : String(e)}`,
);
lines = [];
}
return lines;
}
export function parseJsonlRows<T>(
lines: string[],
validator: (parsed: unknown) => parsed is T,
): T[] {
const candidates: T[] = [];
for (const line of lines) {
try {
const parsed = JSON.parse(line);
if (validator(parsed)) {
candidates.push(parsed);
}
} catch {
// skip invalid lines
}
}
return candidates;
}
export function applySampling<T>(
candidates: T[],
sampleCount?: number,
maxCases: number = 25,
): T[] {
if (sampleCount && sampleCount > 0) {
return sampleUniform(candidates, sampleCount);
} else {
const result: T[] = [];
for (const candidate of candidates) {
result.push(candidate);
if (result.length >= maxCases) break;
}
return result;
}
}
export function normalizeAgentModelEntries(
models: string[] | AgentModelEntry[],
): AgentModelEntry[] {
if (models.length === 0) return [];
if (typeof models[0] !== "string") return models as AgentModelEntry[];
return (models as string[]).map((modelName) => {
const mode = inferDefaultStagehandAgentMode(modelName);
return { modelName, mode, cua: mode === "cua" };
});
}