Repository navigation
Expand file tree
/
Copy pathsteps.ts
More file actions
167 lines (154 loc) · 7.77 KB
/
Copy pathsteps.ts
File metadata and controls
167 lines (154 loc) · 7.77 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
/**
* What an agent did in a run, step by step: the published trace.
*
* node src/steps.ts results/<id> [transcripts/<id>]
*
* writes steps-<label>.jsonl next to every runs-<label>.jsonl from the full
* transcripts the workflow keeps as an artifact (for runs published before
* the runner wrote step logs itself).
*
* One line per run: { runId, steps: [{ tool, action, thought?, url?, output?, error? }] }.
* The action is what the agent issued, as WebJudge sees it (the shell command
* or the tool call); `thought` is the agent's text right before the call,
* `url` the page after it, `output` an excerpt of what the tool replied (its
* start and end) and `error` the start of a failed or denied call's reply.
* Long values are cut, so a full run of 700 traces stays a few MB in git.
*/
import fs from 'node:fs/promises'
import path from 'node:path'
/** one tool call as the harness saw it (screenshots.ts hook, steps.json) */
export interface Step { step: number, toolUseId?: string, tool: string, input: unknown, screenshot?: string, url?: string }
export interface TraceStep { tool: string, action: string, thought?: string, url?: string, output?: string, error?: string }
/** the action as the agent issued it: the shell command or the tool call, never the tool's reply */
export function actionText ({ tool, input }: { tool: string, input: unknown }) {
const command = (input as { command?: unknown })?.command
if (tool === 'Bash' && typeof command === 'string') {
return command
}
return `${tool.replace(/^mcp__browser__/, '')} ${JSON.stringify(input)}`
}
/** tool calls that aren't browser actions: loading the skill, reading its docs */
export const isBrowserAction = (step: { tool: string }) => !['Skill', 'Read', 'ToolSearch', 'TodoWrite'].includes(step.tool)
const cut = (text: string, max: number) => {
const clean = redact(text).trim()
return clean.length > max ? `${clean.slice(0, max)}…` : clean
}
/**
* Credentials that pages and tools leak into what the agent reads (a site's
* map or analytics key in a snapshot, a token in a URL): masked before
* anything is published, since step logs are committed and shown on the site.
*/
const SECRETS: RegExp[] = [
/\b[ps]k\.eyJ[\w-]+\.[\w-]+/g, // Mapbox
/\beyJ[\w-]{8,}\.eyJ[\w-]{8,}\.[\w-]+/g, // JWT
/\b(?:AKIA|ASIA)[0-9A-Z]{16}\b/g, // AWS access key id
/\bAIza[\w-]{35}/g, // Google API key
/\bgh[pousr]_[A-Za-z0-9]{36,}/g, // GitHub
/\bgithub_pat_\w{50,}/g,
/\bxox[abprs]-[\w-]{10,}/g, // Slack
/\b[rs]k_(?:live|test)_[A-Za-z0-9]{16,}/g, // Stripe
/\bsk-(?:ant-|proj-)?[\w-]{20,}/g, // Anthropic, OpenAI
/\bhf_[A-Za-z0-9]{30,}/g, // Hugging Face
/-----BEGIN [A-Z ]*PRIVATE KEY-----[\s\S]*?(?:-----END [A-Z ]*PRIVATE KEY-----|$)/g,
/(?<=[?&#](?:access_token|api_key|apikey|token|secret|password|sig|signature)=)[^&#\s"']{8,}/gi
]
export const redact = (text: string) => SECRETS.reduce((t, re) => t.replace(re, '[redacted]'), text)
/** the start and the end of a long reply: what the agent read first, and where a log or a snapshot ends */
const excerpt = (text: string, head: number, tail: number) => {
const clean = redact(text).trim()
return clean.length > head + tail ? `${clean.slice(0, head).trimEnd()}\n…\n${clean.slice(-tail).trimStart()}` : clean
}
interface Block { type: string, id?: string, name?: string, input?: unknown, text?: string, tool_use_id?: string, is_error?: boolean, content?: unknown }
interface Message { type: string, message?: unknown }
const blocksOf = (message: Message): Block[] => {
const content = (message.message as { content?: unknown } | undefined)?.content
return Array.isArray(content) ? content as Block[] : []
}
function replyText (content: unknown): string {
if (typeof content === 'string') {
return content
}
return Array.isArray(content) ? content.map((c: Block) => c.type === 'text' ? c.text ?? '' : '').join(' ') : ''
}
/**
* Every tool call in the transcript, in order, including the failed and
* denied ones the screenshot hook never sees; the page URL comes from the
* hook's step with the same tool use id.
*/
export function traceOf (transcript: Message[], hookSteps: Step[] = []): TraceStep[] {
const urls = new Map(hookSteps.filter((s) => s.toolUseId && s.url).map((s) => [s.toolUseId!, s.url!]))
const errors = new Map<string, string>()
const outputs = new Map<string, string>()
for (const message of transcript) {
if (message.type === 'user') {
for (const block of blocksOf(message)) {
if (block.type === 'tool_result' && block.tool_use_id) {
const reply = replyText(block.content)
if (block.is_error) {
errors.set(block.tool_use_id, reply)
} else if (reply.trim()) {
outputs.set(block.tool_use_id, reply)
}
}
}
}
}
const steps: TraceStep[] = []
let thought = ''
for (const message of transcript) {
if (message.type !== 'assistant') {
continue
}
for (const block of blocksOf(message)) {
if (block.type === 'text' && block.text) {
thought += (thought ? '\n' : '') + block.text
} else if (block.type === 'tool_use' && block.name) {
const id = block.id ?? ''
steps.push({
tool: block.name.replace(/^mcp__browser__/, ''),
action: cut(actionText({ tool: block.name, input: block.input }), 400),
...(thought.trim() && { thought: cut(thought, 400) }),
...(urls.has(id) && { url: cut(urls.get(id)!, 300) }),
...(outputs.has(id) && { output: excerpt(outputs.get(id)!, 600, 200) }),
...(errors.has(id) && { error: cut(errors.get(id)!, 200) })
})
thought = ''
}
}
}
return steps
}
/** steps-<label>.jsonl for every runs-<label>.jsonl in `resultDir`, from the transcripts in `transcriptDir` */
async function backfill (resultDir: string, transcriptDir: string) {
const files = (await fs.readdir(resultDir)).filter((f) => f.startsWith('runs-') && f.endsWith('.jsonl'))
let written = 0
let missing = 0
for (const file of files) {
const lines: string[] = []
for (const line of (await fs.readFile(path.join(resultDir, file), 'utf8')).split('\n').filter(Boolean)) {
const { runId } = JSON.parse(line) as { runId: string }
const raw = await fs.readFile(path.join(transcriptDir, `${runId}.jsonl`), 'utf8').catch(() => undefined)
if (raw === undefined) {
missing++
continue
}
const transcript = raw.split('\n').filter(Boolean).map((l) => JSON.parse(l) as Message)
const hookSteps: Step[] = JSON.parse(await fs.readFile(path.join(transcriptDir, runId, 'steps.json'), 'utf8').catch(() => '[]'))
lines.push(JSON.stringify({ runId, steps: traceOf(transcript, hookSteps) }))
written++
}
if (lines.length) {
await fs.writeFile(path.join(resultDir, `steps-${file.slice('runs-'.length)}`), lines.join('\n') + '\n')
}
}
console.log(`${path.basename(resultDir)}: step logs for ${written} run(s)${missing ? `, ${missing} without a transcript` : ''}`)
}
if (import.meta.filename === path.resolve(process.argv[1] ?? '')) {
const [dir, transcripts] = process.argv.slice(2)
if (!dir) {
console.error('usage: node src/steps.ts results/<id> [transcripts/<id>]')
process.exit(1)
}
const resultDir = path.resolve(dir)
await backfill(resultDir, path.resolve(transcripts ?? path.join(import.meta.dirname, '..', 'transcripts', path.basename(resultDir))))
}