1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
|
#!/usr/bin/env ruby
require 'json'
require 'net/http'
require 'open3'
goal, start_url = ARGV
abort "usage: ruby agent.rb \"<goal>\" [start_url]" unless goal
def browser(*args)
out, _ = Open3.capture2e('bunx', 'agent-browser', *args)
out
end
def decide(goal, history, snapshot, refs)
questions = {
mode: {
type: 'choice',
instructions:
'Avoid repeating identical sequential calls based on history. Should we interact with an element on the page, or do something with the page/browser?',
criteria: {
click: 'Click one of the interactive elements listed in current_page',
tool: 'Run a browser action: screenshot, extract page text, open the start URL, go back, or finish'
}
},
tool: {
type: 'choice',
instructions: 'If not clicking, which tool call should we make?',
criteria: {
screenshot: 'Save a screenshot of the current page for the user to refer to later.',
extract: 'Read the full text of the current page; the text is added to history',
open: 'Open the start URL again',
back: 'Go back to the previous page',
done: 'The goal has been accomplished; stop'
}
}
}
questions[:element] = {
type: 'choice',
instructions: 'If clicking, which element should be clicked?',
criteria: refs
} unless refs.empty?
res =
Net::HTTP.post(
URI('https://api.typesafe.ai/v1/systemone'),
JSON.generate(
{
state: {
goal: goal,
history: history,
current_page: snapshot
},
model: 'jev-latest',
questions: questions
}
),
'Content-Type' => 'application/json',
'Authorization' => 'Bearer ' + ENV['API_KEY']
)
JSON.parse(res.body).fetch('answers')
end
history = []
browser('open', start_url) if start_url
100.times do |i|
snap = browser('snapshot', '-i')
refs = snap.scan(/^\s*-\s*(.+?)\s*\[.*?ref=(e\d+)\]/).first(255).to_h { |text, ref| [ref, text] }
a = decide(goal, history, snap, refs)
mode = a.dig('mode', 'choice')
puts "[#{i}] mode=#{mode} (#{a.dig('mode', 'confidence')}) tool=#{a.dig('tool', 'choice')} (#{a.dig('tool', 'confidence')}) element=#{a.dig('element', 'choice')} (#{a.dig('element', 'confidence')})"
if mode == 'click' && a['element']
ref = a['element']['choice']
browser('click', ref)
history << "clicked #{ref} (#{refs[ref]})"
else
case tool = a.dig('tool', 'choice')
when 'screenshot'
browser('screenshot', "step#{i}.png")
history << "screenshot saved for user to step#{i}.png"
when 'extract'
text = browser('get', 'text', 'body')[0, 4000].strip
history << "text captured for user: #{text}"
when 'back'
browser('back')
history << 'went back to previous page'
when 'open'
browser('open', start_url.to_s)
history << "opened start URL #{start_url}"
when 'done'
break
end
end
end
puts "\nSteps taken:"
history.each { |h| puts " - #{h}" }
|