mirror of
https://github.com/langchain-ai/langgraph.git
synced 2026-09-06 09:47:51 +02:00
1.3 MiB
1.3 MiB
In [1]:
# %pip install -U --quiet langchain langchain_core langchain_openaiIn [19]:
# Optional: add tracing to visualize the agent trajectories
import os
from getpass import getpass
def _getpass(env_var: str):
if not os.environ.get(env_var):
os.environ[env_var] = getpass(f"{env_var}=")
os.environ["LANGCHAIN_TRACING_V2"] = "true"
os.environ["LANGCHAIN_PROJECT"] = "Web-Voyager"
_getpass("LANGCHAIN_API_KEY")
_getpass("OPENAI_API_KEY")In [2]:
# %pip install --upgrade --quiet playwright > /dev/null
# !playwright installIn [3]:
import nest_asyncio
# This is just required for running async playwright in a Jupyter notebook
nest_asyncio.apply()In [ ]:
from typing import List, Optional, TypedDict
from langchain_core.messages import BaseMessage, SystemMessage
from playwright.async_api import Page
class BBox(TypedDict):
x: float
y: float
class Prediction(TypedDict):
action: str
args: Optional[List[str]]
# This represents the state of the agent
# as it proceeds through execution
class AgentState(TypedDict):
page: Page # The Playwright web page lets us interact with the web environment
input: str # User request
img: str # b64 encoded screenshot
bboxes: List[BBox] # The bounding boxes from the browser annotation function
prediction: Prediction # The Agent's output
# A system message (or messages) containing the intermediate steps
scratchpad: List[BaseMessage]
observation: str # The most recent response from a toolIn [4]:
import asyncio
import platform
async def click(state: AgentState):
# - Click [Numerical_Label]
page = state["page"]
click_args = state["prediction"]["args"]
if click_args is None or len(click_args) != 1:
return f"Failed to click bounding box labeled as number {click_args}"
bbox_id = click_args[0]
bbox_id = int(bbox_id)
bbox = state["bboxes"][bbox_id - 1] # 1-indexed
x, y = bbox["x"], bbox["y"]
res = await page.mouse.click(x, y)
# TODO: In the paper, they automatically parse any downloaded PDFs
# We could add something similar here as well and generally
# improve response format.
return f"Clicked {bbox_id}"
async def type_text(state: AgentState):
page = state["page"]
type_args = state["prediction"]["args"]
if type_args is None or len(type_args) != 2:
return (
f"Failed to type in element from bounding box labeled as number {type_args}"
)
bbox_id = type_args[0]
bbox_id = int(bbox_id)
bbox = state["bboxes"][bbox_id] # - 1] # 1-indexed
x, y = bbox["x"], bbox["y"]
text_content = type_args[1]
await page.mouse.click(x, y)
# Check if MacOS
select_all = "Meta+A" if platform.system() == "Darwin" else "Control+A"
await page.keyboard.press(select_all)
await page.keyboard.press("Backspace")
await page.keyboard.type(text_content)
await page.keyboard.press("Enter")
return f"Typed {text_content} and submitted"
async def scroll(state: AgentState):
page = state["page"]
scroll_args = state["prediction"]["args"]
if scroll_args is None or len(scroll_args) != 2:
return "Failed to scroll due to incorrect arguments."
target, direction = scroll_args
if target.upper() == "WINDOW":
# Not sure the best value for this:
scroll_amount = 500
scroll_direction = (
-scroll_amount if direction.lower() == "up" else scroll_amount
)
await page.evaluate(f"window.scrollBy(0, {scroll_direction})")
else:
# Scrolling within a specific element
scroll_amount = 200
target_id = int(target)
bbox = state["bboxes"][target_id - 1] # 1-indexed
x, y = bbox["x"], bbox["y"]
scroll_direction = (
-scroll_amount if direction.lower() == "up" else scroll_amount
)
await page.mouse.move(x, y)
await page.mouse.wheel(0, scroll_direction)
return f"Scrolled {direction} in {'window' if target.upper() == 'WINDOW' else 'element'}"
async def wait(state: AgentState):
sleep_time = 5
asyncio.sleep(sleep_time)
return f"Waited for {sleep_time}s."
async def go_back(state: AgentState):
page = state["page"]
await page.go_back()
return f"Navigated back a page to {page.url}."
async def to_google(state: AgentState):
page = state["page"]
await page.goto("https://www.google.com/")
return "Navigated to google.com."In [5]:
import asyncio
import base64
from IPython import display
from langchain_core.runnables import chain as chain_decorator
# Some javascript we will run on each step
# to take a screenshot of the page, select the
# elements to annotate, and add bounding boxes
with open("mark_page.js") as f:
mark_page_script = f.read()
@chain_decorator
async def mark_page(page):
await page.evaluate(mark_page_script)
for _ in range(10):
try:
bboxes = await page.evaluate("markPage()")
break
except:
# May be loading...
asyncio.sleep(3)
screenshot = await page.screenshot()
# Ensure the bboxes don't follow us around
await page.evaluate("unmarkPage()")
return {
"img": base64.b64encode(screenshot).decode(),
"bboxes": bboxes,
}In [6]:
from langchain import hub
from langchain_core.output_parsers import StrOutputParser
from langchain_core.prompts import ChatPromptTemplate, MessagesPlaceholder
from langchain_core.runnables import RunnablePassthrough
from langchain_openai import ChatOpenAI
async def annotate(state):
marked_page = await mark_page.with_retry().ainvoke(state["page"])
return {**state, **marked_page}
def parse(text: str) -> dict:
action_prefix = "Action: "
if not text.strip().split("\n")[-1].startswith(action_prefix):
return {"action": "retry", "args": f"Could not parse LLM Output: {text}"}
action_block = text.strip().split("\n")[-1]
action_str = action_block[len(action_prefix) :]
split_output = action_str.split(" ", 1)
if len(split_output) == 1:
action, action_input = split_output[0], None
else:
action, action_input = split_output
action = action.strip()
if action_input is not None:
action_input = [
inp.strip().strip("[]") for inp in action_input.strip().split(";")
]
return {"action": action, "args": action_input}
# Will need a later version of langchain to pull
# this image prompt template
prompt = hub.pull("wfh/web-voyager")In [7]:
llm = ChatOpenAI(model="gpt-4-vision-preview", max_tokens=4096)
agent = annotate | RunnablePassthrough.assign(
prediction=prompt | llm | StrOutputParser() | parse
)In [ ]:
def update_scratchpad(state: AgentState):
"""After a tool is invoked, we want to update
the scratchpad so the agent is aware of its previous steps"""
old = state.get("scratchpad")
if old:
txt = old[0].content
last_line = txt.rsplit("\n", 1)[-1]
step = int(re.match(r"\d+", last_line).group()) + 1
else:
txt = "Previous action observations:\n"
step = 1
txt += f"\n{step}. {state['observation']}"
return {**state, "scratchpad": [SystemMessage(content=txt)]}In [9]:
from langchain_core.runnables import RunnableLambda
from langgraph.graph import END, StateGraph
graph_builder = StateGraph(AgentState)
graph_builder.add_node("agent", agent)
graph_builder.set_entry_point("agent")
graph_builder.add_node("update_scratchpad", update_scratchpad)
graph_builder.add_edge("update_scratchpad", "agent")
tools = {
"Click": click,
"Type": type_text,
"Scroll": scroll,
"Wait": wait,
"GoBack": go_back,
"Google": to_google,
}
for node_name, tool in tools.items():
graph_builder.add_node(
node_name,
# The lambda ensures the function's string output is mapped to the "observation"
# key in the AgentState
RunnableLambda(tool) | (lambda observation: {"observation": observation}),
)
# Always return to the agent (by means of the update-scratchpad node)
graph_builder.add_edge(node_name, "update_scratchpad")
def select_tool(state: AgentState):
# Any time the agent completes, this function
# is called to route the output to a tool or
# to the end user.
action = state["prediction"]["action"]
if action == "ANSWER":
return END
if action == "retry":
return "agent"
return action
graph_builder.add_conditional_edges("agent", select_tool)
graph = graph_builder.compile()In [11]:
from langchain_community.tools.playwright.utils import (
create_async_playwright_browser, # A synchronous browser is available, though it isn't compatible with jupyter.\n", },
)
browser = create_async_playwright_browser(headless=False)
page = await browser.new_page()
_ = await page.goto("https://www.google.com")
async def call_agent(question: str, page, max_steps: int = 150):
event_stream = graph.astream(
{
"page": page,
"input": question,
"scratchpad": [],
},
{
"recursion_limit": max_steps,
},
)
final_answer = None
steps = []
async for event in event_stream:
# We'll display an event stream here
if "agent" not in event:
continue
pred = event["agent"].get("prediction") or {}
action = pred.get("action")
action_input = pred.get("args")
display.clear_output(wait=False)
steps.append(f"{len(steps) + 1}. {action}: {action_input}")
print("\n".join(steps))
display.display(display.Image(base64.b64decode(event["agent"]["img"])))
if "ANSWER" in action:
final_answer = action_input[0]
break
return final_answerIn [14]:
res = await call_agent(
"Could you explain the multimodal Web Voyager paper (on arxiv)?", page
)
print(f"Final response: {res}")1. Google: None 2. Type: ['6', 'Web Voyager paper arxiv'] 3. Click: ['24'] 4. ANSWER;: ['The Web Voyager paper introduces "Voyager," an AI agent powered by large language models (LLMs), specifically designed for the game Minecraft. It\'s an open-ended, embodied agent that can explore the game world, learn a variety of skills, and make discoveries autonomously. The Voyager agent comprises three main components: an automated curriculum for exploration, a skill library for storing and retrieving complex behaviors, and an iterative prompting mechanism which integrates feedback from the environment, detects execution errors, and uses self-verification for program improvement. Voyager leverages GPT-4 for interactions and does not require fine-tuning of model parameters. The paper claims that Voyager has strong learning abilities in-context, can handle more unique items, travel longer distances, and achieve technological milestones faster than previous state-of-the-art systems. It also emphasizes the agent\'s compositional, interpretable, and temporally extended skills, which help rapid capability development and prevent forgetting. The agent\'s proficiency is highlighted in its exceptional performance in a new Minecraft world and its ability to generalize skills to solve novel tasks.']
Final response: The Web Voyager paper introduces "Voyager," an AI agent powered by large language models (LLMs), specifically designed for the game Minecraft. It's an open-ended, embodied agent that can explore the game world, learn a variety of skills, and make discoveries autonomously. The Voyager agent comprises three main components: an automated curriculum for exploration, a skill library for storing and retrieving complex behaviors, and an iterative prompting mechanism which integrates feedback from the environment, detects execution errors, and uses self-verification for program improvement. Voyager leverages GPT-4 for interactions and does not require fine-tuning of model parameters. The paper claims that Voyager has strong learning abilities in-context, can handle more unique items, travel longer distances, and achieve technological milestones faster than previous state-of-the-art systems. It also emphasizes the agent's compositional, interpretable, and temporally extended skills, which help rapid capability development and prevent forgetting. The agent's proficiency is highlighted in its exceptional performance in a new Minecraft world and its ability to generalize skills to solve novel tasks.
In [13]:
res = await call_agent(
"Please explain the today's XKCD comic for me. Why is it funny?", page
)
print(f"Final response: {res}")1. Google: None 2. Type: ['6', "today's XKCD comic"] 3. Click: ['24'] 4. GoBack: None 5. Click: ['27'] 6. ANSWER;: ['The XKCD comic presents a timeline showing significant historical events related to the understanding of the greenhouse effect and industrial activity. The humor lies in the observation that scientists figured out the greenhouse effect almost as early as the beginning of the Industrial Revolution, yet despite this early awareness, effective action on climate change has been slow. This contrast between early knowledge and delayed response is presented in a light-hearted way to highlight the irony of the situation.']
Final response: The XKCD comic presents a timeline showing significant historical events related to the understanding of the greenhouse effect and industrial activity. The humor lies in the observation that scientists figured out the greenhouse effect almost as early as the beginning of the Industrial Revolution, yet despite this early awareness, effective action on climate change has been slow. This contrast between early knowledge and delayed response is presented in a light-hearted way to highlight the irony of the situation.
In [12]:
res = await call_agent("What are the latest blog posts from langchain?", page)
print(f"Final response: {res}")1. Type: ['7', 'latest blog posts from LangChain'] 2. Click: ['24'] 3. ANSWER;: ['The latest blog posts from LangChain are "OpenGPTs," "LangGraph: Multi-Agent Workflows," and "LangGraph."']
Final response: The latest blog posts from LangChain are "OpenGPTs," "LangGraph: Multi-Agent Workflows," and "LangGraph."
In [15]:
res = await call_agent(
"Search a one-way flight from New York To Rejkavik for"
" 1 Adult and analyse the price graph for the"
" next 2 months",
page,
)
print(f"Final response: {res}")1. Google: None 2. Type: ['6', 'one-way flight from New York to Reykjavik for 1 adult'] 3. Click: ['115'] 4. Scroll: ['WINDOW', 'down'] 5. Click: ['32'] 6. Click: ['33'] 7. Click: ['73'] 8. Click: ['1'] 9. Click: ['1'] 10. Click: ['3'] 11. Click: ['5'] 12. Click: ['33'] 13. Click: ['5'] 14. Click: ['33'] 15. Click: ['5'] 16. Click: ['33'] 17. Click: ['2'] 18. Click: ['3'] 19. Click: ['2'] 20. ANSWER;: ['The price graph displayed shows the cost of one-way flights from New York to Reykjavik for the upcoming months of May and June. Prices appear to start from around $412 and show some variation throughout the period, with several peaks that could indicate higher prices on specific days. There appears to be no clear trend of prices decreasing or increasing significantly over the two-month period, and no specific date is highlighted as the cheapest within the visible range on the graph. To find the absolute lowest price, one would typically click on individual bars on the graph to see the price for specific dates, but as this is not possible in this format, this general overview is provided based on the visible information.']
Final response: The price graph displayed shows the cost of one-way flights from New York to Reykjavik for the upcoming months of May and June. Prices appear to start from around $412 and show some variation throughout the period, with several peaks that could indicate higher prices on specific days. There appears to be no clear trend of prices decreasing or increasing significantly over the two-month period, and no specific date is highlighted as the cheapest within the visible range on the graph. To find the absolute lowest price, one would typically click on individual bars on the graph to see the price for specific dates, but as this is not possible in this format, this general overview is provided based on the visible information.
In [ ]:
