From 076219352a62fca8e22b5adadcbcfb6618041622 Mon Sep 17 00:00:00 2001 From: Palash Shah <35114859+Palashio@users.noreply.github.com> Date: Thu, 7 Aug 2025 16:15:24 -0400 Subject: [PATCH] feat: update langbench to run extracted tests (#708) * feat: add test runner * fix: formatting * update: use json formatting * update: fix formatting * update: formatting * update: rename, and small mods * format: remove comment * update: function args * nit: format * fix: update snake case to camel case * update: modify interface * Update apps/open-swe/langbench/evaluator.eval.ts Co-authored-by: Brace Sproul * Update apps/open-swe/langbench/utils.ts Co-authored-by: Brace Sproul * Update apps/open-swe/langbench/utils.ts Co-authored-by: Brace Sproul * Update apps/open-swe/langbench/utils.ts Co-authored-by: Brace Sproul * nit: update commands * nit: update formatting * nit: update commands --------- Co-authored-by: Brace Sproul --- .../{runEvals.eval.ts => evaluator.eval.ts} | 95 +++++-- .../langbench/static/langgraph_prs.json | 86 ++++-- apps/open-swe/langbench/types.ts | 70 +++-- apps/open-swe/langbench/utils.ts | 245 ++++++++++++++++++ apps/open-swe/src/utils/github/git.ts | 47 ++++ 5 files changed, 489 insertions(+), 54 deletions(-) rename apps/open-swe/langbench/{runEvals.eval.ts => evaluator.eval.ts} (53%) create mode 100644 apps/open-swe/langbench/utils.ts diff --git a/apps/open-swe/langbench/runEvals.eval.ts b/apps/open-swe/langbench/evaluator.eval.ts similarity index 53% rename from apps/open-swe/langbench/runEvals.eval.ts rename to apps/open-swe/langbench/evaluator.eval.ts index 86780918..213db616 100644 --- a/apps/open-swe/langbench/runEvals.eval.ts +++ b/apps/open-swe/langbench/evaluator.eval.ts @@ -4,13 +4,15 @@ import { Daytona, Sandbox } from "@daytonaio/sdk"; import { createLogger, LogLevel } from "../src/utils/logger.js"; import { DEFAULT_SANDBOX_CREATE_PARAMS } from "../src/constants.js"; import { readFileSync } from "fs"; -import { cloneRepo } from "../src/utils/github/git.js"; +import { cloneRepo, checkoutFilesFromCommit } from "../src/utils/github/git.js"; import { TargetRepository } from "@open-swe/shared/open-swe/types"; import { getRepoAbsolutePath } from "@open-swe/shared/git"; import { setupEnv } from "../src/utils/env-setup.js"; import { PRData, PRProcessResult } from "./types.js"; +import { runPytestOnFiles } from "./utils.js"; dotenv.config(); + const logger = createLogger(LogLevel.INFO, "PR Processor"); // Load PRs data @@ -28,11 +30,12 @@ logger.info(`Starting evals over ${DATASET.length} PRs...`); */ async function processPR(prData: PRData): Promise { const result: PRProcessResult = { - pr_number: prData.pr_number, - repo_name: prData.repo_name, + prNumber: prData.prNumber, + repoName: prData.repoName, success: false, - evals_found: false, - evals_files: [], + evalsFound: false, + evalsFiles: [], + testFiles: [], }; const daytona = new Daytona({ organizationId: process.env.DAYTONA_ORGANIZATION_ID, @@ -40,22 +43,30 @@ async function processPR(prData: PRData): Promise { let sandbox: Sandbox | undefined; try { - logger.info(`Processing PR #${prData.pr_number}: ${prData.title}`); + logger.info(`Processing PR #${prData.prNumber}: ${prData.title}`); + // Use test files from PR data (already fetched and stored) + const testFiles = prData.testFiles || []; + result.testFiles = testFiles; // Create sandbox sandbox = await daytona.create(DEFAULT_SANDBOX_CREATE_PARAMS); - result.workspace_id = sandbox.id; + // Validate sandbox was created properly + if (!sandbox || !sandbox.id) { + throw new Error("Failed to create valid sandbox"); + } + + result.workspaceId = sandbox.id; logger.info(`Created sandbox: ${sandbox.id}`); // Use the hardcoded pre-merge commit SHA from the dataset - const preMergeCommit = prData.pre_merge_commit_sha; + const preMergeCommit = prData.preMergeCommitSha; logger.info(`Using pre-merge commit: ${preMergeCommit}`); - result.pre_merge_sha = preMergeCommit; + result.preMergeSha = preMergeCommit; const targetRepository: TargetRepository = { - owner: prData.repo_owner, - repo: prData.repo_name, + owner: prData.repoOwner, + repo: prData.repoName, branch: undefined, baseCommit: preMergeCommit, }; @@ -78,11 +89,47 @@ async function processPR(prData: PRData): Promise { logger.warn("Failed to setup Python environment, continuing anyway"); } + // Checkout test files from the merge commit to get the updated test files + if (testFiles.length > 0) { + logger.info( + `Checking out test files from merge commit: ${prData.mergeCommitSha}`, + ); + await checkoutFilesFromCommit({ + sandbox, + repoDir, + commitSha: prData.mergeCommitSha, + filePaths: testFiles, + }); + } + + // Run tests on detected test files + if (testFiles.length > 0) { + logger.info( + `Running pytest on ${testFiles.length} detected test files...`, + ); + const testResults = await runPytestOnFiles({ + sandbox, + testFiles, + repoDir, + timeoutSec: 300, + }); + result.testResults = testResults; + + logger.info(`Test execution completed for PR #${prData.prNumber}`, { + totalTests: testResults.totalTests, + passedTests: testResults.passedTests, + failedTests: testResults.failedTests, + success: testResults.success, + }); + } else { + logger.info(`No test files to run for PR #${prData.prNumber}`); + } + result.success = true; - logger.info(`Successfully processed PR #${prData.pr_number}`); + logger.info(`Successfully processed PR #${prData.prNumber}`); } catch (error) { result.error = error instanceof Error ? error.message : String(error); - logger.error(`Failed to process PR #${prData.pr_number}:`, { error }); + logger.error(`Failed to process PR #${prData.prNumber}:`, { error }); } finally { // Cleanup sandbox if (sandbox) { @@ -104,18 +151,28 @@ ls.describe(DATASET_NAME, () => { ls.test.each(DATASET)( "Can process PR successfully", async ({ inputs: prData }) => { - logger.info(`Processing PR #${prData.pr_number}: ${prData.title}`); + logger.info(`Processing PR #${prData.prNumber}: ${prData.title}`); const result = await processPR(prData); // Log results for visibility - logger.info(`PR #${prData.pr_number} processing completed`, { + logger.info(`PR #${prData.prNumber} processing completed`, { success: result.success, - evals_found: result.evals_found, - evals_files_count: result.evals_files.length, + evalsFound: result.evalsFound, + evalsFilesCount: result.evalsFiles.length, + testFilesCount: result.testFiles.length, + testFiles: result.testFiles, + testResults: result.testResults + ? { + totalTests: result.testResults.totalTests, + passedTests: result.testResults.passedTests, + failedTests: result.testResults.failedTests, + success: result.testResults.success, + } + : null, error: result.error, - workspace_id: result.workspace_id, - pre_merge_sha: result.pre_merge_sha, + workspaceId: result.workspaceId, + preMergeSha: result.preMergeSha, }); // Assert that processing was successful diff --git a/apps/open-swe/langbench/static/langgraph_prs.json b/apps/open-swe/langbench/static/langgraph_prs.json index f75b294e..5cfb9511 100644 --- a/apps/open-swe/langbench/static/langgraph_prs.json +++ b/apps/open-swe/langbench/static/langgraph_prs.json @@ -12,7 +12,13 @@ "body": "## Overview\r\n\r\nThis PR introduces a new API that provides a cleaner, more type-safe way to pass runtime context to LangGraph nodes/tasks. It replaces the current pattern of using `config['configurable']` and `config_schema` with a dedicated `context` parameter and wrapper `Runtime` object.\r\n\r\n## What's Changed\r\n\r\n### Before/After: Basic Context Usage\r\n\r\n### Before (Old Pattern)\r\n```python\r\nfrom langchain_core.runnables import RunnableConfig\r\n\r\ndef node(state: State, config: RunnableConfig):\r\n user_id = config.get(\"configurable\", {}).get(\"user_id\")\r\n return {\"result\": f\"Hello {user_id}\"}\r\n\r\ngraph.invoke(input_data, config={\"configurable\": {\"user_id\": \"123\"}})\r\n```\r\n\r\n### After (New Pattern)\r\n```python\r\nfrom dataclasses import dataclass\r\nfrom langgraph.runtime import Runtime\r\n\r\n@dataclass\r\nclass ContextSchema:\r\n user_id: str\r\n\r\ndef node(state: State, runtime: Runtime[ContextSchema]):\r\n user_id = runtime.context.user_id\r\n return {\"result\": f\"Hello {user_id}\"}\r\n\r\ngraph.invoke(input_data, context={\"user_id\": \"123\"})\r\n```\r\n\r\nOR, you can use the `get_runtime` method:\r\n```python\r\nfrom langgraph.runtime import get_runtime\r\n\r\ndef node(state: State):\r\n user_id = get_runtime(ContextSchema).context.user_id\r\n return {\"result\": f\"Hello {user_id}\"}\r\n```\r\n\r\n
\r\nBefore/After: Store and Stream Writer Access\r\n\r\n### Before (Old pattern)\r\n```py\r\nfrom langgraph.store.base import BaseStore\r\nfrom langchain_core.runnables import RunnableConfig\r\n\r\ndef update_memory(state: MessagesState, config: RunnableConfig, *, store: BaseStore):\r\n user_id = config.get(\"configurable\", {}).get(\"user_id\")\r\n namespace = (user_id, \"memories\")\r\n memory_id = str(uuid.uuid4())\r\n store.put(namespace, memory_id, {\"memory\": memory})\r\n```\r\n\r\n### After (new pattern)\r\n```py\r\nfrom langgraph.runtime import Runtime\r\n\r\ndef update_memory(state: MessagesState, runtime: Runtime[ContextSchema]):\r\n user_id = runtime.context.user_id\r\n namespace = (user_id, \"memories\")\r\n memory_id = str(uuid.uuid4())\r\n runtime.store.put(namespace, memory_id, {\"memory\": memory})\r\n```\r\n
\r\n\r\n## Key Benefits\r\n- **Type Safety**: Type checked `context` input to `invoke` / `stream`, plus typed access to `Runtime` attributes\r\n- **Cleaner API**: Direct `context` parameter instead of nested `config['configurable']`\r\n- **Better DX**: IDE autocomplete for context fields and `Runtime` attributes\r\n- **Unified Runtime**: Single `Runtime` object provides access to context, store, and stream writer, with room for expanding to streamlined config/checkpoint information in the near future.\r\n\r\n## Breaking Changes & Migration\r\n\r\n### Deprecated APIs\r\n- `StateGraph(..., config_schema=X)` -> `StateGraph(..., context_schema=X)`\r\n- `Pregel.config_schema` → `Pregel.get_context_jsonschema()` - this is largely meant to be external and we don't anticipate this affecting many users\r\n\r\n### Migration Details\r\n- Maintains backward compatibility with existing `config['configurable']` usage\r\n- Deprecation warnings guide users to the new API\r\n\r\n## Future Work\r\n\r\n- [ ] Deprecate injection pattern for `store`, `stream_writer`, maybe `previous`, update docs to recommend popping from runtime.\r\n- [ ] Support `Runtime` injection for tools, right now only `get_runtime` is supported\r\n- [ ] LangGraph style guide with recommended best practices\r\n\r\nEventually, I think we should move away from storing and popping things from `config[\"configurable\"]`, which can be a fully internal refactor. Though I would love to do this pre v1, it should largely be internal, so can be done afterwards. Config changes should be in a different PR than this one to keep things reasonably scoped. This PR is already pretty big. Lots of plumbing.\r\n\r\n## Related Issues\r\nCloses #5023\r\n", "created_at": "2025-06-28T01:04:08Z", "merged_at": "2025-07-15T13:20:20Z", - "pre_merge_commit_sha": "e0bf4a7bc35d9bb9b9c52a0c652446ff9c9734ba" + "pre_merge_commit_sha": "e0bf4a7bc35d9bb9b9c52a0c652446ff9c9734ba", + "test_files": [ + "libs/langgraph/tests/test_deprecation.py", + "libs/langgraph/tests/test_pregel.py", + "libs/langgraph/tests/test_runnable.py", + "libs/langgraph/tests/test_runtime.py" + ] }, { "url": "https://api.github.com/repos/langchain-ai/langgraph/pulls/4374", @@ -27,7 +33,12 @@ "body": "This PR does a few things:\r\n1. Surfaces interrupts when `stream_mode='values'` (particularly relevant for `invoke`, where this is the default behavior) \r\n2. Adds an `interrupt_id` property to the `Interrupt` dataclass so that interrupts can effectively be mapped to resumes\r\n3. Minor docs updates to reflect the new pattern (no need for a special section on interrupts with `invoke` and `ainvoke`)\r\n\r\n* In a different PR (the one with the multiple resume values), as it's more relevant there: add an `interrupts` property to `StateSnapshot` so that `interrupts` can easily be iterated over if users are attempting to map interrupts to resumes.\r\n\r\nI **don't** recommend we release this until we have multi-resumes working.\r\n\r\n## Example\r\n\r\nWe have the following setup where we're sending multiple prompts to the child graph, which uses `interrupt`:\r\n\r\n```py\r\ndef child_graph(state):\r\n human_input = interrupt(state[\"prompt\"])\r\n\r\n return {\r\n \"human_inputs\": [human_input],\r\n }\r\n```\r\n\r\n\"Screenshot\r\n\r\nOld behavior:\r\n\r\n```py\r\ninitial_input = {\"prompts\": [\"a\", \"b\"]}\r\n\r\nprint(parent_graph.invoke(input=initial_input,config=thread_config,stream_mode=\"values\"))\r\n#> {'prompts': ['a', 'b'], 'human_inputs': []}\r\n\r\nprint(parent_graph.invoke(Command(resume=\"hello 1\"),config=thread_config,stream_mode=\"values\"))\r\n#> {'prompts': ['a', 'b'], 'human_inputs': ['hello 1']}\r\n\r\nprint(parent_graph.invoke(Command(resume=\"hello 2\"),config=thread_config,stream_mode=\"values\"))\r\n#> {'prompts': ['a', 'b'], 'human_inputs': ['hello 1', 'hello 2']}\r\n```\r\n\r\nNew behavior:\r\n\r\n```py\r\ninitial_input = {\"prompts\": [\"a\", \"b\"]}\r\n\r\nprint(parent_graph.invoke(input=initial_input,config=thread_config,stream_mode=\"values\"))\r\n\"\"\"\r\n{\r\n \"prompts\": [\"a\", \"b\"],\r\n \"human_inputs\": [],\r\n \"__interrupt__\": [\r\n Interrupt(\r\n value=\"a\",\r\n resumable=True,\r\n ns=[\"child_graph:38d43a18-a5e7-8ab2-ca83-9d80f6e9ca83\"]\r\n ),\r\n Interrupt(\r\n value=\"b\",\r\n resumable=True,\r\n ns=[\"child_graph:dad810e8-738e-9f90-41cd-30c0091eb79b\"]\r\n )\r\n ]\r\n}\r\n\"\"\"\r\n\r\nprint(parent_graph.invoke(Command(resume=\"hello 1\"),config=thread_config,stream_mode=\"values\"))\r\n\"\"\"\r\n{\r\n \"prompts\": [\"a\", \"b\"],\r\n \"human_inputs\": [\"hello 1\"],\r\n \"__interrupt__\": [\r\n Interrupt(\r\n value=\"b\",\r\n resumable=True,\r\n ns=[\"child_graph:dad810e8-738e-9f90-41cd-30c0091eb79b\"]\r\n )\r\n ]\r\n}\r\n\"\"\"\r\n\r\nprint(parent_graph.invoke(Command(resume=\"hello 2\"),config=thread_config,stream_mode=\"values\"))\r\n#> {'prompts': ['a', 'b'], 'human_inputs': ['hello 1', 'hello 2']}\r\n```", "created_at": "2025-04-22T17:17:21Z", "merged_at": "2025-04-24T15:21:28Z", - "pre_merge_commit_sha": "b81c21f311fd131d5969c33d238be2eeed5cc522" + "pre_merge_commit_sha": "b81c21f311fd131d5969c33d238be2eeed5cc522", + "test_files": [ + "libs/langgraph/tests/test_large_cases.py", + "libs/langgraph/tests/test_pregel.py", + "libs/langgraph/tests/test_pregel_async.py" + ] }, { "url": "https://api.github.com/repos/langchain-ai/langgraph/pulls/3126", @@ -42,7 +53,8 @@ "body": "Alternative to https://github.com/langchain-ai/langgraph/pull/3124\r\n\r\nCurrently if a tool interrupts, the entire tool node executes again after resuming. So tools can get executed twice if parallel tool calls are generated. Here we allow ToolNode to accept tool calls, so we can use the `Send` API to distribute the tool calls to multiple instances of the tool node.\r\n\r\n```python\r\nfrom langchain_anthropic import ChatAnthropic\r\nfrom langchain_core.tools import tool\r\nfrom langgraph.checkpoint.memory import MemorySaver\r\nfrom langgraph.prebuilt import create_react_agent\r\nfrom langgraph.types import Command, Send, interrupt\r\n\r\n\r\n@tool\r\ndef human_assistance(query: str) -> str:\r\n \"\"\"Request assistance from a human.\"\"\"\r\n human_response = interrupt({\"query\": query})\r\n return human_response[\"data\"]\r\n\r\n\r\n@tool\r\ndef get_weather(location: str) -> str:\r\n \"\"\"Use this tool to get the weather.\"\"\"\r\n return \"It's sunny!\"\r\n\r\n\r\ntools = [get_weather, human_assistance]\r\nllm = ChatAnthropic(model=\"claude-3-5-sonnet-20240620\")\r\n\r\nagent = create_react_agent(\r\n llm,\r\n tools,\r\n checkpointer=MemorySaver(),\r\n tool_call_parallelism=\"parallel_tool_nodes\",\r\n)\r\n\r\n\r\nuser_input = (\r\n \"Could you please (1) request assistance for building an AI agent \"\r\n \"from a human, and (2) search for the weather in Boston, MA? \"\r\n \"Generate two tool calls at once.\"\r\n)\r\n\r\nconfig = {\"configurable\": {\"thread_id\": \"1\"}}\r\n\r\nfor event in agent.stream(\r\n {\"messages\": [{\"role\": \"user\", \"content\": user_input}]},\r\n config,\r\n stream_mode=\"values\",\r\n):\r\n event[\"messages\"][-1].pretty_print()\r\n```\r\n```\r\n...\r\n```\r\n```python\r\nhuman_response = \"You should check out LangGraph to build your agent.\"\r\nhuman_command = Command(resume={\"data\": human_response})\r\n\r\nfor event in agent.stream(human_command, config, stream_mode=\"values\"):\r\n event[\"messages\"][-1].pretty_print()\r\n```", "created_at": "2025-01-21T17:58:50Z", "merged_at": "2025-01-31T17:20:59Z", - "pre_merge_commit_sha": "4b3e07b67aa5a992531cab169286c3cda0c38a0a" + "pre_merge_commit_sha": "4b3e07b67aa5a992531cab169286c3cda0c38a0a", + "test_files": ["libs/langgraph/tests/test_prebuilt.py"] }, { "url": "https://api.github.com/repos/langchain-ai/langgraph/pulls/3095", @@ -57,7 +69,8 @@ "body": "- both issues are related to the fact that waiters for futures are notified of completion before \"done\" callbacks are called\r\n- 1st issue manifested as interrupt stream event being emitted before the result of a task that logically finished first (it's in the line above in body of the entrypoint function) -> this is solved by always returning to use code a fresh future chained on the original future, because chaining is done via done callbacks (therefore the chained future will only resolve after done callbacks of the original feature are called)\r\n- 2nd issue mainfested as sometimes (very rarely) the last stream event not being printed before stream() finishes. this is solved by ensuring we only return out of PregelRunner.tick() once all \"done\" callbacks are called, previously we were approximating this through use of asyncio.sleep(0) / time.sleep(0). The new solution instead waits on a threading/asyncio.Event which will only be set by the last \"done\" callback to fire\r\n- this PR also disables incomplete support for calling sync tasks from async entrypoints", "created_at": "2025-01-17T23:35:26Z", "merged_at": "2025-01-17T23:44:52Z", - "pre_merge_commit_sha": "e4a5c8fd28ceca30171073aebd64b802699c54ba" + "pre_merge_commit_sha": "e4a5c8fd28ceca30171073aebd64b802699c54ba", + "test_files": ["libs/langgraph/tests/test_pregel_async.py"] }, { "url": "https://api.github.com/repos/langchain-ai/langgraph/pulls/2848", @@ -72,7 +85,8 @@ "body": "```python\r\nclass WeatherResponse(BaseModel):\r\n \"\"\"Respond to the user with this\"\"\"\r\n\r\n temperature: float = Field(description=\"The temperature in fahrenheit\")\r\n wind_direction: str = Field(\r\n description=\"The direction of the wind in abbreviated form\"\r\n )\r\n wind_speed: float = Field(description=\"The speed of the wind in mph\")\r\n\r\n@tool\r\ndef get_weather(city: Literal[\"nyc\", \"sf\"]):\r\n \"\"\"Use this to get weather information.\"\"\"\r\n if city == \"nyc\":\r\n return \"It is cloudy in NYC, with 5 mph winds in the North-East direction and a temperature of 70 degrees\"\r\n elif city == \"sf\":\r\n return \"It is 75 degrees and sunny in SF, with 3 mph winds in the South-East direction\"\r\n else:\r\n raise AssertionError(\"Unknown city\")\r\n\r\nmodel = ChatOpenAI()\r\ntools = [get_weather]\r\nagent_with_structured_output = create_react_agent(model, tools, response_format=WeatherResponse)\r\nagent_with_structured_output.invoke({\"messages\": [(\"user\", \"what's the weather in nyc?\")]})\r\n```\r\n\r\n```pycon\r\n{\r\n 'messages': [...],\r\n 'structured_response': WeatherResponse(temperature=70.0, wind_directon='NE', wind_speed=5.0)\r\n}\r\n```", "created_at": "2024-12-20T20:31:20Z", "merged_at": "2025-01-10T16:06:59Z", - "pre_merge_commit_sha": "35c3ba0104804bee045675ee8bce754deccacfc2" + "pre_merge_commit_sha": "35c3ba0104804bee045675ee8bce754deccacfc2", + "test_files": ["libs/langgraph/tests/test_prebuilt.py"] }, { "url": "https://api.github.com/repos/langchain-ai/langgraph/pulls/1004", @@ -87,7 +101,8 @@ "body": "We noticed that it is currently not possible to interrupt a graph multiple times.\r\n\r\nOnce the graph resumes execution after an interruption, it just continues executing, ignoring `interrupt_before` and `interrupt_after`.\r\n\r\nThe reason is that after resuming this condition in `_should_interrupt` seems to always return false:\r\n```\r\nany(\r\n checkpoint[\"channel_versions\"].get(chan, null_version)\r\n > seen.get(chan, null_version)\r\n for chan in snapshot_channels\r\n)\r\n```\r\n\r\nIn this PR I added a unit test in `test_interruption.py` to spec the desired behavior. I modified the code to pass the unit test, but since I do not understand what this code does it's probably not the right thing.\r\n\r\nIt would be great to get some guidance on how to fix this properly.\r\n", "created_at": "2024-07-12T16:42:14Z", "merged_at": "2024-07-12T19:53:17Z", - "pre_merge_commit_sha": "558a513a1acbc0ae88ae24d6e5cc13325ab00ad1" + "pre_merge_commit_sha": "558a513a1acbc0ae88ae24d6e5cc13325ab00ad1", + "test_files": ["libs/langgraph/tests/test_interruption.py"] }, { "url": "https://api.github.com/repos/langchain-ai/langgraph/pulls/5801", @@ -102,7 +117,8 @@ "body": "Fixes https://github.com/langchain-ai/langgraph/issues/5784\r\n\r\n* Removes usage of `is_last_step`, no longer needed with `remaining_steps`\r\n* Make `remaining_steps` `NotRequired` so that json schema doesn't suggest need for user input\r\n* Move `PregelScratchpad` to shared utils file to prevent circular import issue (it's used from `channels/managed` and other pregel files).\r\n* Ensures that managed values wrapped in `NotRequired` or `Required` are still recognized!", "created_at": "2025-08-01T18:31:32Z", "merged_at": "2025-08-03T11:12:54Z", - "pre_merge_commit_sha": "db8ed4e9e424ed29c8165602f17ee800c671681b" + "pre_merge_commit_sha": "db8ed4e9e424ed29c8165602f17ee800c671681b", + "test_files": ["libs/langgraph/tests/test_managed_values.py"] }, { "url": "https://api.github.com/repos/langchain-ai/langgraph/pulls/5796", @@ -117,7 +133,8 @@ "body": "Fixes https://github.com/langchain-ai/langgraph/issues/5795\r\n\r\n* Must use `category=None` on decorator so that we get type checking support but no dupe warning\r\n* Fixed tuple on `config_type` warning causing false warning", "created_at": "2025-08-01T14:27:51Z", "merged_at": "2025-08-01T14:33:46Z", - "pre_merge_commit_sha": "38bbd92e01d8437b70c45355b6005fa40c204844" + "pre_merge_commit_sha": "38bbd92e01d8437b70c45355b6005fa40c204844", + "test_files": ["libs/langgraph/tests/test_deprecation.py"] }, { "url": "https://api.github.com/repos/langchain-ai/langgraph/pulls/5708", @@ -132,7 +149,8 @@ "body": "Fixes https://github.com/langchain-ai/langgraph/issues/5698\r\n\r\nI would like to do a more general refactor of this logic at some point as well using typing introspection utilities. Claude code first pass: https://github.com/langchain-ai/langgraph/pull/5709", "created_at": "2025-07-29T19:14:10Z", "merged_at": "2025-07-29T20:11:37Z", - "pre_merge_commit_sha": "479373bd81f538b81362ae8bea9d1b6923e1f60e" + "pre_merge_commit_sha": "479373bd81f538b81362ae8bea9d1b6923e1f60e", + "test_files": ["libs/langgraph/tests/test_runnable.py"] }, { "url": "https://api.github.com/repos/langchain-ai/langgraph/pulls/4983", @@ -147,7 +165,15 @@ "body": "* rename `input` -> `input_schema`\r\n* rename `output` -> `output_schema`\r\n* make graphs generic on `OutputT` to prep for future type checking\r\n\r\nAll renaming operations are backwards compatible in that we populate old input / output into their respective new schemas!", "created_at": "2025-06-06T18:53:54Z", "merged_at": "2025-06-06T23:44:56Z", - "pre_merge_commit_sha": "5920d8aa92fb8a76c7629a65acac5480387de0a5" + "pre_merge_commit_sha": "5920d8aa92fb8a76c7629a65acac5480387de0a5", + "test_files": [ + "libs/langgraph/tests/test_deprecation.py", + "libs/langgraph/tests/test_large_cases.py", + "libs/langgraph/tests/test_pregel.py", + "libs/langgraph/tests/test_pregel_async.py", + "libs/langgraph/tests/test_state.py", + "libs/langgraph/tests/test_type_checking.py" + ] }, { "url": "https://api.github.com/repos/langchain-ai/langgraph/pulls/3889", @@ -162,7 +188,8 @@ "body": "- Previously the global resume value was passed to subgraphs without being consumed\r\n- This would result in two parallel subgraph calls being able to use the same resume value\r\n- Note this behavior can't be implemented over the wire, that will be fixed in future PR\r\n\r\nCloses #3398 ", "created_at": "2025-03-18T03:32:30Z", "merged_at": "2025-03-18T04:26:34Z", - "pre_merge_commit_sha": "dd16ae4ba5243b4f0e4228f9a00aac477146c301" + "pre_merge_commit_sha": "dd16ae4ba5243b4f0e4228f9a00aac477146c301", + "test_files": ["libs/langgraph/tests/test_pregel.py"] }, { "url": "https://api.github.com/repos/langchain-ai/langgraph/pulls/3110", @@ -177,7 +204,8 @@ "body": "\r\n\r\n- this was not possible in async where all done callbacks are called in next tick\r\n- in sync case this would manifest as the first task done callback seeing counter == 1 and thus setting event\r\n- the fix is to unset the event whenever a task is scheduled", "created_at": "2025-01-20T19:41:44Z", "merged_at": "2025-01-21T18:16:10Z", - "pre_merge_commit_sha": "d48b25420ba7553a7154d3960fb1bf327d497e9d" + "pre_merge_commit_sha": "d48b25420ba7553a7154d3960fb1bf327d497e9d", + "test_files": ["libs/langgraph/tests/test_large_cases.py"] }, { "url": "https://api.github.com/repos/langchain-ai/langgraph/pulls/3037", @@ -192,7 +220,12 @@ "body": "- order was incorrectly based on task id, instead of the correct task path\r\n- this requires storing task paths on checkpointers\r\n- addition of task_path to put_writes is made backwards compatible by checking signature on call, and treating it as an optional arg", "created_at": "2025-01-15T02:12:35Z", "merged_at": "2025-01-15T19:42:08Z", - "pre_merge_commit_sha": "0adbd89d9aaad57e8e4431f308c4372a740e4cbf" + "pre_merge_commit_sha": "0adbd89d9aaad57e8e4431f308c4372a740e4cbf", + "test_files": [ + "libs/langgraph/tests/test_algo.py", + "libs/langgraph/tests/test_large_cases.py", + "libs/langgraph/tests/test_pregel_async.py" + ] }, { "url": "https://api.github.com/repos/langchain-ai/langgraph/pulls/2393", @@ -207,7 +240,11 @@ "body": "- This works similarly to the input() function from stdlib\r\n- calling it in a node interrupts execution\r\n- invoking the graph with Command(resume=...) will set ... as the return value of interrupt() so that the node can access the \"answer\" to the \"question\"\r\n- This PR also starts the work to control the graph on invoke/stream with Command() input, to be continued in a future PR\r\n\r\n```py\r\n class State(TypedDict):\r\n my_key: Annotated[str, operator.add]\r\n market: str\r\n\r\n async def tool_two_node(s: State) -> State:\r\n if s[\"market\"] == \"DE\":\r\n answer = interrupt(\"Just because...\")\r\n else:\r\n answer = \" all good\"\r\n return {\"my_key\": answer}\r\n\r\n tool_two_graph = StateGraph(State)\r\n tool_two_graph.add_node(\"tool_two\", tool_two_node)\r\n tool_two_graph.add_edge(START, \"tool_two\")\r\n tool_two = tool_two_graph.compile()\r\n\r\n tool_two = tool_two_graph.compile(checkpointer=checkpointer)\r\n\r\n # flow: interrupt -> resume with answer\r\n thread2 = {\"configurable\": {\"thread_id\": \"2\"}}\r\n # stop when about to enter node\r\n assert [\r\n c\r\n async for c in tool_two.astream(\r\n {\"my_key\": \"value ⛰️\", \"market\": \"DE\"}, thread2\r\n )\r\n ] == [\r\n {\"__interrupt__\": [Interrupt(value=\"Just because...\", when=\"during\")]},\r\n ]\r\n # resume with answer\r\n assert [\r\n c async for c in tool_two.astream(Command(resume=\" my answer\"), thread2)\r\n ] == [\r\n {\"tool_two\": {\"my_key\": \" my answer\"}},\r\n ]\r\n```", "created_at": "2024-11-12T01:45:14Z", "merged_at": "2024-11-13T21:34:20Z", - "pre_merge_commit_sha": "7a3ea427432dd5e8f4ee101a8c773f3afbc3214c" + "pre_merge_commit_sha": "7a3ea427432dd5e8f4ee101a8c773f3afbc3214c", + "test_files": [ + "libs/langgraph/tests/test_pregel.py", + "libs/langgraph/tests/test_pregel_async.py" + ] }, { "url": "https://api.github.com/repos/langchain-ai/langgraph/pulls/1776", @@ -222,7 +259,11 @@ "body": "- adds the ability for nodes (including in subgraphs) to emit chunks directly to the output stream, emitted chunks can have any type\r\n- when stream_mode=custom isnt requested by the caller emitted chunks are ignored", "created_at": "2024-09-19T23:46:37Z", "merged_at": "2024-09-20T16:18:51Z", - "pre_merge_commit_sha": "531890e35a35f9b381d375167a7b104184378153" + "pre_merge_commit_sha": "531890e35a35f9b381d375167a7b104184378153", + "test_files": [ + "libs/langgraph/tests/test_pregel.py", + "libs/langgraph/tests/test_pregel_async.py" + ] }, { "url": "https://api.github.com/repos/langchain-ai/langgraph/pulls/1735", @@ -237,7 +278,8 @@ "body": "- previous behavior was to buffer all output from subgraph until it finished, now subgraph steps are emitted as soon as produced, while the subgraph is still running\r\n- this is slightly slower in benchmark scripts, but worth it as it's much \"faster\" in real-world latency", "created_at": "2024-09-17T01:04:29Z", "merged_at": "2024-09-17T17:14:26Z", - "pre_merge_commit_sha": "f59435a892e96fb9092e0055a6df2e1dfc5111a9" + "pre_merge_commit_sha": "f59435a892e96fb9092e0055a6df2e1dfc5111a9", + "test_files": ["libs/langgraph/tests/test_pregel_async.py"] }, { "url": "https://api.github.com/repos/langchain-ai/langgraph/pulls/1630", @@ -252,7 +294,11 @@ "body": "- Orchestrator and Executor classes to run LangGraph in a distributed fashion using Kafka as a message bus for communication\r\n- Orchestrator and Executor run on-demand when a new message is published to the topic they listen to\r\n- Orchestrator is responsible for running the Pregel algorithm (deciding next tasks to run) and sending messages to the executor topic\r\n- Executor is responsible for executing each task (node), and sending messages to the orchestrator topic when done", "created_at": "2024-09-06T01:01:06Z", "merged_at": "2024-09-11T00:31:59Z", - "pre_merge_commit_sha": "34d530d5d83837fa080c1db2d055fe952cfc8488" + "pre_merge_commit_sha": "34d530d5d83837fa080c1db2d055fe952cfc8488", + "test_files": [ + "libs/langgraph/tests/test_pregel.py", + "libs/langgraph/tests/test_pregel_async.py" + ] }, { "url": "https://api.github.com/repos/langchain-ai/langgraph/pulls/809", @@ -267,7 +313,8 @@ "body": "When an instance of a callable class is passed as the path arg to add_conditional_edges but no path_map is provided, get_type_hints(path) is called, which raises a TypeError (since get_type_hints only accepts a module, class, method, or function).\r\n\r\nThis patch fixes the error by trying to get type hints from path.\\_\\_call\\_\\_ first, which should work for instances of callable classes.\r\n\r\nTested: Added a test that raises TypeError without the fix in this patch but passes with the fix.", "created_at": "2024-06-25T21:38:27Z", "merged_at": "2024-06-26T23:24:59Z", - "pre_merge_commit_sha": "6ae59581643b51751731ae64c609a8bc21779714" + "pre_merge_commit_sha": "6ae59581643b51751731ae64c609a8bc21779714", + "test_files": ["libs/langgraph/tests/test_pregel.py"] }, { "url": "https://api.github.com/repos/langchain-ai/langgraph/pulls/651", @@ -282,6 +329,7 @@ "body": "This change allows users or graph nodes to remove messages by `id` via `langchain_core.messages.RemoveMessage`\r\n\r\nExamples:\r\n\r\n* allow users to delete messages from state by calling\r\n\r\n```python\r\ngraph.update_state(config, values=[RemoveMessage(id=state.values[-1].id)])\r\n```\r\n\r\n* allow nodes to delete messages\r\n\r\n```python\r\ngraph.add_node(\"delete_messages\", lambda state: [RemoveMessage(id=state[-1].id)])\r\n```", "created_at": "2024-06-12T14:35:51Z", "merged_at": "2024-07-03T05:43:54Z", - "pre_merge_commit_sha": "5e8aa5d9f2e24b197ffa187c6b7b36602761d1a4" + "pre_merge_commit_sha": "5e8aa5d9f2e24b197ffa187c6b7b36602761d1a4", + "test_files": ["libs/langgraph/tests/test_pregel.py"] } ] diff --git a/apps/open-swe/langbench/types.ts b/apps/open-swe/langbench/types.ts index 2bcaea6a..388ed968 100644 --- a/apps/open-swe/langbench/types.ts +++ b/apps/open-swe/langbench/types.ts @@ -1,26 +1,64 @@ +import { Sandbox } from "@daytonaio/sdk"; + export interface PRData { url: string; - html_url: string; - diff_url: string; - patch_url: string; - repo_owner: string; - repo_name: string; - pr_number: number; - merge_commit_sha: string; - pre_merge_commit_sha: string; + htmlUrl: string; + diffUrl: string; + patchUrl: string; + repoOwner: string; + repoName: string; + prNumber: number; + mergeCommitSha: string; + preMergeCommitSha: string; title: string; body: string; - created_at: string; - merged_at: string; + createdAt: string; + mergedAt: string; + testFiles: string[]; +} + +export interface TestResults { + success: boolean; + error: string | null; + totalTests: number; + passedTests: number; + failedTests: number; + testDetails: string[]; +} + +export interface PytestJsonTest { + nodeid: string; + outcome: "passed" | "failed" | "error" | "skipped"; +} + +export interface PytestJsonSummary { + passed?: number; + failed?: number; + error?: number; + skipped?: number; +} + +export interface PytestJsonReport { + tests?: PytestJsonTest[]; + summary?: PytestJsonSummary; } export interface PRProcessResult { - pr_number: number; - repo_name: string; - workspace_id?: string; + prNumber: number; + repoName: string; + workspaceId?: string; success: boolean; - evals_found: boolean; - evals_files: string[]; + evalsFound: boolean; + evalsFiles: string[]; + testFiles: string[]; + testResults?: TestResults; error?: string; - pre_merge_sha?: string; + preMergeSha?: string; +} + +export interface RunPytestOptions { + sandbox: Sandbox; + testFiles: string[]; + repoDir: string; + timeoutSec?: number; } diff --git a/apps/open-swe/langbench/utils.ts b/apps/open-swe/langbench/utils.ts new file mode 100644 index 00000000..28643dbc --- /dev/null +++ b/apps/open-swe/langbench/utils.ts @@ -0,0 +1,245 @@ +import { createLogger, LogLevel } from "../src/utils/logger.js"; +import { ENV_CONSTANTS } from "../src/utils/env-setup.js"; +import { TestResults, PytestJsonReport, RunPytestOptions } from "./types.js"; +import { readFile } from "../src/utils/read-write.js"; + +const logger = createLogger(LogLevel.DEBUG, "Langbench Utils"); + +/** + * Fetch diff content from a diff URL and extract test file names, this function is used in one-off situtations to get the test files from the diff url. + */ +export async function getTestFilesFromDiff(diffUrl: string): Promise { + try { + const response = await fetch(diffUrl); + if (!response.ok) { + throw new Error(`Failed to fetch diff: ${response.statusText}`); + } + + const diffContent = await response.text(); + const testFiles: string[] = []; + + // Parse the diff to find modified files + const lines = diffContent.split("\n"); + for (const line of lines) { + // Look for diff file headers + if (line.startsWith("diff --git ")) { + const match = line.match(/diff --git a\/(.+?) b\//); + if (match) { + const filePath = match[1]; + // Check if this is a test file in libs/langgraph/tests/ + if (isLangGraphTestFile(filePath)) { + testFiles.push(filePath); + } + } + } + } + + return [...new Set(testFiles)]; // Remove duplicates + } catch (error) { + logger.error(`Failed to fetch or parse diff from ${diffUrl}:`, { error }); + return []; + } +} + +/** + * Check if a file path represents a test file in libs/langgraph/tests/ + */ +function isLangGraphTestFile(filePath: string): boolean { + return filePath.includes("libs/langgraph/tests/") && filePath.endsWith(".py"); +} + +// Use shared constants from env-setup utility +const { RUN_PYTHON_IN_VENV, RUN_PIP_IN_VENV } = ENV_CONSTANTS; + +// Installation commands for pytest and dependencies +const PIP_INSTALL_COMMAND = `${RUN_PIP_IN_VENV} install pytest pytest-mock pytest-asyncio syrupy pytest-json-report`; +const LANGGRAPH_INSTALL_COMMAND = `${RUN_PIP_IN_VENV} install -e ./libs/langgraph`; + +/** + * Run pytest on specific test files and return structured results + */ +export async function runPytestOnFiles( + options: RunPytestOptions, +): Promise { + const { sandbox, testFiles, repoDir, timeoutSec = 300 } = options; + if (testFiles.length === 0) { + logger.warn("No test files provided, skipping pytest execution"); + return { + success: true, + error: null, + totalTests: 0, + passedTests: 0, + failedTests: 0, + testDetails: [], + }; + } + + logger.info(`Running pytest on ${testFiles.length} test files`, { + testFiles, + }); + + // Join test files for pytest command + const testFilesArg = testFiles.join(" "); + const command = `${RUN_PYTHON_IN_VENV} -m pytest ${testFilesArg} -v --tb=short --json-report --json-report-file=/tmp/pytest_report.json`; + logger.info("Running pytest command", { command }); + + logger.info( + "Installing pytest, pytest-mock, pytest-asyncio, syrupy, pytest-json-report, and langgraph in virtual environment...", + ); + + // Execute pip install command + logger.info(`Running pip install command: ${PIP_INSTALL_COMMAND}`); + const pipInstallResult = await sandbox.process.executeCommand( + PIP_INSTALL_COMMAND, + repoDir, + undefined, + timeoutSec * 2, + ); + + logger.info(`Pip install command completed`, { + exitCode: pipInstallResult.exitCode, + output: pipInstallResult.result?.slice(0, 500), + }); + + if (pipInstallResult.exitCode !== 0) { + logger.error(`Pip install command failed`, { + command: PIP_INSTALL_COMMAND, + exitCode: pipInstallResult.exitCode, + output: pipInstallResult.result, + }); + } + + // Execute langgraph install command + logger.info( + `Running langgraph install command: ${LANGGRAPH_INSTALL_COMMAND}`, + ); + const langgraphInstallResult = await sandbox.process.executeCommand( + LANGGRAPH_INSTALL_COMMAND, + repoDir, + undefined, + timeoutSec * 2, + ); + + logger.info(`Langgraph install command completed`, { + exitCode: langgraphInstallResult.exitCode, + output: langgraphInstallResult.result?.slice(0, 500), + }); + + if (langgraphInstallResult.exitCode !== 0) { + logger.error(`Langgraph install command failed`, { + command: LANGGRAPH_INSTALL_COMMAND, + exitCode: langgraphInstallResult.exitCode, + output: langgraphInstallResult.result, + }); + } + + try { + const execution = await sandbox.process.executeCommand( + command, + repoDir, + undefined, + timeoutSec, + ); + + // Read the JSON report file + let parsed: Omit; + try { + const jsonReportResult = await readFile({ + sandbox, + filePath: "/tmp/pytest_report.json", + workDir: repoDir, + }); + + if (jsonReportResult.success && jsonReportResult.output) { + const jsonReport = JSON.parse(jsonReportResult.output); + parsed = parsePytestJsonReport(jsonReport); + logger.debug("Successfully parsed JSON report", { jsonReport }); + } else { + throw new Error( + `Failed to read JSON report: ${jsonReportResult.output}`, + ); + } + } catch (jsonError) { + throw new Error("Failed to parse JSON report", { cause: jsonError }); + } + + logger.info("Pytest execution completed", { + exitCode: execution.exitCode, + totalTests: parsed.totalTests, + passedTests: parsed.passedTests, + failedTests: parsed.failedTests, + command, + stdout: execution.result, + fullExecution: JSON.stringify(execution, null, 2), // Show full execution object + }); + + return { + success: execution.exitCode === 0, + error: + execution.exitCode !== 0 ? `Exit code: ${execution.exitCode}` : null, + ...parsed, + }; + } catch (error) { + logger.error("Failed to run pytest", { error }); + return { + success: false, + error: error instanceof Error ? error.message : String(error), + totalTests: 0, + passedTests: 0, + failedTests: 0, + testDetails: [], + }; + } +} + +/** + * Parse pytest JSON report to extract test results + */ +export function parsePytestJsonReport( + jsonReport: PytestJsonReport, +): Omit { + let totalTests = 0; + let passedTests = 0; + let failedTests = 0; + const testDetails: string[] = []; + + if (jsonReport && jsonReport.tests) { + totalTests = jsonReport.tests.length; + + for (const test of jsonReport.tests) { + const testName = `${test.nodeid}`; + const outcome = test.outcome; + + if (outcome === "passed") { + passedTests++; + testDetails.push(`${testName} PASSED`); + } else if (outcome === "failed" || outcome === "error") { + failedTests++; + testDetails.push(`${testName} ${outcome.toUpperCase()}`); + } + } + } + + // Use summary data if available + if (jsonReport && jsonReport.summary) { + const summary = jsonReport.summary; + if (summary.passed !== undefined) passedTests = summary.passed; + if (summary.failed !== undefined) failedTests = summary.failed; + if (summary.error !== undefined) failedTests += summary.error; + totalTests = passedTests + failedTests; + } + + logger.debug("Parsed pytest JSON report", { + totalTests, + passedTests, + failedTests, + detailsCount: testDetails.length, + }); + + return { + totalTests, + passedTests, + failedTests, + testDetails, + }; +} diff --git a/apps/open-swe/src/utils/github/git.ts b/apps/open-swe/src/utils/github/git.ts index 6dc995d8..7e013e5d 100644 --- a/apps/open-swe/src/utils/github/git.ts +++ b/apps/open-swe/src/utils/github/git.ts @@ -613,3 +613,50 @@ async function performClone( return branchName; } + +export interface CheckoutFilesOptions { + sandbox: Sandbox; + repoDir: string; + commitSha: string; + filePaths: string[]; +} + +/** + * Checkout specific files from a given commit + */ +export async function checkoutFilesFromCommit( + options: CheckoutFilesOptions, +): Promise { + const { sandbox, repoDir, commitSha, filePaths } = options; + + if (filePaths.length === 0) { + return; + } + + logger.info( + `Checking out ${filePaths.length} files from commit ${commitSha}`, + ); + + for (const filePath of filePaths) { + try { + const result = await sandbox.process.executeCommand( + `git checkout --force ${commitSha} -- "${filePath}"`, + repoDir, + undefined, + 30, + ); + + if (result.exitCode !== 0) { + logger.warn( + `Failed to checkout file ${filePath} from commit ${commitSha}: ${result.result || "Unknown error"}`, + ); + } else { + logger.info( + `Successfully checked out ${filePath} from commit ${commitSha}`, + ); + } + } catch (error) { + logger.warn(`Error checking out file ${filePath}:`, { error }); + } + } +}