Fix issue #5746 : [Bug]: Broken "Open in VSCode" button

[Bug]: Missing path import (#5747 )
Revert "[Resolver]: Add target branch param" (#5743 )
2026-04-29 03:00:45 -04:00 · 2024-12-22 18:41:57 +00:00 · 2024-12-22 15:58:17 +00:00 · 2024-12-22 01:28:23 +00:00 · 2024-12-22 04:41:39 +08:00 · 2024-12-21 19:07:31 +00:00
515 changed files with 11192 additions and 28561 deletions
--- a/.devcontainer/README.MD
+++ b/.devcontainer/README.MD
@@ -1 +0,0 @@
-The files in this directory configure a development container for GitHub Codespaces.
--- a/.devcontainer/devcontainer.json
+++ b/.devcontainer/devcontainer.json
@@ -1,15 +0,0 @@
-{
-	"name": "OpenHands Codespaces",
-	"image": "mcr.microsoft.com/devcontainers/universal",
-	"customizations":{
-        "vscode":{
-            "extensions": [
-                "ms-python.python"
-            ]
-        }
-    },
-	"onCreateCommand": "sh ./.devcontainer/on_create.sh",
-	"postCreateCommand": "make build",
-	"postStartCommand": "USE_HOST_NETWORK=True nohup bash -c 'make run &'"
-
-}
--- a/.devcontainer/on_create.sh
+++ b/.devcontainer/on_create.sh
@@ -1,6 +0,0 @@
-#!/usr/bin/env bash
-sudo apt update
-sudo apt install -y netcat
-sudo add-apt-repository -y ppa:deadsnakes/ppa
-sudo apt install -y python3.12
-curl -sSL https://install.python-poetry.org | python3.12 -
--- a/.github/dependabot.yml
+++ b/.github/dependabot.yml
@@ -18,7 +18,7 @@ updates:
          - "chromadb"
      browsergym:
        patterns:
-          - "browsergym"
+          - "browsergym*"
      security-all:
        applies-to: "security-updates"
        patterns:
--- a/.github/workflows/fe-unit-tests.yml
+++ b/.github/workflows/fe-unit-tests.yml
@@ -24,7 +24,8 @@ jobs:
    runs-on: ubuntu-latest
    strategy:
      matrix:
-        node-version: [20]
+        node-version: [20, 22]
+      fail-fast: true
    steps:
      - name: Checkout
        uses: actions/checkout@v4
@@ -35,6 +36,9 @@ jobs:
      - name: Install dependencies
        working-directory: ./frontend
        run: npm ci
+      - name: Run TypeScript compilation
+        working-directory: ./frontend
+        run: npm run make-i18n && tsc
      - name: Run tests and collect coverage
        working-directory: ./frontend
        run: npm run test:coverage
--- a/.github/workflows/ghcr-build.yml
+++ b/.github/workflows/ghcr-build.yml
@@ -291,7 +291,7 @@ jobs:
          SANDBOX_RUNTIME_CONTAINER_IMAGE=$image_name \
          TEST_IN_CI=true \
          RUN_AS_OPENHANDS=false \
-          poetry run pytest -n 3 -raRs --reruns 2 --reruns-delay 5 --cov=openhands --cov-report=xml -s ./tests/runtime
+          poetry run pytest -n 3 -raRs --reruns 2 --reruns-delay 5 --cov=openhands --cov-report=xml -s ./tests/runtime --ignore=tests/runtime/test_browsergym_envs.py
      - name: Upload coverage to Codecov
        uses: codecov/codecov-action@v4
        env:
@@ -368,7 +368,7 @@ jobs:
          SANDBOX_RUNTIME_CONTAINER_IMAGE=$image_name \
          TEST_IN_CI=true \
          RUN_AS_OPENHANDS=true \
-          poetry run pytest -n 3 -raRs --reruns 2 --reruns-delay 5 --cov=openhands --cov-report=xml -s ./tests/runtime
+          poetry run pytest -n 3 -raRs --reruns 2 --reruns-delay 5 --cov=openhands --cov-report=xml -s ./tests/runtime --ignore=tests/runtime/test_browsergym_envs.py
      - name: Upload coverage to Codecov
        uses: codecov/codecov-action@v4
        env:
--- a/.github/workflows/lint-fix.yml
+++ b/.github/workflows/lint-fix.yml
@@ -5,9 +5,10 @@ on:
    types: [labeled]

 jobs:
-  lint-fix:
+  # Frontend lint fixes
+  lint-fix-frontend:
    if: github.event.label.name == 'lint-fix'
-    name: Fix linting issues
+    name: Fix frontend linting issues
    runs-on: ubuntu-latest
    permissions:
      contents: write
@@ -20,7 +21,6 @@ jobs:
          fetch-depth: 0
          token: ${{ secrets.GITHUB_TOKEN }}

-      # Frontend lint fixes
      - name: Install Node.js 20
        uses: actions/setup-node@v4
        with:
@@ -34,7 +34,36 @@ jobs:
          cd frontend
          npm run lint:fix

-      # Python lint fixes
+      # Commit and push changes if any
+      - name: Check for changes
+        id: git-check
+        run: |
+          git diff --quiet || echo "changes=true" >> $GITHUB_OUTPUT
+      - name: Commit and push if there are changes
+        if: steps.git-check.outputs.changes == 'true'
+        run: |
+          git config --local user.email "openhands@all-hands.dev"
+          git config --local user.name "OpenHands Bot"
+          git add -A
+          git commit -m "🤖 Auto-fix frontend linting issues"
+          git push
+
+  # Python lint fixes
+  lint-fix-python:
+    if: github.event.label.name == 'lint-fix'
+    name: Fix Python linting issues
+    runs-on: ubuntu-latest
+    permissions:
+      contents: write
+      pull-requests: write
+    steps:
+      - uses: actions/checkout@v4
+        with:
+          ref: ${{ github.head_ref }}
+          repository: ${{ github.event.pull_request.head.repo.full_name }}
+          fetch-depth: 0
+          token: ${{ secrets.GITHUB_TOKEN }}
+
      - name: Set up python
        uses: actions/setup-python@v5
        with:
@@ -58,5 +87,5 @@ jobs:
          git config --local user.email "openhands@all-hands.dev"
          git config --local user.name "OpenHands Bot"
          git add -A
-          git commit -m "🤖 Auto-fix linting issues"
+          git commit -m "🤖 Auto-fix Python linting issues"
          git push
--- a/.github/workflows/lint.yml
+++ b/.github/workflows/lint.yml
@@ -30,10 +30,11 @@ jobs:
        run: |
          cd frontend
          npm install --frozen-lockfile
-      - name: Lint
+      - name: Lint and TypeScript compilation
        run: |
          cd frontend
          npm run lint
+          npm run make-i18n && tsc

  # Run lint on the python code
  lint-python:
--- a/.github/workflows/openhands-resolver.yml
+++ b/.github/workflows/openhands-resolver.yml
@@ -16,17 +16,26 @@ on:
        type: string
        default: "main"
        description: "Target branch to pull and create PR against"
+      LLM_MODEL:
+        required: false
+        type: string
+        default: "anthropic/claude-3-5-sonnet-20241022"
+      base_container_image:
+        required: false
+        type: string
+        default: ""
+        description: "Custom sandbox env"
    secrets:
      LLM_MODEL:
-        required: true
+        required: false
      LLM_API_KEY:
        required: true
      LLM_BASE_URL:
        required: false
      PAT_TOKEN:
-        required: true
+        required: false
      PAT_USERNAME:
-        required: true
+        required: false

  issues:
    types: [labeled]
@@ -50,7 +59,6 @@ jobs:
      github.event_name == 'workflow_call' ||
      github.event.label.name == 'fix-me' ||
      github.event.label.name == 'fix-me-experimental' ||
-
      (
        ((github.event_name == 'issue_comment' || github.event_name == 'pull_request_review_comment') &&
        contains(github.event.comment.body, inputs.macro || '@openhands-agent') &&
@@ -101,13 +109,14 @@ jobs:

      - name: Check required environment variables
        env:
-          LLM_MODEL: ${{ secrets.LLM_MODEL }}
+          LLM_MODEL: ${{ secrets.LLM_MODEL || inputs.LLM_MODEL }}
          LLM_API_KEY: ${{ secrets.LLM_API_KEY }}
          LLM_BASE_URL: ${{ secrets.LLM_BASE_URL }}
          PAT_TOKEN: ${{ secrets.PAT_TOKEN }}
          PAT_USERNAME: ${{ secrets.PAT_USERNAME }}
+          GITHUB_TOKEN: ${{ github.token }}
        run: |
-          required_vars=("LLM_MODEL" "LLM_API_KEY" "PAT_TOKEN" "PAT_USERNAME")
+          required_vars=("LLM_API_KEY")
          for var in "${required_vars[@]}"; do
            if [ -z "${!var}" ]; then
              echo "Error: Required environment variable $var is not set."
@@ -115,17 +124,34 @@ jobs:
            fi
          done

+          # Check optional variables and warn about fallbacks
+          if [ -z "$LLM_BASE_URL" ]; then
+            echo "Warning: LLM_BASE_URL is not set, will use default API endpoint"
+          fi
+
+          if [ -z "$PAT_TOKEN" ]; then
+            echo "Warning: PAT_TOKEN is not set, falling back to GITHUB_TOKEN"
+          fi
+
+          if [ -z "$PAT_USERNAME" ]; then
+            echo "Warning: PAT_USERNAME is not set, will use openhands-agent"
+          fi
+
      - name: Set environment variables
        run: |
-          if [ -n "${{ github.event.review.body }}" ]; then
+          # Handle pull request events first
+          if [ -n "${{ github.event.pull_request.number }}" ]; then
            echo "ISSUE_NUMBER=${{ github.event.pull_request.number }}" >> $GITHUB_ENV
            echo "ISSUE_TYPE=pr" >> $GITHUB_ENV
+          # Handle pull request review events
+          elif [ -n "${{ github.event.review.body }}" ]; then
+            echo "ISSUE_NUMBER=${{ github.event.pull_request.number }}" >> $GITHUB_ENV
+            echo "ISSUE_TYPE=pr" >> $GITHUB_ENV
+          # Handle issue comment events that reference a PR
          elif [ -n "${{ github.event.issue.pull_request }}" ]; then
            echo "ISSUE_NUMBER=${{ github.event.issue.number }}" >> $GITHUB_ENV
            echo "ISSUE_TYPE=pr" >> $GITHUB_ENV
-          elif [ -n "${{ github.event.pull_request.number }}" ]; then
-            echo "ISSUE_NUMBER=${{ github.event.pull_request.number }}" >> $GITHUB_ENV
-            echo "ISSUE_TYPE=pr" >> $GITHUB_ENV
+          # Handle regular issue events
          else
            echo "ISSUE_NUMBER=${{ github.event.issue.number }}" >> $GITHUB_ENV
            echo "ISSUE_TYPE=issue" >> $GITHUB_ENV
@@ -138,7 +164,8 @@ jobs:
          fi

          echo "MAX_ITERATIONS=${{ inputs.max_iterations || 50 }}" >> $GITHUB_ENV
-          echo "SANDBOX_ENV_GITHUB_TOKEN=${{ secrets.GITHUB_TOKEN }}" >> $GITHUB_ENV
+          echo "SANDBOX_ENV_GITHUB_TOKEN=${{ secrets.PAT_TOKEN || github.token }}" >> $GITHUB_ENV
+          echo "SANDBOX_ENV_BASE_CONTAINER_IMAGE=${{ inputs.base_container_image }}" >> $GITHUB_ENV

          # Set branch variables
          echo "TARGET_BRANCH=${{ inputs.target_branch }}" >> $GITHUB_ENV
@@ -146,7 +173,7 @@ jobs:
      - name: Comment on issue with start message
        uses: actions/github-script@v7
        with:
-          github-token: ${{secrets.GITHUB_TOKEN}}
+          github-token: ${{ secrets.PAT_TOKEN || github.token }}
          script: |
            const issueType = process.env.ISSUE_TYPE;
            github.rest.issues.createComment({
@@ -157,23 +184,38 @@ jobs:
            });

      - name: Install OpenHands
-        run: |
-          if [[ "${{ github.event.label.name }}" == "fix-me-experimental" ]] ||
-             ([[ "${{ github.event_name }}" == "issue_comment" || "${{ github.event_name }}" == "pull_request_review_comment" ]] &&
-              [[ "${{ github.event.comment.body }}" == "@openhands-agent-exp"* ]]) ||
-             ([[ "${{ github.event_name }}" == "pull_request_review" ]] &&
-              [[ "${{ github.event.review.body }}" == "@openhands-agent-exp"* ]]); then
-            python -m pip install --upgrade pip
-            pip install git+https://github.com/all-hands-ai/openhands.git
-          else
-            python -m pip install --upgrade -r requirements.txt
-          fi
+        uses: actions/github-script@v7
+        with:
+          script: |
+            const commentBody = `${{ github.event.comment.body || '' }}`.trim();
+            const reviewBody = `${{ github.event.review.body || '' }}`.trim();
+            const labelName = `${{ github.event.label.name || '' }}`.trim();
+            const eventName = `${{ github.event_name }}`.trim();
+
+            // Check conditions
+            const isExperimentalLabel = labelName === "fix-me-experimental";
+            const isIssueCommentExperimental =
+              (eventName === "issue_comment" || eventName === "pull_request_review_comment") &&
+              commentBody.includes("@openhands-agent-exp");
+            const isReviewCommentExperimental =
+              eventName === "pull_request_review" && reviewBody.includes("@openhands-agent-exp");
+
+            // Perform package installation
+            if (isExperimentalLabel || isIssueCommentExperimental || isReviewCommentExperimental) {
+              console.log("Installing experimental OpenHands...");
+              await exec.exec("python -m pip install --upgrade pip");
+              await exec.exec("pip install git+https://github.com/all-hands-ai/openhands.git");
+            } else {
+              console.log("Installing from requirements.txt...");
+              await exec.exec("python -m pip install --upgrade pip");
+              await exec.exec("pip install -r requirements.txt");
+            }

      - name: Attempt to resolve issue
        env:
-          GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
-          GITHUB_USERNAME: ${{ secrets.PAT_USERNAME }}
-          LLM_MODEL: ${{ secrets.LLM_MODEL }}
+          GITHUB_TOKEN: ${{ secrets.PAT_TOKEN || github.token }}
+          GITHUB_USERNAME: ${{ secrets.PAT_USERNAME || 'openhands-agent' }}
+          LLM_MODEL: ${{ secrets.LLM_MODEL || inputs.LLM_MODEL }}
          LLM_API_KEY: ${{ secrets.LLM_API_KEY }}
          LLM_BASE_URL: ${{ secrets.LLM_BASE_URL }}
          PYTHONPATH: ""
@@ -183,7 +225,7 @@ jobs:
            --issue-number ${{ env.ISSUE_NUMBER }} \
            --issue-type ${{ env.ISSUE_TYPE }} \
            --max-iterations ${{ env.MAX_ITERATIONS }} \
-            --comment-id ${{ env.COMMENT_ID }} \
+            --comment-id ${{ env.COMMENT_ID }}

      - name: Check resolution result
        id: check_result
@@ -205,9 +247,9 @@ jobs:
      - name: Create draft PR or push branch
        if: always() # Create PR or branch even if the previous steps fail
        env:
-          GITHUB_TOKEN: ${{ secrets.PAT_TOKEN }}
-          GITHUB_USERNAME: ${{ secrets.PAT_USERNAME }}
-          LLM_MODEL: ${{ secrets.LLM_MODEL }}
+          GITHUB_TOKEN: ${{ secrets.PAT_TOKEN || github.token }}
+          GITHUB_USERNAME: ${{ secrets.PAT_USERNAME || 'openhands-agent' }}
+          LLM_MODEL: ${{ secrets.LLM_MODEL || inputs.LLM_MODEL }}
          LLM_API_KEY: ${{ secrets.LLM_API_KEY }}
          LLM_BASE_URL: ${{ secrets.LLM_BASE_URL }}
          PYTHONPATH: ""
@@ -215,7 +257,8 @@ jobs:
          if [ "${{ steps.check_result.outputs.RESOLUTION_SUCCESS }}" == "true" ]; then
            cd /tmp && python -m openhands.resolver.send_pull_request \
              --issue-number ${{ env.ISSUE_NUMBER }} \
-              --pr-type draft | tee pr_result.txt && \
+              --pr-type draft \
+              --reviewer ${{ github.actor }} | tee pr_result.txt && \
              grep "draft created" pr_result.txt | sed 's/.*\///g' > pr_number.txt
          else
            cd /tmp && python -m openhands.resolver.send_pull_request \
@@ -225,30 +268,58 @@ jobs:
              grep "branch created" branch_result.txt | sed 's/.*\///g; s/.expand=1//g' > branch_name.txt
          fi

-      - name: Comment on issue
+      # Step leaves comment for when agent is invoked on PR
+      - name: Analyze Push Logs (Updated PR or No Changes) # Skip comment if PR update was successful OR leave comment if the agent made no code changes
        uses: actions/github-script@v7
-        if: always() # Comment on issue even if the previous steps fail
+        if: always()
+        env:
+          AGENT_RESPONDED: ${{ env.AGENT_RESPONDED || 'false' }}
        with:
-          github-token: ${{secrets.GITHUB_TOKEN}}
+          github-token: ${{ secrets.PAT_TOKEN || github.token }}
          script: |
            const fs = require('fs');
            const issueNumber = ${{ env.ISSUE_NUMBER }};
+            let logContent = '';
+
+            try {
+              logContent = fs.readFileSync('/tmp/pr_result.txt', 'utf8').trim();
+            } catch (error) {
+              console.error('Error reading pr_result.txt file:', error);
+            }
+
+            const noChangesMessage = `No changes to commit for issue #${issueNumber}. Skipping commit.`;
+
+            // Check logs from send_pull_request.py (pushes code to GitHub)
+            if (logContent.includes("Updated pull request")) {
+              console.log("Updated pull request found. Skipping comment.");
+              process.env.AGENT_RESPONDED = 'true';
+            } else if (logContent.includes(noChangesMessage)) {
+              github.rest.issues.createComment({
+                issue_number: issueNumber,
+                owner: context.repo.owner,
+                repo: context.repo.repo,
+                body: `The workflow to fix this issue encountered an error. Openhands failed to create any code changes.`
+              });
+              process.env.AGENT_RESPONDED = 'true';
+            }
+
+      # Step leaves comment for when agent is invoked on issue
+      - name: Comment on issue # Comment link to either PR or branch created by agent
+        uses: actions/github-script@v7
+        if: always() # Comment on issue even if the previous steps fail
+        env:
+          AGENT_RESPONDED: ${{ env.AGENT_RESPONDED || 'false' }}
+        with:
+          github-token: ${{ secrets.PAT_TOKEN || github.token }}
+          script: |
+            const fs = require('fs');
+            const path = require('path');
+            const issueNumber = ${{ env.ISSUE_NUMBER }};
            const success = ${{ steps.check_result.outputs.RESOLUTION_SUCCESS }};

            let prNumber = '';
            let branchName = '';
-            let logContent = '';
-            const noChangesMessage = `No changes to commit for issue #${issueNumber}. Skipping commit.`;
-
-            try {
-              if (success){
-                logContent = fs.readFileSync('/tmp/pr_result.txt', 'utf8').trim();
-              } else {
-                logContent = fs.readFileSync('/tmp/branch_result.txt', 'utf8').trim();
-              }
-            } catch (error) {
-              console.error('Error reading results file:', error);
-            }
+            let resultExplanation = '';

            try {
              if (success) {
@@ -260,32 +331,63 @@ jobs:
              console.error('Error reading file:', error);
            }

-            if (logContent.includes(noChangesMessage)) {
-              github.rest.issues.createComment({
-                issue_number: issueNumber,
-                owner: context.repo.owner,
-                repo: context.repo.repo,
-                body: `The workflow to fix this issue encountered an error. Openhands failed to create any code changes.`
-              });
-            } else if (success && prNumber) {
+
+            try {
+              if (!success){
+                // Read result_explanation from JSON file for failed resolution
+                const outputFilePath = path.resolve('/tmp/output/output.jsonl');
+                if (fs.existsSync(outputFilePath)) {
+                  const outputContent = fs.readFileSync(outputFilePath, 'utf8');
+                  const jsonLines = outputContent.split('\n').filter(line => line.trim() !== '');
+
+                  if (jsonLines.length > 0) {
+                    // First entry in JSON lines has the key 'result_explanation'
+                    const firstEntry = JSON.parse(jsonLines[0]);
+                    resultExplanation = firstEntry.result_explanation || '';
+                  }
+                }
+              }
+            } catch (error){
+              console.error('Error reading file:', error);
+            }
+
+            // Check "success" log from resolver output
+            if (success && prNumber) {
              github.rest.issues.createComment({
                issue_number: issueNumber,
                owner: context.repo.owner,
                repo: context.repo.repo,
                body: `A potential fix has been generated and a draft PR #${prNumber} has been created. Please review the changes.`
              });
+              process.env.AGENT_RESPONDED = 'true';
            } else if (!success && branchName) {
+              let commentBody = `An attempt was made to automatically fix this issue, but it was unsuccessful. A branch named '${branchName}' has been created with the attempted changes. You can view the branch [here](https://github.com/${context.repo.owner}/${context.repo.repo}/tree/${branchName}). Manual intervention may be required.`;
+
+              if (resultExplanation) {
+                commentBody += `\n\nAdditional details about the failure:\n${resultExplanation}`;
+              }
+
              github.rest.issues.createComment({
                issue_number: issueNumber,
                owner: context.repo.owner,
                repo: context.repo.repo,
-                body: `An attempt was made to automatically fix this issue, but it was unsuccessful. A branch named '${branchName}' has been created with the attempted changes. You can view the branch [here](https://github.com/${context.repo.owner}/${context.repo.repo}/tree/${branchName}). Manual intervention may be required.`
-              });
-            } else {
-              github.rest.issues.createComment({
-                issue_number: issueNumber,
-                owner: context.repo.owner,
-                repo: context.repo.repo,
-                body: `The workflow to fix this issue encountered an error. Please check the [workflow logs](https://github.com/${context.repo.owner}/${context.repo.repo}/actions/runs/${context.runId}) for more information.`
+                body: commentBody
              });
+              process.env.AGENT_RESPONDED = 'true';
            }
+
+      # Leave error comment when both PR/Issue comment handling fail
+      - name: Fallback Error Comment
+        uses: actions/github-script@v7
+        if: ${{ env.AGENT_RESPONDED == 'false' }} # Only run if no conditions were met in previous steps
+        with:
+          github-token: ${{ secrets.PAT_TOKEN || github.token }}
+          script: |
+            const issueNumber = ${{ env.ISSUE_NUMBER }};
+
+            github.rest.issues.createComment({
+              issue_number: issueNumber,
+              owner: context.repo.owner,
+              repo: context.repo.repo,
+              body: `The workflow to fix this issue encountered an error. Please check the [workflow logs](https://github.com/${context.repo.owner}/${context.repo.repo}/actions/runs/${context.runId}) for more information.`
+            });
--- a/.github/workflows/py-unit-tests.yml
+++ b/.github/workflows/py-unit-tests.yml
@@ -42,7 +42,7 @@ jobs:
      - name: Build Environment
        run: make build
      - name: Run Tests
-        run: poetry run pytest --forked --cov=openhands --cov-report=xml -svv ./tests/unit --ignore=tests/unit/test_memory.py
+        run: poetry run pytest --forked -n auto --cov=openhands --cov-report=xml -svv ./tests/unit --ignore=tests/unit/test_memory.py
      - name: Upload coverage to Codecov
        uses: codecov/codecov-action@v4
        env:
--- a/.nvmrc
+++ b/.nvmrc
@@ -0,0 +1 @@
+22
--- a/CONTRIBUTING.md
+++ b/CONTRIBUTING.md
@@ -21,14 +21,14 @@ There are many ways that you can contribute:

 1. **Download and use** OpenHands, and send [issues](https://github.com/All-Hands-AI/OpenHands/issues) when you encounter something that isn't working or a feature that you'd like to see.
 2. **Send feedback** after each session by [clicking the thumbs-up thumbs-down buttons](https://docs.all-hands.dev/modules/usage/feedback), so we can see where things are working and failing, and also build an open dataset for training code agents.
-3. **Improve the Codebase** by sending PRs (see details below). In particular, we have some [good first issues](https://github.com/All-Hands-AI/OpenHands/labels/good%20first%20issue) that may be ones to start on.
+3. **Improve the Codebase** by sending [PRs](#sending-pull-requests-to-openhands) (see details below). In particular, we have some [good first issues](https://github.com/All-Hands-AI/OpenHands/labels/good%20first%20issue) that may be ones to start on.

 ## What can I build?
 Here are a few ways you can help improve the codebase.

 #### UI/UX
 We're always looking to improve the look and feel of the application. If you've got a small fix
-for something that's bugging you, feel free to open up a PR that changes the `./frontend` directory.
+for something that's bugging you, feel free to open up a PR that changes the [`./frontend`](./frontend) directory.

 If you're looking to make a bigger change, add a new UI element, or significantly alter the style
 of the application, please open an issue first, or better, join the #frontend channel in our Slack
@@ -46,7 +46,7 @@ We use the [SWE-bench](https://www.swebench.com/) benchmark to test our agent. Y
 channel in Slack to learn more.

 #### Adding a new agent
-You may want to experiment with building new types of agents. You can add an agent to `openhands/agenthub`
+You may want to experiment with building new types of agents. You can add an agent to [`openhands/agenthub`](./openhands/agenthub)
 to help expand the capabilities of OpenHands.

 #### Adding a new runtime
@@ -57,8 +57,8 @@ If you work for a company that provides a cloud-based runtime, you could help us
 by implementing the [interface specified here](https://github.com/All-Hands-AI/OpenHands/blob/main/openhands/runtime/base.py).

 #### Testing
-When you write code, it is also good to write tests. Please navigate to the `tests` folder to see existing test suites.
-At the moment, we have two kinds of tests: `unit` and `integration`. Please refer to the README for each test suite. These tests also run on GitHub's continuous integration to ensure quality of the project.
+When you write code, it is also good to write tests. Please navigate to the [`./tests`](./tests) folder to see existing test suites.
+At the moment, we have two kinds of tests: [`unit`](./tests/unit) and [`integration`](./evaluation/integration_tests). Please refer to the README for each test suite. These tests also run on GitHub's continuous integration to ensure quality of the project.

 ## Sending Pull Requests to OpenHands

@@ -103,7 +103,7 @@ Further, if you see an issue you like, please leave a "thumbs-up" or a comment,

 ### Making Pull Requests

-We're generally happy to consider all PRs, with the evaluation process varying based on the type of change:
+We're generally happy to consider all [PRs](https://github.com/All-Hands-AI/OpenHands/pulls), with the evaluation process varying based on the type of change:

 #### For Small Improvements

--- a/Development.md
+++ b/Development.md
@@ -100,7 +100,7 @@ poetry run pytest ./tests/unit/test_*.py
 To reduce build time (e.g., if no changes were made to the client-runtime component), you can use an existing Docker container image by
 setting the SANDBOX_RUNTIME_CONTAINER_IMAGE environment variable to the desired Docker image.

-Example: `export SANDBOX_RUNTIME_CONTAINER_IMAGE=ghcr.io/all-hands-ai/runtime:0.14-nikolaik`
+Example: `export SANDBOX_RUNTIME_CONTAINER_IMAGE=ghcr.io/all-hands-ai/runtime:0.16-nikolaik`

 ## Develop inside Docker container

--- a/README.md
+++ b/README.md
@@ -12,7 +12,7 @@
  <a href="https://codecov.io/github/All-Hands-AI/OpenHands?branch=main"><img alt="CodeCov" src="https://img.shields.io/codecov/c/github/All-Hands-AI/OpenHands?style=for-the-badge&color=blue"></a>
  <a href="https://github.com/All-Hands-AI/OpenHands/blob/main/LICENSE"><img src="https://img.shields.io/github/license/All-Hands-AI/OpenHands?style=for-the-badge&color=blue" alt="MIT License"></a>
  <br/>
-  <a href="https://join.slack.com/t/openhands-ai/shared_invite/zt-2tom0er4l-JeNUGHt_AxpEfIBstbLPiw"><img src="https://img.shields.io/badge/Slack-Join%20Us-red?logo=slack&logoColor=white&style=for-the-badge" alt="Join our Slack community"></a>
+  <a href="https://join.slack.com/t/openhands-ai/shared_invite/zt-2vbfigwev-G03twSpXaErwzYVD4CFiBg"><img src="https://img.shields.io/badge/Slack-Join%20Us-red?logo=slack&logoColor=white&style=for-the-badge" alt="Join our Slack community"></a>
  <a href="https://discord.gg/ESHStjSjD4"><img src="https://img.shields.io/badge/Discord-Join%20Us-purple?logo=discord&logoColor=white&style=for-the-badge" alt="Join our Discord community"></a>
  <a href="https://github.com/All-Hands-AI/OpenHands/blob/main/CREDITS.md"><img src="https://img.shields.io/badge/Project-Credits-blue?style=for-the-badge&color=FFE165&logo=github&logoColor=white" alt="Credits"></a>
  <br/>
@@ -29,6 +29,11 @@ call APIs, and yes—even copy code snippets from StackOverflow.

 Learn more at [docs.all-hands.dev](https://docs.all-hands.dev), or jump to the [Quick Start](#-quick-start).

+> [!IMPORTANT]
+> Using OpenHands for work? We'd love to chat! Fill out
+> [this short form](https://docs.google.com/forms/d/e/1FAIpQLSet3VbGaz8z32gW9Wm-Grl4jpt5WgMXPgJ4EDPVmCETCBpJtQ/viewform)
+> to join our Design Partner program, where you'll get early access to commercial features and the opportunity to provide input on our product roadmap.
+
 ![App screenshot](./docs/static/img/screenshot.png)

 ## ⚡ Quick Start
@@ -38,16 +43,17 @@ See the [Installation](https://docs.all-hands.dev/modules/usage/installation) gu
 system requirements and more information.

 ```bash
-docker pull docker.all-hands.dev/all-hands-ai/runtime:0.14-nikolaik
+docker pull docker.all-hands.dev/all-hands-ai/runtime:0.16-nikolaik

-docker run -it --pull=always \
-    -e SANDBOX_RUNTIME_CONTAINER_IMAGE=docker.all-hands.dev/all-hands-ai/runtime:0.14-nikolaik \
+docker run -it --rm --pull=always \
+    -e SANDBOX_RUNTIME_CONTAINER_IMAGE=docker.all-hands.dev/all-hands-ai/runtime:0.16-nikolaik \
    -e LOG_ALL_EVENTS=true \
    -v /var/run/docker.sock:/var/run/docker.sock \
+    -v ~/.openhands:/home/openhands/.openhands \
    -p 3000:3000 \
    --add-host host.docker.internal:host-gateway \
    --name openhands-app \
-    docker.all-hands.dev/all-hands-ai/openhands:0.14
+    docker.all-hands.dev/all-hands-ai/openhands:0.16
 ```

 You'll find OpenHands running at [http://localhost:3000](http://localhost:3000)!
@@ -65,6 +71,14 @@ or run it on tagged issues with [a github action](https://github.com/All-Hands-A

 Visit [Installation](https://docs.all-hands.dev/modules/usage/installation) for more information and setup instructions.

+> [!CAUTION]
+> OpenHands is meant to be run by a single user on their local workstation.
+> It is not appropriate for multi-tenant deployments, where multiple users share the same instance--there is no built-in isolation or scalability.
+>
+> If you're interested in running OpenHands in a multi-tenant environment, please
+> [get in touch with us](https://docs.google.com/forms/d/e/1FAIpQLSet3VbGaz8z32gW9Wm-Grl4jpt5WgMXPgJ4EDPVmCETCBpJtQ/viewform)
+> for advanced deployment options.
+
 If you want to modify the OpenHands source code, check out [Development.md](https://github.com/All-Hands-AI/OpenHands/blob/main/Development.md).

 Having issues? The [Troubleshooting Guide](https://docs.all-hands.dev/modules/usage/troubleshooting) can help.
@@ -82,7 +96,7 @@ troubleshooting resources, and advanced configuration options.
 OpenHands is a community-driven project, and we welcome contributions from everyone. We do most of our communication
 through Slack, so this is the best place to start, but we also are happy to have you contact us on Discord or Github:

- [Join our Slack workspace](https://join.slack.com/t/openhands-ai/shared_invite/zt-2tom0er4l-JeNUGHt_AxpEfIBstbLPiw) - Here we talk about research, architecture, and future development.
+- [Join our Slack workspace](https://join.slack.com/t/openhands-ai/shared_invite/zt-2vbfigwev-G03twSpXaErwzYVD4CFiBg) - Here we talk about research, architecture, and future development.
 - [Join our Discord server](https://discord.gg/ESHStjSjD4) - This is a community-run server for general discussion, questions, and feedback.
 - [Read or post Github Issues](https://github.com/All-Hands-AI/OpenHands/issues) - Check out the issues we're working on, or add your own ideas.

--- a/compose.yml
+++ b/compose.yml
@@ -7,7 +7,7 @@ services:
    image: openhands:latest
    container_name: openhands-app-${DATE:-}
    environment:
-      - SANDBOX_RUNTIME_CONTAINER_IMAGE=${SANDBOX_RUNTIME_CONTAINER_IMAGE:-ghcr.io/all-hands-ai/runtime:0.14-nikolaik}
+      - SANDBOX_RUNTIME_CONTAINER_IMAGE=${SANDBOX_RUNTIME_CONTAINER_IMAGE:-ghcr.io/all-hands-ai/runtime:0.16-nikolaik}
      - SANDBOX_USER_ID=${SANDBOX_USER_ID:-1234}
      - WORKSPACE_MOUNT_PATH=${WORKSPACE_BASE:-$PWD/workspace}
    ports:
--- a/config.template.toml
+++ b/config.template.toml
@@ -95,10 +95,10 @@ workspace_base = "./workspace"
 # AWS secret access key
 #aws_secret_access_key = ""

-# API key to use
+# API key to use (For Headless / CLI only -  In Web this is overridden by Session Init)
 api_key = "your-api-key"

-# API base URL
+# API base URL (For Headless / CLI only -  In Web this is overridden by Session Init)
 #base_url = ""

 # API version
@@ -131,7 +131,7 @@ embedding_model = "local"
 # Maximum number of output tokens
 #max_output_tokens = 0

-# Model to use
+# Model to use. (For Headless / CLI only -  In Web this is overridden by Session Init)
 model = "gpt-4o"

 # Number of retries to attempt when an operation fails with the LLM.
@@ -154,6 +154,10 @@ model = "gpt-4o"
 # Drop any unmapped (unsupported) params without causing an exception
 #drop_params = false

+# Modify params for litellm to do transformations like adding a default message, when a message is empty.
+# Note: this setting is global, unlike drop_params, it cannot be overridden in each call to litellm.
+#modify_params = true
+
 # Using the prompt caching feature if provided by the LLM and supported
 #caching_prompt = true

@@ -172,6 +176,10 @@ model = "gpt-4o"
 # If model is vision capable, this option allows to disable image processing (useful for cost reduction).
 #disable_vision = true

+# Custom tokenizer to use for token counting
+# https://docs.litellm.ai/docs/completion/token_usage
+#custom_tokenizer = ""
+
 [llm.gpt4o-mini]
 api_key = "your-api-key"
 model = "gpt-4o"
@@ -217,6 +225,9 @@ llm_config = 'gpt3'
 # Use host network
 #use_host_network = false

+# runtime extra build args
+#runtime_extra_build_args = ["--network=host", "--add-host=host.docker.internal:host-gateway"]
+
 # Enable auto linting after editing
 #enable_auto_lint = false

@@ -237,10 +248,10 @@ llm_config = 'gpt3'
 ##############################################################################
 [security]

-# Enable confirmation mode
+# Enable confirmation mode (For Headless / CLI only -  In Web this is overridden by Session Init)
 #confirmation_mode = false

-# The security analyzer to use
+# The security analyzer to use (For Headless / CLI only -  In Web this is overridden by Session Init)
 #security_analyzer = ""

 #################################### Eval ####################################
--- a/containers/app/Dockerfile
+++ b/containers/app/Dockerfile
@@ -42,6 +42,8 @@ ENV USE_HOST_NETWORK=false
 ENV WORKSPACE_BASE=/opt/workspace_base
 ENV OPENHANDS_BUILD_VERSION=$OPENHANDS_BUILD_VERSION
 ENV SANDBOX_USER_ID=0
+ENV FILE_STORE=local
+ENV FILE_STORE_PATH=~/.openhands
 RUN mkdir -p $WORKSPACE_BASE

 RUN apt-get update -y \
--- a/containers/dev/README.md
+++ b/containers/dev/README.md
@@ -1,5 +1,8 @@
 # Develop in Docker

+> [!WARNING]
+> This is not officially supported and may not work.
+
 Install [Docker](https://docs.docker.com/engine/install/) on your host machine and run:

 ```bash
--- a/containers/dev/compose.yml
+++ b/containers/dev/compose.yml
@@ -11,7 +11,7 @@ services:
      - BACKEND_HOST=${BACKEND_HOST:-"0.0.0.0"}
      - SANDBOX_API_HOSTNAME=host.docker.internal
      #
-      - SANDBOX_RUNTIME_CONTAINER_IMAGE=${SANDBOX_RUNTIME_CONTAINER_IMAGE:-ghcr.io/all-hands-ai/runtime:0.14-nikolaik}
+      - SANDBOX_RUNTIME_CONTAINER_IMAGE=${SANDBOX_RUNTIME_CONTAINER_IMAGE:-ghcr.io/all-hands-ai/runtime:0.16-nikolaik}
      - SANDBOX_USER_ID=${SANDBOX_USER_ID:-1234}
      - WORKSPACE_MOUNT_PATH=${WORKSPACE_BASE:-$PWD/workspace}
    ports:
--- a/docs/i18n/fr/docusaurus-plugin-content-docs/current/usage/about.md
+++ b/docs/i18n/fr/docusaurus-plugin-content-docs/current/usage/about.md
@@ -27,7 +27,7 @@ Pour plus de détails, veuillez consulter [ce document](https://github.com/All-H

 Nous avons à la fois un espace de travail Slack pour la collaboration sur la construction d'OpenHands et un serveur Discord pour discuter de tout ce qui est lié, par exemple, à ce projet, LLM, agent, etc.

- [Espace de travail Slack](https://join.slack.com/t/opendevin/shared_invite/zt-2oikve2hu-UDxHeo8nsE69y6T7yFX_BA)
+- [Espace de travail Slack](https://join.slack.com/t/openhands-ai/shared_invite/zt-2vbfigwev-G03twSpXaErwzYVD4CFiBg)
 - [Serveur Discord](https://discord.gg/ESHStjSjD4)

 Si vous souhaitez contribuer, n'hésitez pas à rejoindre notre communauté. Simplifions ensemble l'ingénierie logicielle !
--- a/docs/i18n/fr/docusaurus-plugin-content-docs/current/usage/how-to/openshift-example.md
+++ b/docs/i18n/fr/docusaurus-plugin-content-docs/current/usage/how-to/openshift-example.md
@@ -1,338 +0,0 @@
-
-
-# Kubernetes
-
-Il existe différentes façons d'exécuter OpenHands sur Kubernetes ou OpenShift. Ce guide présente une façon possible :
-1. Créer un PV "en tant qu'administrateur du cluster" pour mapper les données workspace_base et le répertoire docker au pod via le nœud worker
-2. Créer un PVC pour pouvoir monter ces PV sur le pod
-3. Créer un pod qui contient deux conteneurs : les conteneurs OpenHands et Sandbox
-
-## Étapes détaillées pour l'exemple ci-dessus
-
-> Remarque : Assurez-vous d'être connecté au cluster avec le compte approprié pour chaque étape. La création de PV nécessite un administrateur de cluster !
-
-> Assurez-vous d'avoir les autorisations de lecture/écriture sur le hostPath utilisé ci-dessous (c'est-à-dire /tmp/workspace)
-
-1. Créer le PV :
-Le fichier yaml d'exemple ci-dessous peut être utilisé par un administrateur de cluster pour créer le PV.
- workspace-pv.yaml
-
-```yamlfile
-apiVersion: v1
-kind: PersistentVolume
-metadata:
-  name: workspace-pv
-spec:
-  capacity:
-    storage: 2Gi
-  accessModes:
-    - ReadWriteOnce
-  persistentVolumeReclaimPolicy: Retain
-  hostPath:
-    path: /tmp/workspace
-```
-
-```bash
-# appliquer le fichier yaml
-$ oc create -f workspace-pv.yaml
-persistentvolume/workspace-pv created
-
-# vérifier :
-$ oc get pv
-NAME                                       CAPACITY   ACCESS MODES   RECLAIM POLICY   STATUS      CLAIM                STORAGECLASS     REASON   AGE
-workspace-pv                               2Gi        RWO            Retain           Available                                                  7m23s
-```
-
- docker-pv.yaml
-
-```yamlfile
-apiVersion: v1
-kind: PersistentVolume
-metadata:
-  name: docker-pv
-spec:
-  capacity:
-    storage: 2Gi
-  accessModes:
-    - ReadWriteOnce
-  persistentVolumeReclaimPolicy: Retain
-  hostPath:
-    path: /var/run/docker.sock
-```
-
-```bash
-# appliquer le fichier yaml
-$ oc create -f docker-pv.yaml
-persistentvolume/docker-pv created
-
-# vérifier :
-oc get pv
-NAME                                       CAPACITY   ACCESS MODES   RECLAIM POLICY   STATUS      CLAIM                STORAGECLASS     REASON   AGE
-docker-pv                                  2Gi        RWO            Retain           Available                                                  6m55s
-workspace-pv                               2Gi        RWO            Retain           Available                                                  7m23s
-```
-
-2. Créer le PVC :
-Exemple de fichier yaml PVC ci-dessous :
-
- workspace-pvc.yaml
-
-```yamlfile
-apiVersion: v1
-kind: PersistentVolumeClaim
-metadata:
-  name: workspace-pvc
-spec:
-  accessModes:
-    - ReadWriteOnce
-  resources:
-    requests:
-      storage: 1Gi
-```
-
-```bash
-# créer le pvc
-$ oc create -f workspace-pvc.yaml
-persistentvolumeclaim/workspace-pvc created
-
-# vérifier
-$ oc get pvc
-NAME            STATUS    VOLUME   CAPACITY   ACCESS MODES   STORAGECLASS     AGE
-workspace-pvc   Pending                                      hcloud-volumes   4s
-
-$ oc get events
-LAST SEEN   TYPE     REASON                 OBJECT                                MESSAGE
-8s          Normal   WaitForFirstConsumer   persistentvolumeclaim/workspace-pvc   waiting for first consumer to be created before binding
-```
-
- docker-pvc.yaml
-
-```yamlfile
-apiVersion: v1
-kind: PersistentVolumeClaim
-metadata:
-  name: docker-pvc
-spec:
-  accessModes:
-    - ReadWriteOnce
-  resources:
-    requests:
-      storage: 1Gi
-```
-
-```bash
-# créer le pvc
-$ oc create -f docker-pvc.yaml
-persistentvolumeclaim/docker-pvc created
-
-# vérifier
-$ oc get pvc
-NAME            STATUS    VOLUME   CAPACITY   ACCESS MODES   STORAGECLASS     AGE
-docker-pvc      Pending                                      hcloud-volumes   4s
-workspace-pvc   Pending                                      hcloud-volumes   2m53s
-
-$ oc get events
-LAST SEEN   TYPE     REASON                 OBJECT                                MESSAGE
-10s         Normal   WaitForFirstConsumer   persistentvolumeclaim/docker-pvc      waiting for first consumer to be created before binding
-10s         Normal   WaitForFirstConsumer   persistentvolumeclaim/workspace-pvc   waiting for first consumer to be created before binding
-```
-
-3. Créer le fichier yaml du pod :
-Exemple de fichier yaml de pod ci-dessous :
-
- pod.yaml
-
-```yamlfile
-apiVersion: v1
-kind: Pod
-metadata:
-  name: openhands-app-2024
-  labels:
-    app: openhands-app-2024
-spec:
-  containers:
-  - name: openhands-app-2024
-    image: ghcr.io/all-hands-ai/openhands:main
-    env:
-    - name: SANDBOX_USER_ID
-      value: "1000"
-    - name: WORKSPACE_MOUNT_PATH
-      value: "/opt/workspace_base"
-    volumeMounts:
-    - name: workspace-volume
-      mountPath: /opt/workspace_base
-    - name: docker-sock
-      mountPath: /var/run/docker.sock
-    ports:
-    - containerPort: 3000
-  - name: openhands-sandbox-2024
-    image: ghcr.io/all-hands-ai/sandbox:main
-    ports:
-    - containerPort: 51963
-    command: ["/usr/sbin/sshd", "-D", "-p 51963", "-o", "PermitRootLogin=yes"]
-  volumes:
-  - name: workspace-volume
-    persistentVolumeClaim:
-      claimName: workspace-pvc
-  - name: docker-sock
-    persistentVolumeClaim:
-      claimName: docker-pvc
-```
-
-
-```bash
-# créer le pod
-$ oc create -f pod.yaml
-W0716 11:22:07.776271  107626 warnings.go:70] would violate PodSecurity "restricted:v1.24": allowPrivilegeEscalation != false (containers "openhands-app-2024", "openhands-sandbox-2024" must set securityContext.allowPrivilegeEscalation=false), unrestricted capabilities (containers "openhands-app-2024", "openhands-sandbox-2024" must set securityContext.capabilities.drop=["ALL"]), runAsNonRoot != true (pod or containers "openhands-app-2024", "openhands-sandbox-2024" must set securityContext.runAsNonRoot=true), seccompProfile (pod or containers "openhands-app-2024", "openhands-sandbox-2024" must set securityContext.seccompProfile.type to "RuntimeDefault" or "Localhost")
-pod/openhands-app-2024 created
-
-# L'avertissement ci-dessus peut être ignoré pour l'instant car nous ne modifierons pas les restrictions SCC.
-
-# vérifier
-$ oc get pods
-NAME                 READY   STATUS    RESTARTS   AGE
-openhands-app-2024   0/2     Pending   0          5s
-
-$ oc get pods
-NAME                 READY   STATUS              RESTARTS   AGE
-openhands-app-2024   0/2     ContainerCreating   0          15s
-
-$ oc get events
-LAST SEEN   TYPE     REASON                   OBJECT                                MESSAGE
-38s         Normal   WaitForFirstConsumer     persistentvolumeclaim/docker-pvc      waiting for first consumer to be created before binding
-23s         Normal   ExternalProvisioning     persistentvolumeclaim/docker-pvc      waiting for a volume to be created, either by external provisioner "csi.hetzner.cloud" or manually created by system administrator
-27s         Normal   Provisioning             persistentvolumeclaim/docker-pvc      External provisioner is provisioning volume for claim "openhands/docker-pvc"
-17s         Normal   ProvisioningSucceeded    persistentvolumeclaim/docker-pvc      Successfully provisioned volume pvc-2b1d223a-1c8f-4990-8e3d-68061a9ae252
-16s         Normal   Scheduled                pod/openhands-app-2024                Successfully assigned All-Hands-AI/OpenHands-app-2024 to worker1.hub.internal.blakane.com
-9s          Normal   SuccessfulAttachVolume   pod/openhands-app-2024                AttachVolume.Attach succeeded for volume "pvc-2b1d223a-1c8f-4990-8e3d-68061a9ae252"
-9s          Normal   SuccessfulAttachVolume   pod/openhands-app-2024                AttachVolume.Attach succeeded for volume "pvc-31f15b25-faad-4665-a25f-201a530379af"
-6s          Normal   AddedInterface           pod/openhands-app-2024                Add eth0 [10.128.2.48/23] from openshift-sdn
-6s          Normal   Pulled                   pod/openhands-app-2024                Container image "ghcr.io/all-hands-ai/openhands:main" already present on machine
-6s          Normal   Created                  pod/openhands-app-2024                Created container openhands-app-2024
-6s          Normal   Started                  pod/openhands-app-2024                Started container openhands-app-2024
-6s          Normal   Pulled                   pod/openhands-app-2024                Container image "ghcr.io/all-hands-ai/sandbox:main" already present on machine
-5s          Normal   Created                  pod/openhands-app-2024                Created container openhands-sandbox-2024
-5s          Normal   Started                  pod/openhands-app-2024                Started container openhands-sandbox-2024
-83s         Normal   WaitForFirstConsumer     persistentvolumeclaim/workspace-pvc   waiting for first consumer to be created before binding
-27s         Normal   Provisioning             persistentvolumeclaim/workspace-pvc   External provisioner is provisioning volume for claim "openhands/workspace-pvc"
-17s         Normal   ProvisioningSucceeded    persistentvolumeclaim/workspace-pvc   Successfully provisioned volume pvc-31f15b25-faad-4665-a25f-201a530379af
-
-$ oc get pods
-NAME                 READY   STATUS    RESTARTS   AGE
-openhands-app-2024   2/2     Running   0          23s
-
-$ oc get pvc
-NAME            STATUS   VOLUME                                     CAPACITY   ACCESS MODES   STORAGECLASS     AGE
-docker-pvc      Bound    pvc-2b1d223a-1c8f-4990-8e3d-68061a9ae252   10Gi       RWO            hcloud-volumes   10m
-workspace-pvc   Bound    pvc-31f15b25-faad-4665-a25f-201a530379af   10Gi       RWO            hcloud-volumes   13m
-
-```
-
-4. Créer un service NodePort.
-Exemple de commande de création de service ci-dessous :
-
-```bash
-# créer le service de type NodePort
-$ oc create svc nodeport  openhands-app-2024  --tcp=3000:3000
-service/openhands-app-2024 created
-
-# vérifier
-
-$ oc get svc
-NAME                 TYPE       CLUSTER-IP      EXTERNAL-IP   PORT(S)          AGE
-openhands-app-2024   NodePort   172.30.225.42   <none>        3000:30495/TCP   4s
-
-$ oc describe svc openhands-app-2024
-Name:                     openhands-app-2024
-Namespace:                openhands
-Labels:                   app=openhands-app-2024
-Annotations:              <none>
-Selector:                 app=openhands-app-2024
-Type:                     NodePort
-IP Family Policy:         SingleStack
-IP Families:              IPv4
-IP:                       172.30.225.42
-IPs:                      172.30.225.42
-Port:                     3000-3000  3000/TCP
-TargetPort:               3000/TCP
-NodePort:                 3000-3000  30495/TCP
-Endpoints:                10.128.2.48:3000
-Session Affinity:         None
-External Traffic Policy:  Cluster
-Events:                   <none>
-```
-
-6. Se connecter à l'interface utilisateur d'OpenHands, configurer l'Agent, puis tester :
-
-![image](https://github.com/user-attachments/assets/12f94804-a0c7-4744-b873-e003c9caf40e)
-
-
-
-## Déploiement d'Openhands sur GCP GKE
-
-**Avertissement** : ce déploiement accorde à l'application OpenHands l'accès au socket docker de Kubernetes, ce qui crée un risque de sécurité. Utilisez à vos propres risques.
-1- Créer une politique pour l'accès privilégié
-2- Créer des informations d'identification gke (facultatif)
-3- Créer le déploiement openhands
-4- Commandes de vérification et d'accès à l'interface utilisateur
-5- Dépanner le pod pour vérifier le conteneur interne
-
-1. créer une politique pour l'accès privilégié
-```bash
-apiVersion: rbac.authorization.k8s.io/v1
-kind: ClusterRole
-metadata:
-  name: privileged-role
-rules:
- apiGroups: [""]
-  resources: ["pods"]
-  verbs: ["create", "get", "list", "watch", "delete"]
- apiGroups: ["apps"]
-  resources: ["deployments"]
-  verbs: ["create", "get", "list", "watch", "delete"]
- apiGroups: [""]
-  resources: ["pods/exec"]
-  verbs: ["create"]
- apiGroups: [""]
-  resources: ["pods/log"]
-  verbs: ["get"]
---
-apiVersion: rbac.authorization.k8s.io/v1
-kind: ClusterRoleBinding
-metadata:
-  name: privileged-role-binding
-roleRef:
-  apiGroup: rbac.authorization.k8s.io
-  kind: ClusterRole
-  name: privileged-role
-subjects:
- kind: ServiceAccount
-  name: default  # Remplacez par le nom de votre compte de service
-  namespace: default
-```
-2. créer des informations d'identification gke (facultatif)
-```bash
-kubectl create secret generic google-cloud-key \
-  --from-file=key.json=/path/to/your/google-cloud-key.json
-  ```
-3. créer le déploiement openhands
-## comme cela est testé pour le nœud worker unique, si vous en avez plusieurs, spécifiez l'indicateur pour le worker unique
-
-```bash
-kind: Deployment
-metadata:
-  name: openhands-app-2024
-  labels:
-    app: openhands-app-2024
-spec:
-  replicas: 1  # Vous pouvez augmenter ce nombre pour plusieurs réplicas
-  selector:
-    matchLabels:
-      app: openhands-app-2024
-  template:
-    metadata:
-      labels:
-        app: openhands-app-2024
-    spec:
-      containers:
-      -
--- a/docs/i18n/fr/docusaurus-plugin-content-docs/current/usage/troubleshooting/troubleshooting.md
+++ b/docs/i18n/fr/docusaurus-plugin-content-docs/current/usage/troubleshooting/troubleshooting.md
@@ -9,7 +9,6 @@ Si vous trouvez plus d'informations ou une solution de contournement pour l'un d
 :::tip
 OpenHands ne prend en charge Windows que via [WSL](https://learn.microsoft.com/en-us/windows/wsl/install).
 Veuillez vous assurer d'exécuter toutes les commandes à l'intérieur de votre terminal WSL.
-Consultez les [Notes pour les utilisateurs de WSL sur Windows](troubleshooting/windows) pour des guides de dépannage.
 :::

 ## Problèmes courants
--- a/docs/i18n/fr/docusaurus-plugin-content-docs/current/usage/troubleshooting/windows.md
+++ b/docs/i18n/fr/docusaurus-plugin-content-docs/current/usage/troubleshooting/windows.md
@@ -1,66 +0,0 @@
-
-
-# Notes pour les utilisateurs de WSL sur Windows
-
-OpenHands ne prend en charge Windows que via [WSL](https://learn.microsoft.com/en-us/windows/wsl/install).
-Veuillez vous assurer d'exécuter toutes les commandes dans votre terminal WSL.
-
-## Dépannage
-
-### Recommandation : Ne pas exécuter en tant qu'utilisateur root
-
-Pour des raisons de sécurité, il est fortement recommandé de ne pas exécuter OpenHands en tant qu'utilisateur root, mais en tant qu'utilisateur avec un UID non nul.
-
-Références :
-
-* [Pourquoi il est mauvais de se connecter en tant que root](https://askubuntu.com/questions/16178/why-is-it-bad-to-log-in-as-root)
-* [Définir l'utilisateur par défaut dans WSL](https://www.tenforums.com/tutorials/128152-set-default-user-windows-subsystem-linux-distro-windows-10-a.html#option2)
-Astuce concernant la 2ème référence : pour les utilisateurs d'Ubuntu, la commande pourrait en fait être "ubuntupreview" au lieu de "ubuntu".
-
---
-### Erreur : 'docker' n'a pas pu être trouvé dans cette distribution WSL 2.
-
-Si vous utilisez Docker Desktop, assurez-vous de le démarrer avant d'appeler toute commande docker depuis WSL.
-Docker doit également avoir l'option d'intégration WSL activée.
-
---
-### Installation de Poetry
-
-* Si vous rencontrez des problèmes pour exécuter Poetry même après l'avoir installé pendant le processus de build, vous devrez peut-être ajouter son chemin binaire à votre environnement :
-
-```sh
-export PATH="$HOME/.local/bin:$PATH"
-```
-
-* Si make build s'arrête sur une erreur comme celle-ci :
-
-```sh
-ModuleNotFoundError: no module named <module-name>
-```
-
-Cela pourrait être un problème avec le cache de Poetry.
-Essayez d'exécuter ces 2 commandes l'une après l'autre :
-
-```sh
-rm -r ~/.cache/pypoetry
-make build
-```
-
---
-### L'objet NoneType n'a pas d'attribut 'request'
-
-Si vous rencontrez des problèmes liés au réseau, tels que `NoneType object has no attribute 'request'` lors de l'exécution de `make run`, vous devrez peut-être configurer les paramètres réseau de WSL2. Suivez ces étapes :
-
-* Ouvrez ou créez le fichier `.wslconfig` situé à `C:\Users\%username%\.wslconfig` sur votre machine hôte Windows.
-* Ajoutez la configuration suivante au fichier `.wslconfig` :
-
-```sh
-[wsl2]
-networkingMode=mirrored
-localhostForwarding=true
-```
-
-* Enregistrez le fichier `.wslconfig`.
-* Redémarrez complètement WSL2 en quittant toutes les instances WSL2 en cours d'exécution et en exécutant la commande `wsl --shutdown` dans votre invite de commande ou terminal.
-* Après avoir redémarré WSL, essayez d'exécuter à nouveau `make run`.
-Le problème de réseau devrait être résolu.
--- a/docs/i18n/zh-Hans/docusaurus-plugin-content-docs/current/usage/about.md
+++ b/docs/i18n/zh-Hans/docusaurus-plugin-content-docs/current/usage/about.md
@@ -27,7 +27,7 @@ OpenHands 是一个社区驱动的项目，我们欢迎每个人的贡献。无

 我们有 Slack 工作区用于协作构建 OpenHands，也有 Discord 服务器用于讨论任何相关的内容，例如此项目、大语言模型、代理等。

- [Slack 工作区](https://join.slack.com/t/opendevin/shared_invite/zt-2oikve2hu-UDxHeo8nsE69y6T7yFX_BA)
+- [Slack 工作区](https://join.slack.com/t/openhands-ai/shared_invite/zt-2vbfigwev-G03twSpXaErwzYVD4CFiBg)
 - [Discord 服务器](https://discord.gg/ESHStjSjD4)

 如果你想做出贡献，欢迎加入我们的社区。让我们一起简化软件工程！
--- a/docs/i18n/zh-Hans/docusaurus-plugin-content-docs/current/usage/how-to/openshift-example.md
+++ b/docs/i18n/zh-Hans/docusaurus-plugin-content-docs/current/usage/how-to/openshift-example.md
@@ -1,343 +0,0 @@
-以下是翻译后的内容:
-
-# Kubernetes
-
-在 Kubernetes 或 OpenShift 上运行 OpenHands 有不同的方式。本指南介绍了一种可能的方式:
-1. 作为集群管理员,创建一个 PV 将 workspace_base 数据和 docker 目录映射到 worker 节点上的 pod
-2. 创建一个 PVC 以便将这些 PV 挂载到 pod
-3. 创建一个包含两个容器的 pod:OpenHands 和 Sandbox 容器
-
-## 上述示例的详细步骤
-
-> 注意:确保首先使用适当的帐户登录到集群以执行每个步骤。创建 PV 需要集群管理员权限!
-
-> 确保你对下面使用的 hostPath(即 /tmp/workspace)有读写权限
-
-1. 创建 PV:
-集群管理员可以使用下面的示例 yaml 文件创建 PV。
- workspace-pv.yaml
-
-```yamlfile
-apiVersion: v1
-kind: PersistentVolume
-metadata:
-  name: workspace-pv
-spec:
-  capacity:
-    storage: 2Gi
-  accessModes:
-    - ReadWriteOnce
-  persistentVolumeReclaimPolicy: Retain
-  hostPath:
-    path: /tmp/workspace
-```
-
-```bash
-# 应用 yaml 文件
-$ oc create -f workspace-pv.yaml
-persistentvolume/workspace-pv created
-
-# 查看:
-$ oc get pv
-NAME                                       CAPACITY   ACCESS MODES   RECLAIM POLICY   STATUS      CLAIM                STORAGECLASS     REASON   AGE
-workspace-pv                               2Gi        RWO            Retain           Available                                                  7m23s
-```
-
- docker-pv.yaml
-
-```yamlfile
-apiVersion: v1
-kind: PersistentVolume
-metadata:
-  name: docker-pv
-spec:
-  capacity:
-    storage: 2Gi
-  accessModes:
-    - ReadWriteOnce
-  persistentVolumeReclaimPolicy: Retain
-  hostPath:
-    path: /var/run/docker.sock
-```
-
-```bash
-# 应用 yaml 文件
-$ oc create -f docker-pv.yaml
-persistentvolume/docker-pv created
-
-# 查看:
-oc get pv
-NAME                                       CAPACITY   ACCESS MODES   RECLAIM POLICY   STATUS      CLAIM                STORAGECLASS     REASON   AGE
-docker-pv                                  2Gi        RWO            Retain           Available                                                  6m55s
-workspace-pv                               2Gi        RWO            Retain           Available                                                  7m23s
-```
-
-2. 创建 PVC:
-下面是示例 PVC yaml 文件:
-
- workspace-pvc.yaml
-
-```yamlfile
-apiVersion: v1
-kind: PersistentVolumeClaim
-metadata:
-  name: workspace-pvc
-spec:
-  accessModes:
-    - ReadWriteOnce
-  resources:
-    requests:
-      storage: 1Gi
-```
-
-```bash
-# 创建 pvc
-$ oc create -f workspace-pvc.yaml
-persistentvolumeclaim/workspace-pvc created
-
-# 查看
-$ oc get pvc
-NAME            STATUS    VOLUME   CAPACITY   ACCESS MODES   STORAGECLASS     AGE
-workspace-pvc   Pending                                      hcloud-volumes   4s
-
-$ oc get events
-LAST SEEN   TYPE     REASON                 OBJECT                                MESSAGE
-8s          Normal   WaitForFirstConsumer   persistentvolumeclaim/workspace-pvc   waiting for first consumer to be created before binding
-```
-
- docker-pvc.yaml
-
-```yamlfile
-apiVersion: v1
-kind: PersistentVolumeClaim
-metadata:
-  name: docker-pvc
-spec:
-  accessModes:
-    - ReadWriteOnce
-  resources:
-    requests:
-      storage: 1Gi
-```
-
-```bash
-# 创建 pvc
-$ oc create -f docker-pvc.yaml
-persistentvolumeclaim/docker-pvc created
-
-# 查看
-$ oc get pvc
-NAME            STATUS    VOLUME   CAPACITY   ACCESS MODES   STORAGECLASS     AGE
-docker-pvc      Pending                                      hcloud-volumes   4s
-workspace-pvc   Pending                                      hcloud-volumes   2m53s
-
-$ oc get events
-LAST SEEN   TYPE     REASON                 OBJECT                                MESSAGE
-10s         Normal   WaitForFirstConsumer   persistentvolumeclaim/docker-pvc      waiting for first consumer to be created before binding
-10s         Normal   WaitForFirstConsumer   persistentvolumeclaim/workspace-pvc   waiting for first consumer to be created before binding
-```
-
-3. 创建 pod yaml 文件:
-下面是示例 pod yaml 文件:
-
- pod.yaml
-
-```yamlfile
-apiVersion: v1
-kind: Pod
-metadata:
-  name: openhands-app-2024
-  labels:
-    app: openhands-app-2024
-spec:
-  containers:
-  - name: openhands-app-2024
-    image: ghcr.io/all-hands-ai/openhands:main
-    env:
-    - name: SANDBOX_USER_ID
-      value: "1000"
-    - name: WORKSPACE_MOUNT_PATH
-      value: "/opt/workspace_base"
-    volumeMounts:
-    - name: workspace-volume
-      mountPath: /opt/workspace_base
-    - name: docker-sock
-      mountPath: /var/run/docker.sock
-    ports:
-    - containerPort: 3000
-  - name: openhands-sandbox-2024
-    image: ghcr.io/all-hands-ai/sandbox:main
-    ports:
-    - containerPort: 51963
-    command: ["/usr/sbin/sshd", "-D", "-p 51963", "-o", "PermitRootLogin=yes"]
-  volumes:
-  - name: workspace-volume
-    persistentVolumeClaim:
-      claimName: workspace-pvc
-  - name: docker-sock
-    persistentVolumeClaim:
-      claimName: docker-pvc
-```
-
-
-```bash
-# 创建 pod
-$ oc create -f pod.yaml
-W0716 11:22:07.776271  107626 warnings.go:70] would violate PodSecurity "restricted:v1.24": allowPrivilegeEscalation != false (containers "openhands-app-2024", "openhands-sandbox-2024" must set securityContext.allowPrivilegeEscalation=false), unrestricted capabilities (containers "openhands-app-2024", "openhands-sandbox-2024" must set securityContext.capabilities.drop=["ALL"]), runAsNonRoot != true (pod or containers "openhands-app-2024", "openhands-sandbox-2024" must set securityContext.runAsNonRoot=true), seccompProfile (pod or containers "openhands-app-2024", "openhands-sandbox-2024" must set securityContext.seccompProfile.type to "RuntimeDefault" or "Localhost")
-pod/openhands-app-2024 created
-
-# 上面的警告可以暂时忽略,因为我们不会修改 SCC 限制。
-
-# 查看
-$ oc get pods
-NAME                 READY   STATUS    RESTARTS   AGE
-openhands-app-2024   0/2     Pending   0          5s
-
-$ oc get pods
-NAME                 READY   STATUS              RESTARTS   AGE
-openhands-app-2024   0/2     ContainerCreating   0          15s
-
-$ oc get events
-LAST SEEN   TYPE     REASON                   OBJECT                                MESSAGE
-38s         Normal   WaitForFirstConsumer     persistentvolumeclaim/docker-pvc      waiting for first consumer to be created before binding
-23s         Normal   ExternalProvisioning     persistentvolumeclaim/docker-pvc      waiting for a volume to be created, either by external provisioner "csi.hetzner.cloud" or manually created by system administrator
-27s         Normal   Provisioning             persistentvolumeclaim/docker-pvc      External provisioner is provisioning volume for claim "openhands/docker-pvc"
-17s         Normal   ProvisioningSucceeded    persistentvolumeclaim/docker-pvc      Successfully provisioned volume pvc-2b1d223a-1c8f-4990-8e3d-68061a9ae252
-16s         Normal   Scheduled                pod/openhands-app-2024                Successfully assigned All-Hands-AI/OpenHands-app-2024 to worker1.hub.internal.blakane.com
-9s          Normal   SuccessfulAttachVolume   pod/openhands-app-2024                AttachVolume.Attach succeeded for volume "pvc-2b1d223a-1c8f-4990-8e3d-68061a9ae252"
-9s          Normal   SuccessfulAttachVolume   pod/openhands-app-2024                AttachVolume.Attach succeeded for volume "pvc-31f15b25-faad-4665-a25f-201a530379af"
-6s          Normal   AddedInterface           pod/openhands-app-2024                Add eth0 [10.128.2.48/23] from openshift-sdn
-6s          Normal   Pulled                   pod/openhands-app-2024                Container image "ghcr.io/all-hands-ai/openhands:main" already present on machine
-6s          Normal   Created                  pod/openhands-app-2024                Created container openhands-app-2024
-6s          Normal   Started                  pod/openhands-app-2024                Started container openhands-app-2024
-6s          Normal   Pulled                   pod/openhands-app-2024                Container image "ghcr.io/all-hands-ai/sandbox:main" already present on machine
-5s          Normal   Created                  pod/openhands-app-2024                Created container openhands-sandbox-2024
-5s          Normal   Started                  pod/openhands-app-2024                Started container openhands-sandbox-2024
-83s         Normal   WaitForFirstConsumer     persistentvolumeclaim/workspace-pvc   waiting for first consumer to be created before binding
-27s         Normal   Provisioning             persistentvolumeclaim/workspace-pvc   External provisioner is provisioning volume for claim "openhands/workspace-pvc"
-17s         Normal   ProvisioningSucceeded    persistentvolumeclaim/workspace-pvc   Successfully provisioned volume pvc-31f15b25-faad-4665-a25f-201a530379af
-
-$ oc get pods
-NAME                 READY   STATUS    RESTARTS   AGE
-openhands-app-2024   2/2     Running   0          23s
-
-$ oc get pvc
-NAME            STATUS   VOLUME                                     CAPACITY   ACCESS MODES   STORAGECLASS     AGE
-docker-pvc      Bound    pvc-2b1d223a-1c8f-4990-8e3d-68061a9ae252   10Gi       RWO            hcloud-volumes   10m
-workspace-pvc   Bound    pvc-31f15b25-faad-4665-a25f-201a530379af   10Gi       RWO            hcloud-volumes   13m
-
-```
-
-4. 创建一个 NodePort 服务。
-下面是示例服务创建命令:
-
-```bash
-# 创建 NodePort 类型的服务
-$ oc create svc nodeport  openhands-app-2024  --tcp=3000:3000
-service/openhands-app-2024 created
-
-# 查看
-
-$ oc get svc
-NAME                 TYPE       CLUSTER-IP      EXTERNAL-IP   PORT(S)          AGE
-openhands-app-2024   NodePort   172.30.225.42   <none>        3000:30495/TCP   4s
-
-$ oc describe svc openhands-app-2024
-Name:                     openhands-app-2024
-Namespace:                openhands
-Labels:                   app=openhands-app-2024
-Annotations:              <none>
-Selector:                 app=openhands-app-2024
-Type:                     NodePort
-IP Family Policy:         SingleStack
-IP Families:              IPv4
-IP:                       172.30.225.42
-IPs:                      172.30.225.42
-Port:                     3000-3000  3000/TCP
-TargetPort:               3000/TCP
-NodePort:                 3000-3000  30495/TCP
-Endpoints:                10.128.2.48:3000
-Session Affinity:         None
-External Traffic Policy:  Cluster
-Events:                   <none>
-```
-
-6. 连接到 OpenHands UI,配置 Agent,然后测试:
-
-![image](https://github.com/user-attachments/assets/12f94804-a0c7-4744-b873-e003c9caf40e)
-
-
-
-## GCP GKE OpenHands 部署
-
-**警告**:此部署授予 OpenHands 应用程序访问 Kubernetes docker socket 的权限,这会带来安全风险。请自行决定是否使用。
-1- 创建特权访问策略
-2- 创建 gke 凭证(可选)
-3- 创建 openhands 部署
-4- 验证和 UI 访问命令
-5- 排查 pod 以验证内部容器
-
-1. 创建特权访问策略
-```bash
-apiVersion: rbac.authorization.k8s.io/v1
-kind: ClusterRole
-metadata:
-  name: privileged-role
-rules:
- apiGroups: [""]
-  resources: ["pods"]
-  verbs: ["create", "get", "list", "watch", "delete"]
- apiGroups: ["apps"]
-  resources: ["deployments"]
-  verbs: ["create", "get", "list", "watch", "delete"]
- apiGroups: [""]
-  resources: ["pods/exec"]
-  verbs: ["create"]
- apiGroups: [""]
-  resources: ["pods/log"]
-  verbs: ["get"]
---
-apiVersion: rbac.authorization.k8s.io/v1
-kind: ClusterRoleBinding
-metadata:
-  name: privileged-role-binding
-roleRef:
-  apiGroup: rbac.authorization.k8s.io
-  kind: ClusterRole
-  name: privileged-role
-subjects:
- kind: ServiceAccount
-  name: default  # 更改为你的服务帐户名称
-  namespace: default
-```
-2. 创建 gke 凭证(可选)
-```bash
-kubectl create secret generic google-cloud-key \
-  --from-file=key.json=/path/to/your/google-cloud-key.json
-  ```
-3. 创建 openhands 部署
-## 由于这是针对单个工作节点进行测试的,如果你有多个节点,请指定单个工作节点的标志
-
-```bash
-kind: Deployment
-metadata:
-  name: openhands-app-2024
-  labels:
-    app: openhands-app-2024
-spec:
-  replicas: 1  # 你可以增加这个数字以获得多个副本
-  selector:
-    matchLabels:
-      app: openhands-app-2024
-  template:
-    metadata:
-      labels:
-        app: openhands-app-2024
-    spec:
-      containers:
-      - name: openhands-app-2024
-        image: ghcr.io/all-hands-ai/openhands:main
-        env:
-        - name: SANDBOX_USER_ID
-          value: "1000"
-        - name: SANDBOX_API
--- a/docs/i18n/zh-Hans/docusaurus-plugin-content-docs/current/usage/troubleshooting/troubleshooting.md
+++ b/docs/i18n/zh-Hans/docusaurus-plugin-content-docs/current/usage/troubleshooting/troubleshooting.md
@@ -7,7 +7,6 @@
 :::tip
 OpenHands 仅通过 [WSL](https://learn.microsoft.com/en-us/windows/wsl/install) 支持 Windows。
 请确保在您的 WSL 终端内运行所有命令。
-查看 [Windows 用户的 WSL 注意事项](troubleshooting/windows) 以获取一些故障排除指南。
 :::

 ## 常见问题
--- a/docs/i18n/zh-Hans/docusaurus-plugin-content-docs/current/usage/troubleshooting/windows.md
+++ b/docs/i18n/zh-Hans/docusaurus-plugin-content-docs/current/usage/troubleshooting/windows.md
@@ -1,66 +0,0 @@
-以下是翻译后的内容:
-
-# 针对 Windows 上 WSL 用户的注意事项
-
-OpenHands 仅通过 [WSL](https://learn.microsoft.com/en-us/windows/wsl/install) 支持 Windows。
-请确保在您的 WSL 终端内运行所有命令。
-
-## 故障排除
-
-### 建议: 不要以 root 用户身份运行
-
-出于安全原因,强烈建议不要以 root 用户身份运行 OpenHands,而是以具有非零 UID 的用户身份运行。
-
-参考:
-
-* [为什么以 root 身份登录不好](https://askubuntu.com/questions/16178/why-is-it-bad-to-log-in-as-root)
-* [在 WSL 中设置默认用户](https://www.tenforums.com/tutorials/128152-set-default-user-windows-subsystem-linux-distro-windows-10-a.html#option2)
-关于第二个参考的提示:对于 Ubuntu 用户,命令实际上可能是 "ubuntupreview" 而不是 "ubuntu"。
-
---
-### 错误: 在此 WSL 2 发行版中找不到 'docker'。
-
-如果您正在使用 Docker Desktop,请确保在从 WSL 内部调用任何 docker 命令之前启动它。
-Docker 还需要激活 WSL 集成选项。
-
---
-### Poetry 安装
-
-* 如果您在构建过程中安装 Poetry 后仍然面临运行 Poetry 的问题,您可能需要将其二进制路径添加到环境中:
-
-```sh
-export PATH="$HOME/.local/bin:$PATH"
-```
-
-* 如果 make build 在如下错误上停止:
-
-```sh
-ModuleNotFoundError: no module named <module-name>
-```
-
-这可能是 Poetry 缓存的问题。
-尝试依次运行这两个命令:
-
-```sh
-rm -r ~/.cache/pypoetry
-make build
-```
-
---
-### NoneType 对象没有属性 'request'
-
-如果您在执行 `make run` 时遇到与网络相关的问题,例如 `NoneType 对象没有属性 'request'`,您可能需要配置 WSL2 网络设置。请按照以下步骤操作:
-
-* 在 Windows 主机上打开或创建位于 `C:\Users\%username%\.wslconfig` 的 `.wslconfig` 文件。
-* 将以下配置添加到 `.wslconfig` 文件中:
-
-```sh
-[wsl2]
-networkingMode=mirrored
-localhostForwarding=true
-```
-
-* 保存 `.wslconfig` 文件。
-* 通过退出任何正在运行的 WSL2 实例并在命令提示符或终端中执行 `wsl --shutdown` 命令来完全重启 WSL2。
-* 重新启动 WSL 后,再次尝试执行 `make run`。
-网络问题应该得到解决。
--- a/docs/modules/usage/how-to/cli-mode.md
+++ b/docs/modules/usage/how-to/cli-mode.md
@@ -50,7 +50,7 @@ LLM_API_KEY="sk_test_12345"
 ```bash
 docker run -it \
    --pull=always \
-    -e SANDBOX_RUNTIME_CONTAINER_IMAGE=docker.all-hands.dev/all-hands-ai/runtime:0.14-nikolaik \
+    -e SANDBOX_RUNTIME_CONTAINER_IMAGE=docker.all-hands.dev/all-hands-ai/runtime:0.16-nikolaik \
    -e SANDBOX_USER_ID=$(id -u) \
    -e WORKSPACE_MOUNT_PATH=$WORKSPACE_BASE \
    -e LLM_API_KEY=$LLM_API_KEY \
@@ -59,7 +59,7 @@ docker run -it \
    -v /var/run/docker.sock:/var/run/docker.sock \
    --add-host host.docker.internal:host-gateway \
    --name openhands-app-$(date +%Y%m%d%H%M%S) \
-    docker.all-hands.dev/all-hands-ai/openhands:0.14 \
+    docker.all-hands.dev/all-hands-ai/openhands:0.16 \
    python -m openhands.core.cli
 ```

--- a/docs/modules/usage/how-to/github-action.md
+++ b/docs/modules/usage/how-to/github-action.md
@@ -37,24 +37,33 @@ the [README for the OpenHands Resolver](https://github.com/All-Hands-AI/OpenHand

 You can provide custom directions for OpenHands by following the [README for the resolver](https://github.com/All-Hands-AI/OpenHands/blob/main/openhands/resolver/README.md#providing-custom-instructions).

-### Configure custom macro
+### Custom configurations

-To customize the default macro (`@openhands-agent`):
+Github resolver will automatically check for valid [repository secrets](https://docs.github.com/en/actions/security-for-github-actions/security-guides/using-secrets-in-github-actions?tool=webui#creating-secrets-for-a-repository) or [repository variables](https://docs.github.com/en/actions/writing-workflows/choosing-what-your-workflow-does/store-information-in-variables#creating-configuration-variables-for-a-repository) to customize its behavior.
+The customization options you can set are:

-1. [Create a repository variable](https://docs.github.com/en/actions/writing-workflows/choosing-what-your-workflow-does/store-information-in-variables#creating-configuration-variables-for-a-repository) named `OPENHANDS_MACRO`
-2. Assign the variable a custom value
+| **Attribute name**               | **Type** | **Purpose**                                                                                                 | **Example**                                          |
+|----------------------------------| -------- |-------------------------------------------------------------------------------------------------------------|------------------------------------------------------|
+| `LLM_MODEL`                      | Variable | Set the LLM to use with OpenHands                                                                           | `LLM_MODEL="anthropic/claude-3-5-sonnet-20241022"`   |
+| `OPENHANDS_MAX_ITER`             | Variable | Set max limit for agent iterations                                                                          | `OPENHANDS_MAX_ITER=10`                              |
+| `OPENHANDS_MACRO`                | Variable | Customize default macro for invoking the resolver                                                           | `OPENHANDS_MACRO=@resolveit`                         |
+| `OPENHANDS_BASE_CONTAINER_IMAGE` | Variable | Custom Sandbox ([learn more](https://docs.all-hands.dev/modules/usage/how-to/custom-sandbox-guide))         | `OPENHANDS_BASE_CONTAINER_IMAGE="custom_image"`      |

 ## Writing Effective .openhands_instructions Files

-The `.openhands_instructions` file is a file that you can put in the root directory of your repository to guide OpenHands in understanding and working with your repository effectively. Here are key tips for writing high-quality instructions:
+The `.openhands_instructions` file is a file that you can put in the root directory of your repository to guide OpenHands
+in understanding and working with your repository effectively. Here are key tips for writing high-quality instructions:

 ### Core Principles

-1. **Concise but Informative**: Provide a clear, focused overview of the repository that emphasizes the most common actions OpenHands will need to perform.
+1. **Concise but Informative**: Provide a clear, focused overview of the repository that emphasizes the most common
+     actions OpenHands will need to perform.

-2. **Repository Structure**: Explain the key directories and their purposes, especially highlighting where different types of code (e.g., frontend, backend) are located.
+2. **Repository Structure**: Explain the key directories and their purposes, especially highlighting where different
+     types of code (e.g., frontend, backend) are located.

 3. **Development Workflows**: Document the essential commands for:
+
   - Building and setting up the project
   - Running tests
   - Linting and code quality checks
@@ -69,24 +78,29 @@ The `.openhands_instructions` file is a file that you can put in the root direct

 ```markdown
 # Repository Overview
+
 [Brief description of the project]

 ## General Setup
+
 - Main build command
 - Development environment setup
 - Pre-commit checks

 ## Backend
+
 - Location and structure
 - Testing instructions
 - Environment requirements

 ## Frontend
+
 - Setup prerequisites
 - Build and test commands
 - Environment variables

 ## Additional Guidelines
+
 - Code style requirements
 - Special considerations
 - Common workflows
--- a/docs/modules/usage/how-to/gui-mode.md
+++ b/docs/modules/usage/how-to/gui-mode.md
@@ -23,10 +23,75 @@ OpenHands provides a user-friendly Graphical User Interface (GUI) mode for inter

 OpenHands automatically exports a `GITHUB_TOKEN` to the shell environment if it is available. This can happen in two ways:

-1. Locally (OSS): The user directly inputs their GitHub token.
-2. Online (SaaS): The token is obtained through GitHub OAuth authentication.
+1. **Locally (OSS)**: The user directly inputs their GitHub token
+2. **Online (SaaS)**: The token is obtained through GitHub OAuth authentication

-When you reach the `/app` route, the app checks if a token is present. If it finds one, it sets it in the environment for the agent to use.
+#### Setting Up a Local GitHub Token
+
+1. **Generate a Personal Access Token (PAT)**:
+   - Go to GitHub Settings > Developer Settings > Personal Access Tokens > Tokens (classic)
+   - Click "Generate new token (classic)"
+   - Required scopes:
+     - `repo` (Full control of private repositories)
+     - `workflow` (Update GitHub Action workflows)
+     - `read:org` (Read organization data)
+
+2. **Enter Token in OpenHands**:
+   - Click the Settings button (gear icon) in the top right
+   - Navigate to the "GitHub" section
+   - Paste your token in the "GitHub Token" field
+   - Click "Save" to apply the changes
+
+#### Organizational Token Policies
+
+If you're working with organizational repositories, additional setup may be required:
+
+1. **Check Organization Requirements**:
+   - Organization admins may enforce specific token policies
+   - Some organizations require tokens to be created with SSO enabled
+   - Review your organization's [token policy settings](https://docs.github.com/en/organizations/managing-programmatic-access-to-your-organization/setting-a-personal-access-token-policy-for-your-organization)
+
+2. **Verify Organization Access**:
+   - Go to your token settings on GitHub
+   - Look for the organization under "Organization access"
+   - If required, click "Enable SSO" next to your organization
+   - Complete the SSO authorization process
+
+#### OAuth Authentication (Online Mode)
+
+When using OpenHands in online mode, the GitHub OAuth flow:
+
+1. Requests the following permissions:
+   - Repository access (read/write)
+   - Workflow management
+   - Organization read access
+
+2. Authentication steps:
+   - Click "Sign in with GitHub" when prompted
+   - Review the requested permissions
+   - Authorize OpenHands to access your GitHub account
+   - If using an organization, authorize organization access if prompted
+
+#### Troubleshooting
+
+Common issues and solutions:
+
+1. **Token Not Recognized**:
+   - Ensure the token is properly saved in settings
+   - Check that the token hasn't expired
+   - Verify the token has the required scopes
+   - Try regenerating the token
+
+2. **Organization Access Denied**:
+   - Check if SSO is required but not enabled
+   - Verify organization membership
+   - Contact organization admin if token policies are blocking access
+
+3. **Verifying Token Works**:
+   - The app will show a green checkmark if the token is valid
+   - Try accessing a repository to confirm permissions
+   - Check the browser console for any error messages
+   - Use the "Test Connection" button in settings if available

 ### Advanced Settings

--- a/docs/modules/usage/how-to/headless-mode.md
+++ b/docs/modules/usage/how-to/headless-mode.md
@@ -44,7 +44,7 @@ LLM_API_KEY="sk_test_12345"
 ```bash
 docker run -it \
    --pull=always \
-    -e SANDBOX_RUNTIME_CONTAINER_IMAGE=docker.all-hands.dev/all-hands-ai/runtime:0.14-nikolaik \
+    -e SANDBOX_RUNTIME_CONTAINER_IMAGE=docker.all-hands.dev/all-hands-ai/runtime:0.16-nikolaik \
    -e SANDBOX_USER_ID=$(id -u) \
    -e WORKSPACE_MOUNT_PATH=$WORKSPACE_BASE \
    -e LLM_API_KEY=$LLM_API_KEY \
@@ -54,6 +54,6 @@ docker run -it \
    -v /var/run/docker.sock:/var/run/docker.sock \
    --add-host host.docker.internal:host-gateway \
    --name openhands-app-$(date +%Y%m%d%H%M%S) \
-    docker.all-hands.dev/all-hands-ai/openhands:0.14 \
-    python -m openhands.core.main -t "write a bash script that prints hi"
+    docker.all-hands.dev/all-hands-ai/openhands:0.16 \
+    python -m openhands.core.main -t "write a bash script that prints hi" --no-auto-continue
 ```
--- a/docs/modules/usage/how-to/openshift-example.md
+++ b/docs/modules/usage/how-to/openshift-example.md
@@ -1,429 +0,0 @@
-# Kubernetes
-
-There are different ways you might run OpenHands on Kubernetes or OpenShift. This guide goes through one possible way:
-1. Create a PV "as a cluster admin" to map workspace_base data and docker directory to the pod through the worker node
-2. Create a PVC to be able to mount those PVs to the pod
-3. Create a pod which contains two containers; the OpenHands and Sandbox containers
-
-## Detailed Steps for the Example Above
-
-> Note: Make sure you are logged in to the cluster first with the proper account for each step. PV creation requires cluster administrator!
-
-> Make sure you have read/write permissions on the hostPath used below (i.e. /tmp/workspace)
-
-1. Create the PV:
-Sample yaml file below can be used by a cluster admin to create the PV.
- workspace-pv.yaml
-
-```yamlfile
-apiVersion: v1
-kind: PersistentVolume
-metadata:
-  name: workspace-pv
-spec:
-  capacity:
-    storage: 2Gi
-  accessModes:
-    - ReadWriteOnce
-  persistentVolumeReclaimPolicy: Retain
-  hostPath:
-    path: /tmp/workspace
-```
-
-```bash
-# apply yaml file
-$ oc create -f workspace-pv.yaml
-persistentvolume/workspace-pv created
-
-# review:
-$ oc get pv
-NAME                                       CAPACITY   ACCESS MODES   RECLAIM POLICY   STATUS      CLAIM                STORAGECLASS     REASON   AGE
-workspace-pv                               2Gi        RWO            Retain           Available                                                  7m23s
-```
-
- docker-pv.yaml
-
-```yamlfile
-apiVersion: v1
-kind: PersistentVolume
-metadata:
-  name: docker-pv
-spec:
-  capacity:
-    storage: 2Gi
-  accessModes:
-    - ReadWriteOnce
-  persistentVolumeReclaimPolicy: Retain
-  hostPath:
-    path: /var/run/docker.sock
-```
-
-```bash
-# apply yaml file
-$ oc create -f docker-pv.yaml
-persistentvolume/docker-pv created
-
-# review:
-oc get pv
-NAME                                       CAPACITY   ACCESS MODES   RECLAIM POLICY   STATUS      CLAIM                STORAGECLASS     REASON   AGE
-docker-pv                                  2Gi        RWO            Retain           Available                                                  6m55s
-workspace-pv                               2Gi        RWO            Retain           Available                                                  7m23s
-```
-
-2. Create the PVC:
-Sample PVC yaml file below:
-
- workspace-pvc.yaml
-
-```yamlfile
-apiVersion: v1
-kind: PersistentVolumeClaim
-metadata:
-  name: workspace-pvc
-spec:
-  accessModes:
-    - ReadWriteOnce
-  resources:
-    requests:
-      storage: 1Gi
-```
-
-```bash
-# create the pvc
-$ oc create -f workspace-pvc.yaml
-persistentvolumeclaim/workspace-pvc created
-
-# review
-$ oc get pvc
-NAME            STATUS    VOLUME   CAPACITY   ACCESS MODES   STORAGECLASS     AGE
-workspace-pvc   Pending                                      hcloud-volumes   4s
-
-$ oc get events
-LAST SEEN   TYPE     REASON                 OBJECT                                MESSAGE
-8s          Normal   WaitForFirstConsumer   persistentvolumeclaim/workspace-pvc   waiting for first consumer to be created before binding
-```
-
- docker-pvc.yaml
-
-```yamlfile
-apiVersion: v1
-kind: PersistentVolumeClaim
-metadata:
-  name: docker-pvc
-spec:
-  accessModes:
-    - ReadWriteOnce
-  resources:
-    requests:
-      storage: 1Gi
-```
-
-```bash
-# create pvc
-$ oc create -f docker-pvc.yaml
-persistentvolumeclaim/docker-pvc created
-
-# review
-$ oc get pvc
-NAME            STATUS    VOLUME   CAPACITY   ACCESS MODES   STORAGECLASS     AGE
-docker-pvc      Pending                                      hcloud-volumes   4s
-workspace-pvc   Pending                                      hcloud-volumes   2m53s
-
-$ oc get events
-LAST SEEN   TYPE     REASON                 OBJECT                                MESSAGE
-10s         Normal   WaitForFirstConsumer   persistentvolumeclaim/docker-pvc      waiting for first consumer to be created before binding
-10s         Normal   WaitForFirstConsumer   persistentvolumeclaim/workspace-pvc   waiting for first consumer to be created before binding
-```
-
-3. Create the pod yaml file:
-Sample pod yaml file below:
-
- pod.yaml
-
-```yamlfile
-apiVersion: v1
-kind: Pod
-metadata:
-  name: openhands-app-2024
-  labels:
-    app: openhands-app-2024
-spec:
-  containers:
-  - name: openhands-app-2024
-    image: docker.all-hands.dev/all-hands-ai/openhands:main
-    env:
-    - name: SANDBOX_USER_ID
-      value: "1000"
-    - name: WORKSPACE_MOUNT_PATH
-      value: "/opt/workspace_base"
-    volumeMounts:
-    - name: workspace-volume
-      mountPath: /opt/workspace_base
-    - name: docker-sock
-      mountPath: /var/run/docker.sock
-    ports:
-    - containerPort: 3000
-  - name: openhands-sandbox-2024
-    image: docker.all-hands.dev/all-hands-ai/runtime:main
-    ports:
-    - containerPort: 51963
-    command: ["/usr/sbin/sshd", "-D", "-p 51963", "-o", "PermitRootLogin=yes"]
-  volumes:
-  - name: workspace-volume
-    persistentVolumeClaim:
-      claimName: workspace-pvc
-  - name: docker-sock
-    persistentVolumeClaim:
-      claimName: docker-pvc
-```
-
-
-```bash
-# create the pod
-$ oc create -f pod.yaml
-W0716 11:22:07.776271  107626 warnings.go:70] would violate PodSecurity "restricted:v1.24": allowPrivilegeEscalation != false (containers "openhands-app-2024", "openhands-sandbox-2024" must set securityContext.allowPrivilegeEscalation=false), unrestricted capabilities (containers "openhands-app-2024", "openhands-sandbox-2024" must set securityContext.capabilities.drop=["ALL"]), runAsNonRoot != true (pod or containers "openhands-app-2024", "openhands-sandbox-2024" must set securityContext.runAsNonRoot=true), seccompProfile (pod or containers "openhands-app-2024", "openhands-sandbox-2024" must set securityContext.seccompProfile.type to "RuntimeDefault" or "Localhost")
-pod/openhands-app-2024 created
-
-# Above warning can be ignored for now as we will not modify SCC restrictions.
-
-# review
-$ oc get pods
-NAME                 READY   STATUS    RESTARTS   AGE
-openhands-app-2024   0/2     Pending   0          5s
-
-$ oc get pods
-NAME                 READY   STATUS              RESTARTS   AGE
-openhands-app-2024   0/2     ContainerCreating   0          15s
-
-$ oc get events
-LAST SEEN   TYPE     REASON                   OBJECT                                MESSAGE
-38s         Normal   WaitForFirstConsumer     persistentvolumeclaim/docker-pvc      waiting for first consumer to be created before binding
-23s         Normal   ExternalProvisioning     persistentvolumeclaim/docker-pvc      waiting for a volume to be created, either by external provisioner "csi.hetzner.cloud" or manually created by system administrator
-27s         Normal   Provisioning             persistentvolumeclaim/docker-pvc      External provisioner is provisioning volume for claim "openhands/docker-pvc"
-17s         Normal   ProvisioningSucceeded    persistentvolumeclaim/docker-pvc      Successfully provisioned volume pvc-2b1d223a-1c8f-4990-8e3d-68061a9ae252
-16s         Normal   Scheduled                pod/openhands-app-2024                Successfully assigned All-Hands-AI/OpenHands-app-2024 to worker1.hub.internal.blakane.com
-9s          Normal   SuccessfulAttachVolume   pod/openhands-app-2024                AttachVolume.Attach succeeded for volume "pvc-2b1d223a-1c8f-4990-8e3d-68061a9ae252"
-9s          Normal   SuccessfulAttachVolume   pod/openhands-app-2024                AttachVolume.Attach succeeded for volume "pvc-31f15b25-faad-4665-a25f-201a530379af"
-6s          Normal   AddedInterface           pod/openhands-app-2024                Add eth0 [10.128.2.48/23] from openshift-sdn
-6s          Normal   Pulled                   pod/openhands-app-2024                Container image "docker.all-hands.dev/all-hands-ai/openhands:main" already present on machine
-6s          Normal   Created                  pod/openhands-app-2024                Created container openhands-app-2024
-6s          Normal   Started                  pod/openhands-app-2024                Started container openhands-app-2024
-6s          Normal   Pulled                   pod/openhands-app-2024                Container image "docker.all-hands.dev/all-hands-ai/sandbox:main" already present on machine
-5s          Normal   Created                  pod/openhands-app-2024                Created container openhands-sandbox-2024
-5s          Normal   Started                  pod/openhands-app-2024                Started container openhands-sandbox-2024
-83s         Normal   WaitForFirstConsumer     persistentvolumeclaim/workspace-pvc   waiting for first consumer to be created before binding
-27s         Normal   Provisioning             persistentvolumeclaim/workspace-pvc   External provisioner is provisioning volume for claim "openhands/workspace-pvc"
-17s         Normal   ProvisioningSucceeded    persistentvolumeclaim/workspace-pvc   Successfully provisioned volume pvc-31f15b25-faad-4665-a25f-201a530379af
-
-$ oc get pods
-NAME                 READY   STATUS    RESTARTS   AGE
-openhands-app-2024   2/2     Running   0          23s
-
-$ oc get pvc
-NAME            STATUS   VOLUME                                     CAPACITY   ACCESS MODES   STORAGECLASS     AGE
-docker-pvc      Bound    pvc-2b1d223a-1c8f-4990-8e3d-68061a9ae252   10Gi       RWO            hcloud-volumes   10m
-workspace-pvc   Bound    pvc-31f15b25-faad-4665-a25f-201a530379af   10Gi       RWO            hcloud-volumes   13m
-
-```
-
-4. Create a NodePort service.
-Sample service creation command below:
-
-```bash
-# create the service of type NodePort
-$ oc create svc nodeport  openhands-app-2024  --tcp=3000:3000
-service/openhands-app-2024 created
-
-# review
-
-$ oc get svc
-NAME                 TYPE       CLUSTER-IP      EXTERNAL-IP   PORT(S)          AGE
-openhands-app-2024   NodePort   172.30.225.42   <none>        3000:30495/TCP   4s
-
-$ oc describe svc openhands-app-2024
-Name:                     openhands-app-2024
-Namespace:                openhands
-Labels:                   app=openhands-app-2024
-Annotations:              <none>
-Selector:                 app=openhands-app-2024
-Type:                     NodePort
-IP Family Policy:         SingleStack
-IP Families:              IPv4
-IP:                       172.30.225.42
-IPs:                      172.30.225.42
-Port:                     3000-3000  3000/TCP
-TargetPort:               3000/TCP
-NodePort:                 3000-3000  30495/TCP
-Endpoints:                10.128.2.48:3000
-Session Affinity:         None
-External Traffic Policy:  Cluster
-Events:                   <none>
-```
-
-6. Connect to OpenHands UI, configure the Agent, then test:
-
-![image](https://github.com/user-attachments/assets/12f94804-a0c7-4744-b873-e003c9caf40e)
-
-
-
-## GCP GKE Openhands deployment
-
-**Warning**: this deployment grants the OpenHands application access to the Kubernetes docker socket, which creates security risk. Use at your own discretion.
-1- Create policy for privillege access
-2- Create gke credentials(optional)
-3- Create openhands deployment
-4- Verification and ui access commands
-5- Tshoot pod to verify the internal container
-
-1. create policy for privillege access
-```bash
-apiVersion: rbac.authorization.k8s.io/v1
-kind: ClusterRole
-metadata:
-  name: privileged-role
-rules:
- apiGroups: [""]
-  resources: ["pods"]
-  verbs: ["create", "get", "list", "watch", "delete"]
- apiGroups: ["apps"]
-  resources: ["deployments"]
-  verbs: ["create", "get", "list", "watch", "delete"]
- apiGroups: [""]
-  resources: ["pods/exec"]
-  verbs: ["create"]
- apiGroups: [""]
-  resources: ["pods/log"]
-  verbs: ["get"]
---
-apiVersion: rbac.authorization.k8s.io/v1
-kind: ClusterRoleBinding
-metadata:
-  name: privileged-role-binding
-roleRef:
-  apiGroup: rbac.authorization.k8s.io
-  kind: ClusterRole
-  name: privileged-role
-subjects:
- kind: ServiceAccount
-  name: default  # Change to your service account name
-  namespace: default
-```
-2. create gke credentials(optional)
-```bash
-kubectl create secret generic google-cloud-key \
-  --from-file=key.json=/path/to/your/google-cloud-key.json
-  ```
-3. create openhands deployment
-## as this is tested for the single worker node if you have multiple specify the flag for the single worker
-
-```bash
-kind: Deployment
-metadata:
-  name: openhands-app-2024
-  labels:
-    app: openhands-app-2024
-spec:
-  replicas: 1  # You can increase this number for multiple replicas
-  selector:
-    matchLabels:
-      app: openhands-app-2024
-  template:
-    metadata:
-      labels:
-        app: openhands-app-2024
-    spec:
-      containers:
-      - name: openhands-app-2024
-        image: docker.all-hands.dev/all-hands-ai/openhands:main
-        env:
-        - name: SANDBOX_USER_ID
-          value: "1000"
-        - name: SANDBOX_API_HOSTNAME
-          value: '10.164.0.4'
-        - name: WORKSPACE_MOUNT_PATH
-          value: "/tmp/workspace_base"
-        - name: GOOGLE_APPLICATION_CREDENTIALS
-          value: "/tmp/workspace_base/google-cloud-key.json"
-        volumeMounts:
-        - name: workspace-volume
-          mountPath: /tmp/workspace_base
-        - name: docker-sock
-          mountPath: /var/run/docker.sock
-        - name: google-credentials
-          mountPath: "/tmp/workspace_base/google-cloud-key.json"
-        securityContext:
-          privileged: true  # Add this to allow privileged access
-        ports:
-        - containerPort: 3000
-      - name: openhands-sandbox-2024
-        image: docker.all-hands.dev/all-hands-ai/runtime:main
-    #    securityContext:
-    #      privileged: true  # Add this to allow privileged access
-        ports:
-        - containerPort: 51963
-        command: ["/usr/sbin/sshd", "-D", "-p 51963", "-o", "PermitRootLogin=yes"]
-      volumes:
-      #- name: workspace-volume
-      #  persistentVolumeClaim:
-      #    claimName: workspace-pvc
-      - name: workspace-volume
-        emptyDir: {}
-      - name: docker-sock
-        hostPath:
-          path: /var/run/docker.sock       # Use host's Docker socket
-          type: Socket
-      - name: google-credentials
-        secret:
-          secretName: google-cloud-key
---
-apiVersion: v1
-kind: Service
-metadata:
-  name: openhands-app-2024-svc
-spec:
-  selector:
-    app: openhands-app-2024
-  ports:
-  - name: http
-    protocol: TCP
-    port: 80
-    targetPort: 3000
-  - name: ssh
-    protocol: TCP
-    port: 51963
-    targetPort: 51963
-  type: LoadBalancer
-  ```
-
-5. Tshoot pod to verify the internal container
-### if you want to know more regarding the internal container runtime use below mention pod deployment use kubectl exec -it to enter into container and you can check the contaienr run time using normal docker commands like "docker ps -a"
-
-```bash
-apiVersion: apps/v1
-kind: Deployment
-metadata:
-  name: docker-in-docker
-spec:
-  replicas: 1
-  selector:
-    matchLabels:
-      app: docker-in-docker
-  template:
-    metadata:
-      labels:
-        app: docker-in-docker
-    spec:
-      containers:
-      - name: dind
-        image: docker:20.10-dind
-        securityContext:
-          privileged: true
-        volumeMounts:
-        - name: docker-sock
-          mountPath: /var/run/docker.sock
-      volumes:
-      - name: docker-sock
-        hostPath:
-          path: /var/run/docker.sock
-          type: Socket
-```
--- a/docs/modules/usage/how-to/persist-session-data.md
+++ b/docs/modules/usage/how-to/persist-session-data.md
@@ -0,0 +1,16 @@
+# Persisting Session Data
+
+Using the standard installation, the session data is stored in memory. Currently, if OpenHands' service is restarted,
+previous sessions become invalid (a new secret is generated) and thus not recoverable.
+
+## How to Persist Session Data
+
+### Development Workflow
+In the `config.toml` file, specify the following:
+```
+[core]
+...
+file_store="local"
+file_store_path="/absolute/path/to/openhands/cache/directory"
+jwt_secret="secretpass"
+```
--- a/docs/modules/usage/installation.mdx
+++ b/docs/modules/usage/installation.mdx
@@ -11,16 +11,16 @@
 The easiest way to run OpenHands is in Docker.

 ```bash
-docker pull docker.all-hands.dev/all-hands-ai/runtime:0.14-nikolaik
+docker pull docker.all-hands.dev/all-hands-ai/runtime:0.16-nikolaik

 docker run -it --rm --pull=always \
-    -e SANDBOX_RUNTIME_CONTAINER_IMAGE=docker.all-hands.dev/all-hands-ai/runtime:0.14-nikolaik \
+    -e SANDBOX_RUNTIME_CONTAINER_IMAGE=docker.all-hands.dev/all-hands-ai/runtime:0.16-nikolaik \
    -e LOG_ALL_EVENTS=true \
    -v /var/run/docker.sock:/var/run/docker.sock \
    -p 3000:3000 \
    --add-host host.docker.internal:host-gateway \
    --name openhands-app \
-    docker.all-hands.dev/all-hands-ai/openhands:0.14
+    docker.all-hands.dev/all-hands-ai/openhands:0.16
 ```

 You can also run OpenHands in a scriptable [headless mode](https://docs.all-hands.dev/modules/usage/how-to/headless-mode), as an [interactive CLI](https://docs.all-hands.dev/modules/usage/how-to/cli-mode), or using the [OpenHands GitHub Action](https://docs.all-hands.dev/modules/usage/how-to/github-action).
--- a/docs/modules/usage/micro-agents.md
+++ b/docs/modules/usage/micro-agents.md
@@ -0,0 +1,213 @@
+# Micro-Agents
+
+OpenHands uses specialized micro-agents to handle specific tasks and contexts efficiently. These micro-agents are small, focused components that provide specialized behavior and knowledge for particular scenarios.
+
+## Overview
+
+Micro-agents are defined in markdown files under the `openhands/agenthub/codeact_agent/micro/` directory. Each micro-agent is configured with:
+
+- A unique name
+- The agent type (typically CodeActAgent)
+- Trigger keywords that activate the agent
+- Specific instructions and capabilities
+
+## Available Micro-Agents
+
+### GitHub Agent
+**File**: `github.md`
+**Triggers**: `github`, `git`
+
+The GitHub agent specializes in GitHub API interactions and repository management. It:
+- Has access to a `GITHUB_TOKEN` for API authentication
+- Follows strict guidelines for repository interactions
+- Handles branch management and pull requests
+- Uses the GitHub API instead of web browser interactions
+
+Key features:
+- Branch protection (prevents direct pushes to main/master)
+- Automated PR creation
+- Git configuration management
+- API-first approach for GitHub operations
+
+### NPM Agent
+**File**: `npm.md`
+**Triggers**: `npm`
+
+Specializes in handling npm package management with specific focus on:
+- Non-interactive shell operations
+- Automated confirmation handling using Unix 'yes' command
+- Package installation automation
+
+### Custom Micro-Agents
+
+You can create your own micro-agents by adding new markdown files to the micro-agents directory. Each file should follow this structure:
+
+```markdown
+---
+name: agent_name
+agent: CodeActAgent
+triggers:
+- trigger_word1
+- trigger_word2
+---
+
+Instructions and capabilities for the micro-agent...
+```
+
+## Best Practices
+
+When working with micro-agents:
+
+1. **Use Appropriate Triggers**: Ensure your commands include the relevant trigger words to activate the correct micro-agent
+2. **Follow Agent Guidelines**: Each agent has specific instructions and limitations - respect these for optimal results
+3. **API-First Approach**: When available, use API endpoints rather than web interfaces
+4. **Automation Friendly**: Design commands that work well in non-interactive environments
+
+## Integration
+
+Micro-agents are automatically integrated into OpenHands' workflow. They:
+- Monitor incoming commands for their trigger words
+- Activate when relevant triggers are detected
+- Apply their specialized knowledge and capabilities
+- Follow their specific guidelines and restrictions
+
+## Example Usage
+
+```bash
+# GitHub agent example
+git checkout -b feature-branch
+git commit -m "Add new feature"
+git push origin feature-branch
+
+# NPM agent example
+yes | npm install package-name
+```
+
+For more information about specific agents, refer to their individual documentation files in the micro-agents directory.
+
+## Contributing a Micro-Agent
+
+To contribute a new micro-agent to OpenHands, follow these guidelines:
+
+### 1. Planning Your Micro-Agent
+
+Before creating a micro-agent, consider:
+- What specific problem or use case will it address?
+- What unique capabilities or knowledge should it have?
+- What trigger words make sense for activating it?
+- What constraints or guidelines should it follow?
+
+### 2. File Structure
+
+Create a new markdown file in `openhands/agenthub/codeact_agent/micro/` with a descriptive name (e.g., `docker.md` for a Docker-focused agent).
+
+### 3. Required Components
+
+Your micro-agent file must include:
+
+1. **Front Matter**: YAML metadata at the start of the file:
+```markdown
+---
+name: your_agent_name
+agent: CodeActAgent
+triggers:
+- trigger_word1
+- trigger_word2
+---
+```
+
+2. **Instructions**: Clear, specific guidelines for the agent's behavior:
+```markdown
+You are responsible for [specific task/domain].
+
+Key responsibilities:
+1. [Responsibility 1]
+2. [Responsibility 2]
+
+Guidelines:
+- [Guideline 1]
+- [Guideline 2]
+
+Examples of usage:
+[Example 1]
+[Example 2]
+```
+
+### 4. Best Practices for Micro-Agent Development
+
+1. **Clear Scope**: Keep the agent focused on a specific domain or task
+2. **Explicit Instructions**: Provide clear, unambiguous guidelines
+3. **Useful Examples**: Include practical examples of common use cases
+4. **Safety First**: Include necessary warnings and constraints
+5. **Integration Awareness**: Consider how the agent interacts with other components
+
+### 5. Testing Your Micro-Agent
+
+Before submitting:
+1. Test the agent with various prompts
+2. Verify trigger words activate the agent correctly
+3. Ensure instructions are clear and comprehensive
+4. Check for potential conflicts with existing agents
+
+### 6. Example Implementation
+
+Here's a template for a new micro-agent:
+
+```markdown
+---
+name: docker
+agent: CodeActAgent
+triggers:
+- docker
+- container
+---
+
+You are responsible for Docker container management and Dockerfile creation.
+
+Key responsibilities:
+1. Create and modify Dockerfiles
+2. Manage container lifecycle
+3. Handle Docker Compose configurations
+
+Guidelines:
+- Always use official base images when possible
+- Include necessary security considerations
+- Follow Docker best practices for layer optimization
+
+Examples:
+1. Creating a Dockerfile:
+   ```dockerfile
+   FROM node:18-alpine
+   WORKDIR /app
+   COPY package*.json ./
+   RUN npm install
+   COPY . .
+   CMD ["npm", "start"]
+   ```
+
+2. Docker Compose usage:
+   ```yaml
+   version: '3'
+   services:
+     web:
+       build: .
+       ports:
+         - "3000:3000"
+   ```
+
+Remember to:
+- Validate Dockerfile syntax
+- Check for security vulnerabilities
+- Optimize for build time and image size
+```
+
+### 7. Submission Process
+
+1. Create your micro-agent file in the correct directory
+2. Test thoroughly
+3. Submit a pull request with:
+   - The new micro-agent file
+   - Updated documentation if needed
+   - Description of the agent's purpose and capabilities
+
+Remember that micro-agents are a powerful way to extend OpenHands' capabilities in specific domains. Well-designed agents can significantly improve the system's ability to handle specialized tasks.
--- a/docs/modules/usage/prompting-best-practices.md
+++ b/docs/modules/usage/prompting-best-practices.md
@@ -2,6 +2,11 @@

 When working with OpenHands AI software developer, it's crucial to provide clear and effective prompts. This guide outlines best practices for creating prompts that will yield the most accurate and useful responses.

+## Table of Contents
+
+- [Characteristics of Good Prompts](#characteristics-of-good-prompts)
+- [Customizing Prompts for your Project](#customizing-prompts-for-your-project)
+
 ## Characteristics of Good Prompts

 Good prompts are:
@@ -39,3 +44,63 @@ Good prompts are:
 Remember, the more precise and informative your prompt is, the better the AI can assist you in developing or modifying the OpenHands software.

 See [Getting Started with OpenHands](./getting-started) for more examples of helpful prompts.
+
+## Customizing Prompts for your Project
+
+OpenHands can be customized to work more effectively with specific repositories by providing repository-specific context and guidelines. This section explains how to optimize OpenHands for your project.
+
+### Repository Configuration
+
+You can customize OpenHands' behavior for your repository by creating a `.openhands_instructions` file in your repository's root directory. This file should contain:
+
+1. **Repository Overview**: A brief description of your project's purpose and architecture
+2. **Directory Structure**: Key directories and their purposes
+3. **Development Guidelines**: Project-specific coding standards and practices
+4. **Testing Requirements**: How to run tests and what types of tests are required
+5. **Setup Instructions**: Steps needed to build and run the project
+
+Example `.openhands_instructions` file:
+```
+Repository: MyProject
+Description: A web application for task management
+
+Directory Structure:
+- src/: Main application code
+- tests/: Test files
+- docs/: Documentation
+
+Setup:
+- Run `npm install` to install dependencies
+- Use `npm run dev` for development
+- Run `npm test` for testing
+
+Guidelines:
+- Follow ESLint configuration
+- Write tests for all new features
+- Use TypeScript for new code
+```
+
+### Customizing Prompts
+
+When working with a customized repository:
+
+1. **Reference Project Standards**: Mention specific coding standards or patterns used in your project
+2. **Include Context**: Reference relevant documentation or existing implementations
+3. **Specify Testing Requirements**: Include project-specific testing requirements in your prompts
+
+Example customized prompt:
+```
+Add a new task completion feature to src/components/TaskList.tsx following our existing component patterns.
+Include unit tests in tests/components/ and update the documentation in docs/features/.
+The component should use our shared styling from src/styles/components.
+```
+
+### Best Practices for Repository Customization
+
+1. **Keep Instructions Updated**: Regularly update your `.openhands_instructions` file as your project evolves
+2. **Be Specific**: Include specific paths, patterns, and requirements unique to your project
+3. **Document Dependencies**: List all tools and dependencies required for development
+4. **Include Examples**: Provide examples of good code patterns from your project
+5. **Specify Conventions**: Document naming conventions, file organization, and code style preferences
+
+By customizing OpenHands for your repository, you'll get more accurate and consistent results that align with your project's standards and requirements.
--- a/docs/modules/usage/runtimes.md
+++ b/docs/modules/usage/runtimes.md
@@ -16,7 +16,7 @@ some flags being passed to `docker run` that make this possible:

 ```
 docker run # ...
-    -e SANDBOX_RUNTIME_CONTAINER_IMAGE=docker.all-hands.dev/all-hands-ai/runtime:0.11-nikolaik \
+    -e SANDBOX_RUNTIME_CONTAINER_IMAGE=docker.all-hands.dev/all-hands-ai/runtime:0.16-nikolaik \
    -v /var/run/docker.sock:/var/run/docker.sock \
    # ...
 ```
@@ -28,12 +28,22 @@ You can also [build your own runtime image](how-to/custom-sandbox-guide).
 ### Connecting to Your filesystem
 One useful feature here is the ability to connect to your local filesystem.

-To mount your filesystem into the runtime, add the following options to
-the `docker run` command:
-
+To mount your filesystem into the runtime, first set WORKSPACE_BASE:
 ```bash
 export WORKSPACE_BASE=/path/to/your/code

+# Linux and Mac Example
+# export WORKSPACE_BASE=$HOME/OpenHands
+# Will set $WORKSPACE_BASE to /home/<username>/OpenHands
+#
+# WSL on Windows Example
+# export WORKSPACE_BASE=/mnt/c/dev/OpenHands
+# Will set $WORKSPACE_BASE to C:\dev\OpenHands
+```
+
+then add the following options to the `docker run` command:
+
+```bash
 docker run # ...
    -e SANDBOX_USER_ID=$(id -u) \
    -e WORKSPACE_MOUNT_PATH=$WORKSPACE_BASE \
--- a/docs/modules/usage/troubleshooting/troubleshooting.md
+++ b/docs/modules/usage/troubleshooting/troubleshooting.md
@@ -1,180 +1,44 @@
 # 🚧 Troubleshooting

-There are some error messages that frequently get reported by users.
-We'll try to make the install process easier, but for now you can look for your error message below and see if there are any workarounds.
-If you find more information or a workaround for one of these issues, please open a *PR* to add details to this file.
-
 :::tip
-OpenHands only supports Windows via [WSL](https://learn.microsoft.com/en-us/windows/wsl/install).
-Please be sure to run all commands inside your WSL terminal.
-Check out [Notes for WSL on Windows Users](troubleshooting/windows) for some troubleshooting guides.
+OpenHands only supports Windows via WSL. Please be sure to run all commands inside your WSL terminal.
 :::

-## Common Issues
+### Launch docker client failed

-* [Unable to connect to Docker](#unable-to-connect-to-docker)
-* [404 Resource not found](#404-resource-not-found)
-* [`make build` getting stuck on package installations](#make-build-getting-stuck-on-package-installations)
-* [Sessions are not restored](#sessions-are-not-restored)
-* [Connection to host.docker.internal timed out](#connection-to-host-docker-internal-timed-out)
+**Description**

-### Unable to connect to Docker
-
-[GitHub Issue](https://github.com/All-Hands-AI/OpenHands/issues/1226)
-
-**Symptoms**
-
-```bash
-Error creating controller. Please check Docker is running and visit `https://docs.all-hands.dev/modules/usage/troubleshooting` for more debugging information.
+When running OpenHands, the following error is seen:
+```
+Launch docker client failed. Please make sure you have installed docker and started docker desktop/daemon.
 ```

-```bash
-docker.errors.DockerException: Error while fetching server API version: ('Connection aborted.', FileNotFoundError(2, 'No such file or directory'))
-```
-
-**Details**
-
-OpenHands uses a Docker container to do its work safely, without potentially breaking your machine.
-
-**Workarounds**
-
-* Run `docker ps` to ensure that docker is running
-* Make sure you don't need `sudo` to run docker [see here](https://www.baeldung.com/linux/docker-run-without-sudo)
-* If you are on a Mac, check the [permissions requirements](https://docs.docker.com/desktop/mac/permission-requirements/) and in particular consider enabling the `Allow the default Docker socket to be used` under `Settings > Advanced` in Docker Desktop.
-* In addition, upgrade your Docker to the latest version under `Check for Updates`
+**Resolution**

+Try these in order:
+* Confirm `docker` is running on your system. You should be able to run `docker ps` in the terminal successfully.
+* If using Docker Desktop, ensure `Settings > Advanced > Allow the default Docker socket to be used` is enabled.
+* Depending on your configuration you may need `Settings > Resources > Network > Enable host networking` enabled in Docker Desktop.
+* Reinstall Docker Desktop.
 ---
-### `404 Resource not found`

-**Symptoms**
+# Development Workflow Specific
+### Error building runtime docker image

-```python
-Traceback (most recent call last):
-  File "/app/.venv/lib/python3.12/site-packages/litellm/llms/openai.py", line 414, in completion
-    raise e
-  File "/app/.venv/lib/python3.12/site-packages/litellm/llms/openai.py", line 373, in completion
-    response = openai_client.chat.completions.create(**data, timeout=timeout)  # type: ignore
-               ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
-  File "/app/.venv/lib/python3.12/site-packages/openai/_utils/_utils.py", line 277, in wrapper
-    return func(*args, **kwargs)
-           ^^^^^^^^^^^^^^^^^^^^^
-  File "/app/.venv/lib/python3.12/site-packages/openai/resources/chat/completions.py", line 579, in create
-    return self._post(
-           ^^^^^^^^^^^
-  File "/app/.venv/lib/python3.12/site-packages/openai/_base_client.py", line 1232, in post
-    return cast(ResponseT, self.request(cast_to, opts, stream=stream, stream_cls=stream_cls))
-                           ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
-  File "/app/.venv/lib/python3.12/site-packages/openai/_base_client.py", line 921, in request
-    return self._request(
-           ^^^^^^^^^^^^^^
-  File "/app/.venv/lib/python3.12/site-packages/openai/_base_client.py", line 1012, in _request
-    raise self._make_status_error_from_response(err.response) from None
-openai.NotFoundError: Error code: 404 - {'error': {'code': '404', 'message': 'Resource not found'}}
+**Description**
+
+Attempts to start a new session fail, and errors with terms like the following appear in the logs:
+```
+debian-security bookworm-security
+InRelease At least one invalid signature was encountered.
 ```

-**Details**
+This seems to happen when the hash of an existing external library changes and your local docker instance has
+cached a previous version. To work around this, please try the following:

-This happens when LiteLLM (our library for connecting to different LLM providers) can't find
-the API endpoint you're trying to connect to. Most often this happens for Azure or ollama users.
-
-**Workarounds**
-
-* Check that you've set `LLM_BASE_URL` properly
-* Check that the model is set properly, based on the [LiteLLM docs](https://docs.litellm.ai/docs/providers)
-  * If you're running inside the UI, be sure to set the `model` in the settings modal
-  * If you're running headless (via main.py) be sure to set `LLM_MODEL` in your env/config
-* Make sure you've followed any special instructions for your LLM provider
-  * [Azure](/modules/usage/llms/azure-llms)
-  * [Google](/modules/usage/llms/google-llms)
-* Make sure your API key is correct
-* See if you can connect to the LLM using `curl`
-* Try [connecting via LiteLLM directly](https://github.com/BerriAI/litellm) to test your setup
-
---
-### `make build` getting stuck on package installations
-
-**Symptoms**
-
-Package installation stuck on `Pending...` without any error message:
-
-```bash
-Package operations: 286 installs, 0 updates, 0 removals
-
-  - Installing certifi (2024.2.2): Pending...
-  - Installing h11 (0.14.0): Pending...
-  - Installing idna (3.7): Pending...
-  - Installing sniffio (1.3.1): Pending...
-  - Installing typing-extensions (4.11.0): Pending...
-```
-
-**Details**
-
-In rare cases, `make build` can seemingly get stuck on package installations
-without any error message.
-
-**Workarounds**
-
-The package installer Poetry may miss a configuration setting for where credentials are to be looked up (keyring).
-
-First check with `env` if a value for `PYTHON_KEYRING_BACKEND` exists.
-If not, run the below command to set it to a known value and retry the build:
-
-```bash
-export PYTHON_KEYRING_BACKEND=keyring.backends.null.Keyring
-```
-
---
-### Sessions are not restored
-
-**Symptoms**
-
-OpenHands usually asks whether to resume or start a new session when opening the UI.
-But clicking "Resume" still starts a fresh new chat.
-
-**Details**
-
-With a standard installation as of today session data is stored in memory.
-Currently, if OpenHands's service is restarted, previous sessions become
-invalid (a new secret is generated) and thus not recoverable.
-
-**Workarounds**
-
-* Change configuration to make sessions persistent by editing the `config.toml`
-file (in OpenHands's root folder) by specifying a `file_store` and an
-absolute `file_store_path`:
-
-```toml
-file_store="local"
-file_store_path="/absolute/path/to/openhands/cache/directory"
-```
-
-* Add a fixed jwt secret in your .bashrc, like below, so that previous session id's
-should stay accepted.
-
-```bash
-EXPORT JWT_SECRET=A_CONST_VALUE
-```
-
---
-### Connection to host docker internal timed out
-
-**Symptoms**
-
-When you start the server using the docker command from the main [README](https://github.com/All-Hands-AI/OpenHands/README.md), you get a long timeout
-followed by the a stack trace containing messages like:
-
-* `Connection to host.docker.internal timed out. (connect timeout=310)`
-* `Max retries exceeded with url: /alive`
-
-**Details**
-
-If Docker Engine is installed rather than Docker Desktop, the main command will not work as expected.
-Docker Desktop includes easy DNS configuration for connecting processes running in different containers
-which OpenHands makes use of when the main server is running inside a docker container.
-(Further details: https://forums.docker.com/t/difference-between-docker-desktop-and-docker-engine/124612)
-
-**Workarounds**
-
-* [Install Docker Desktop](https://www.docker.com/products/docker-desktop/)
-* Run OpenHands in [Development Mode](https://github.com/All-Hands-AI/OpenHands/blob/main/Development.md),
-  So that the main server is not run inside a container, but still creates dockerized runtime sandboxes.
+* Stop any containers where the name has the prefix `openhands-runtime-` :
+  `docker ps --filter name=openhands-runtime- --filter status=running -aq | xargs docker stop`
+* Remove any containers where the name has the prefix `openhands-runtime-` :
+  `docker rmi $(docker images --filter name=openhands-runtime- -q --no-trunc)`
+* Stop and Remove any containers / images where the name has the prefix `openhands-runtime-`
+* Prune containers / images : `docker container prune -f && docker image prune -f`
--- a/docs/modules/usage/troubleshooting/windows.md
+++ b/docs/modules/usage/troubleshooting/windows.md
@@ -1,64 +0,0 @@
-# Notes for WSL on Windows Users
-
-OpenHands only supports Windows via [WSL](https://learn.microsoft.com/en-us/windows/wsl/install).
-Please be sure to run all commands inside your WSL terminal.
-
-## Troubleshooting
-
-### Recommendation: Do not run as root user
-
-For security reasons, it is highly recommended to not run OpenHands as the root user, but a user with a non-zero UID.
-
-References:
-
-* [Why it is bad to login as root](https://askubuntu.com/questions/16178/why-is-it-bad-to-log-in-as-root)
-* [Set default user in WSL](https://www.tenforums.com/tutorials/128152-set-default-user-windows-subsystem-linux-distro-windows-10-a.html#option2)
-Hint about the 2nd reference: for Ubuntu users, the command could actually be "ubuntupreview" instead of "ubuntu".
-
---
-### Error: 'docker' could not be found in this WSL 2 distro.
-
-If you are using Docker Desktop, make sure to start it before calling any docker command from inside WSL.
-Docker also needs to have the WSL integration option activated.
-
---
-### Poetry Installation
-
-* If you face issues running Poetry even after installing it during the build process, you may need to add its binary path to your environment:
-
-```sh
-export PATH="$HOME/.local/bin:$PATH"
-```
-
-* If make build stops on an error like this:
-
-```sh
-ModuleNotFoundError: no module named <module-name>
-```
-
-This could be an issue with Poetry's cache.
-Try to run these 2 commands after another:
-
-```sh
-rm -r ~/.cache/pypoetry
-make build
-```
-
---
-### NoneType object has no attribute 'request'
-
-If you are experiencing issues related to networking, such as `NoneType object has no attribute 'request'` when executing `make run`, you may need to configure your WSL2 networking settings. Follow these steps:
-
-* Open or create the `.wslconfig` file located at `C:\Users\%username%\.wslconfig` on your Windows host machine.
-* Add the following configuration to the `.wslconfig` file:
-
-```sh
-[wsl2]
-networkingMode=mirrored
-localhostForwarding=true
-```
-
-* Save the `.wslconfig` file.
-* Restart WSL2 completely by exiting any running WSL2 instances and executing the command `wsl --shutdown` in your command prompt or terminal.
-* After restarting WSL, attempt to execute `make run` again.
-The networking issue should be resolved.
--- a/docs/modules/usage/upgrade-guide.md
+++ b/docs/modules/usage/upgrade-guide.md
@@ -1,71 +0,0 @@
-# ⬆️ Upgrade Guide
-
-## 0.8.0 (2024-07-13)
-
-### Config breaking changes
-
-In this release we introduced a few breaking changes to backend configurations.
-If you have only been using OpenHands via frontend (web GUI), nothing needs
-to be taken care of.
-
-Here's a list of breaking changes in configs. They only apply to users who
-use OpenHands CLI via `main.py`. For more detail, see [#2756](https://github.com/All-Hands-AI/OpenHands/pull/2756).
-
-#### Removal of --model-name option from main.py
-
-Please note that `--model-name`, or `-m` option, no longer exists. You should set up the LLM
-configs in `config.toml` or via environmental variables.
-
-#### LLM config groups must be subgroups of 'llm'
-
-Prior to release 0.8, you can use arbitrary name for llm config in `config.toml`, e.g.
-
-```toml
-[gpt-4o]
-model="gpt-4o"
-api_key="<your_api_key>"
-```
-
-and then use `--llm-config` CLI argument to specify the desired LLM config group
-by name. This no longer works. Instead, the config group must be under `llm` group,
-e.g.:
-
-```toml
-[llm.gpt-4o]
-model="gpt-4o"
-api_key="<your_api_key>"
-```
-
-If you have a config group named `llm`, no need to change it, it will be used
-as the default LLM config group.
-
-#### 'agent' group no longer contains 'name' field
-
-Prior to release 0.8, you may or may not have a config group named `agent` that
-looks like this:
-
-```toml
-[agent]
-name="CodeActAgent"
-memory_max_threads=2
-```
-
-Note the `name` field is now removed. Instead, you should put `default_agent` field
-under `core` group, e.g.
-
-```toml
-[core]
-# other configs
-default_agent='CodeActAgent'
-
-[agent]
-llm_config='llm'
-memory_max_threads=2
-
-[agent.CodeActAgent]
-llm_config='gpt-4o'
-```
-
-Note that similar to `llm` subgroups, you can also define `agent` subgroups.
-Moreover, an agent can be associated with a specific LLM config group. For more
-detail, see the examples in `config.template.toml`.
--- a/docs/package-lock.json
+++ b/docs/package-lock.json
@@ -14,17 +14,17 @@
        "@docusaurus/theme-mermaid": "^3.6.3",
        "@mdx-js/react": "^3.1.0",
        "clsx": "^2.0.0",
-        "prism-react-renderer": "^2.4.0",
+        "prism-react-renderer": "^2.4.1",
        "react": "^18.3.1",
        "react-dom": "^18.3.1",
-        "react-icons": "^5.3.0",
-        "react-use": "^17.5.1"
+        "react-icons": "^5.4.0",
+        "react-use": "^17.6.0"
      },
      "devDependencies": {
        "@docusaurus/module-type-aliases": "^3.5.1",
        "@docusaurus/tsconfig": "^3.6.3",
        "@docusaurus/types": "^3.5.1",
-        "typescript": "~5.6.3"
+        "typescript": "~5.7.2"
      },
      "engines": {
        "node": ">=18.0"
@@ -14781,9 +14781,9 @@
      }
    },
    "node_modules/prism-react-renderer": {
-      "version": "2.4.0",
-      "resolved": "https://registry.npmjs.org/prism-react-renderer/-/prism-react-renderer-2.4.0.tgz",
-      "integrity": "sha512-327BsVCD/unU4CNLZTWVHyUHKnsqcvj2qbPlQ8MiBE2eq2rgctjigPA1Gp9HLF83kZ20zNN6jgizHJeEsyFYOw==",
+      "version": "2.4.1",
+      "resolved": "https://registry.npmjs.org/prism-react-renderer/-/prism-react-renderer-2.4.1.tgz",
+      "integrity": "sha512-ey8Ls/+Di31eqzUxC46h8MksNuGx/n0AAC8uKpwFau4RPDYLuE3EXTp8N8G2vX2N7UC/+IXeNUnlWBGGcAG+Ig==",
      "dependencies": {
        "@types/prismjs": "^1.26.0",
        "clsx": "^2.0.0"
@@ -15155,9 +15155,10 @@
      }
    },
    "node_modules/react-icons": {
-      "version": "5.3.0",
-      "resolved": "https://registry.npmjs.org/react-icons/-/react-icons-5.3.0.tgz",
-      "integrity": "sha512-DnUk8aFbTyQPSkCfF8dbX6kQjXA9DktMeJqfjrg6cK9vwQVMxmcA3BfP4QoiztVmEHtwlTgLFsPuH2NskKT6eg==",
+      "version": "5.4.0",
+      "resolved": "https://registry.npmjs.org/react-icons/-/react-icons-5.4.0.tgz",
+      "integrity": "sha512-7eltJxgVt7X64oHh6wSWNwwbKTCtMfK35hcjvJS0yxEAhPM8oUKdS3+kqaW1vicIltw+kR2unHaa12S9pPALoQ==",
+      "license": "MIT",
      "peerDependencies": {
        "react": "*"
      }
@@ -15263,9 +15264,9 @@
      }
    },
    "node_modules/react-use": {
-      "version": "17.5.1",
-      "resolved": "https://registry.npmjs.org/react-use/-/react-use-17.5.1.tgz",
-      "integrity": "sha512-LG/uPEVRflLWMwi3j/sZqR00nF6JGqTTDblkXK2nzXsIvij06hXl1V/MZIlwj1OKIQUtlh1l9jK8gLsRyCQxMg==",
+      "version": "17.6.0",
+      "resolved": "https://registry.npmjs.org/react-use/-/react-use-17.6.0.tgz",
+      "integrity": "sha512-OmedEScUMKFfzn1Ir8dBxiLLSOzhKe/dPZwVxcujweSj45aNM7BEGPb9BEVIgVEqEXx6f3/TsXzwIktNgUR02g==",
      "dependencies": {
        "@types/js-cookie": "^2.2.6",
        "@xobotyi/scrollbar-width": "^1.9.5",
@@ -16985,9 +16986,10 @@
      }
    },
    "node_modules/typescript": {
-      "version": "5.6.3",
-      "resolved": "https://registry.npmjs.org/typescript/-/typescript-5.6.3.tgz",
-      "integrity": "sha512-hjcS1mhfuyi4WW8IWtjP7brDrG2cuDZukyrYrSauoXGNgx0S7zceP07adYkJycEr56BOUTNPzbInooiN3fn1qw==",
+      "version": "5.7.2",
+      "resolved": "https://registry.npmjs.org/typescript/-/typescript-5.7.2.tgz",
+      "integrity": "sha512-i5t66RHxDvVN40HfDd1PsEThGNnlMCMT3jMUuoh9/0TaqWevNontacunWyN02LA9/fIbEWlcHZcgTKb9QoaLfg==",
+      "license": "Apache-2.0",
      "bin": {
        "tsc": "bin/tsc",
        "tsserver": "bin/tsserver"
--- a/docs/package.json
+++ b/docs/package.json
@@ -21,17 +21,17 @@
    "@docusaurus/theme-mermaid": "^3.6.3",
    "@mdx-js/react": "^3.1.0",
    "clsx": "^2.0.0",
-    "prism-react-renderer": "^2.4.0",
+    "prism-react-renderer": "^2.4.1",
    "react": "^18.3.1",
    "react-dom": "^18.3.1",
-    "react-icons": "^5.3.0",
-    "react-use": "^17.5.1"
+    "react-icons": "^5.4.0",
+    "react-use": "^17.6.0"
  },
  "devDependencies": {
    "@docusaurus/module-type-aliases": "^3.5.1",
    "@docusaurus/tsconfig": "^3.6.3",
    "@docusaurus/types": "^3.5.1",
-    "typescript": "~5.6.3"
+    "typescript": "~5.7.2"
  },
  "browserslist": {
    "production": [
--- a/docs/sidebars.ts
+++ b/docs/sidebars.ts
@@ -14,9 +14,20 @@ const sidebars: SidebarsConfig = {
      id: 'usage/getting-started',
    },
    {
-      type: 'doc',
-      label: 'Prompting Best Practices',
-      id: 'usage/prompting-best-practices',
+      type: 'category',
+      label: 'Prompting',
+      items: [
+        {
+          type: 'doc',
+          label: 'Best Practices',
+          id: 'usage/prompting-best-practices',
+        },
+        {
+          type: 'doc',
+          label: 'Micro-Agents',
+          id: 'usage/micro-agents',
+        },
+      ],
    },
    {
      type: 'category',
@@ -110,6 +121,11 @@ const sidebars: SidebarsConfig = {
          label: 'Custom Sandbox',
          id: 'usage/how-to/custom-sandbox-guide',
        },
+        {
+          type: 'doc',
+          label: 'Persist Session Data',
+          id: 'usage/how-to/persist-session-data',
+        },
      ],
    },
    {
@@ -152,11 +168,6 @@ const sidebars: SidebarsConfig = {
          label: 'Evaluation',
          id: 'usage/how-to/evaluation-harness',
        },
-        {
-          type: 'doc',
-          label: 'Kubernetes Deployment',
-          id: 'usage/how-to/openshift-example',
-        },
      ],
    },
    {
--- a/docs/src/components/CustomFooter.tsx
+++ b/docs/src/components/CustomFooter.tsx
@@ -8,7 +8,7 @@ function CustomFooter() {
    <footer className="custom-footer">
      <div className="footer-content">
        <div className="footer-icons">
-          <a href="https://join.slack.com/t/opendevin/shared_invite/zt-2oikve2hu-UDxHeo8nsE69y6T7yFX_BA" target="_blank" rel="noopener noreferrer">
+          <a href="https://join.slack.com/t/openhands-ai/shared_invite/zt-2vbfigwev-G03twSpXaErwzYVD4CFiBg" target="_blank" rel="noopener noreferrer">
            <FaSlack />
          </a>
          <a href="https://discord.gg/ESHStjSjD4" target="_blank" rel="noopener noreferrer">
--- a/docs/src/components/HomepageHeader/HomepageHeader.tsx
+++ b/docs/src/components/HomepageHeader/HomepageHeader.tsx
@@ -23,7 +23,7 @@ export function HomepageHeader() {
          <a href="https://codecov.io/github/All-Hands-AI/OpenHands?branch=main"><img alt="CodeCov" src="https://img.shields.io/codecov/c/github/All-Hands-AI/OpenHands?style=for-the-badge&color=blue" /></a>
          <a href="https://github.com/All-Hands-AI/OpenHands/blob/main/LICENSE"><img src="https://img.shields.io/github/license/All-Hands-AI/OpenHands?style=for-the-badge&color=blue" alt="MIT License" /></a>
          <br/>
-          <a href="https://join.slack.com/t/openhands-ai/shared_invite/zt-2tom0er4l-JeNUGHt_AxpEfIBstbLPiw"><img src="https://img.shields.io/badge/Slack-Join%20Us-red?logo=slack&logoColor=white&style=for-the-badge" alt="Join our Slack community" /></a>
+          <a href="https://join.slack.com/t/openhands-ai/shared_invite/zt-2vbfigwev-G03twSpXaErwzYVD4CFiBg"><img src="https://img.shields.io/badge/Slack-Join%20Us-red?logo=slack&logoColor=white&style=for-the-badge" alt="Join our Slack community" /></a>
          <a href="https://discord.gg/ESHStjSjD4"><img src="https://img.shields.io/badge/Discord-Join%20Us-purple?logo=discord&logoColor=white&style=for-the-badge" alt="Join our Discord community" /></a>
          <a href="https://github.com/All-Hands-AI/OpenHands/blob/main/CREDITS.md"><img src="https://img.shields.io/badge/Project-Credits-blue?style=for-the-badge&color=FFE165&logo=github&logoColor=white" alt="Credits" /></a>
          <br/>
--- a/docs/yarn.lock
+++ b/docs/yarn.lock
--- a/evaluation/benchmarks/EDA/README.md
+++ b/evaluation/benchmarks/EDA/README.md
@@ -4,12 +4,10 @@ This folder contains evaluation harness for evaluating agents on the Entity-dedu

 ## Setup Environment and LLM Configuration

-Please follow instruction [here](../README.md#setup) to setup your local development environment and LLM.
-
+Please follow instruction [here](../../README.md#setup) to setup your local development environment and LLM.

 ## Start the evaluation

-
 ```bash
 export OPENAI_API_KEY="sk-XXX"; # This is required for evaluation (to simulate another party of conversation)
 ./evaluation/benchmarks/EDA/scripts/run_infer.sh [model_config] [git-version] [agent] [dataset] [eval_limit]
@@ -37,7 +35,8 @@ For example,
 ```

 ## Reference
-```
+
+```bibtex
@inproceedings{zhang2023entity,
  title={Probing the Multi-turn Planning Capabilities of LLMs via 20 Question Games},
  author={Zhang, Yizhe and Lu, Jiarui and Jaitly, Navdeep},
--- a/evaluation/benchmarks/EDA/run_infer.py
+++ b/evaluation/benchmarks/EDA/run_infer.py
@@ -202,6 +202,9 @@ if __name__ == '__main__':
    llm_config = None
    if args.llm_config:
        llm_config = get_llm_config_arg(args.llm_config)
+        # modify_params must be False for evaluation purpose, for reproducibility and accurancy of results
+        llm_config.modify_params = False
+
    if llm_config is None:
        raise ValueError(f'Could not find LLM config: --llm_config {args.llm_config}')

--- a/evaluation/benchmarks/EDA/scripts/run_infer.sh
+++ b/evaluation/benchmarks/EDA/scripts/run_infer.sh
@@ -21,7 +21,7 @@ if [ -z "$AGENT" ]; then
  AGENT="CodeActAgent"
 fi

-get_agent_version
+get_openhands_version

 if [ -z "$DATASET" ]; then
  echo "Dataset not specified, use default 'things'"
@@ -34,12 +34,9 @@ if [ -z "$OPENAI_API_KEY" ]; then
  exit 1
 fi

-# IMPORTANT: Because Agent's prompt changes fairly often in the rapidly evolving codebase of OpenHands
-# We need to track the version of Agent in the evaluation to make sure results are comparable
-AGENT_VERSION=v$(poetry run python -c "import openhands.agenthub; from openhands.controller.agent import Agent; print(Agent.get_cls('$AGENT').VERSION)")

 echo "AGENT: $AGENT"
-echo "AGENT_VERSION: $AGENT_VERSION"
+echo "OPENHANDS_VERSION: $OPENHANDS_VERSION"
 echo "MODEL_CONFIG: $MODEL_CONFIG"
 echo "DATASET: $DATASET"

@@ -51,7 +48,7 @@ COMMAND="poetry run python evaluation/benchmarks/EDA/run_infer.py \
  --max-iterations 20 \
  --OPENAI_API_KEY $OPENAI_API_KEY \
  --eval-num-workers $NUM_WORKERS \
-  --eval-note ${AGENT_VERSION}_${DATASET}"
+  --eval-note ${OPENHANDS_VERSION}_${DATASET}"

 if [ -n "$EVAL_LIMIT" ]; then
  echo "EVAL_LIMIT: $EVAL_LIMIT"
--- a/evaluation/benchmarks/agent_bench/README.md
+++ b/evaluation/benchmarks/agent_bench/README.md
@@ -4,7 +4,7 @@ This folder contains evaluation harness for evaluating agents on the [AgentBench

 ## Setup Environment and LLM Configuration

-Please follow instruction [here](../README.md#setup) to setup your local development environment and LLM.
+Please follow instruction [here](../../README.md#setup) to setup your local development environment and LLM.

 ## Start the evaluation

--- a/evaluation/benchmarks/agent_bench/run_infer.py
+++ b/evaluation/benchmarks/agent_bench/run_infer.py
@@ -307,6 +307,8 @@ if __name__ == '__main__':
    llm_config = None
    if args.llm_config:
        llm_config = get_llm_config_arg(args.llm_config)
+        # modify_params must be False for evaluation purpose, for reproducibility and accurancy of results
+        llm_config.modify_params = False

    if llm_config is None:
        raise ValueError(f'Could not find LLM config: --llm_config {args.llm_config}')
--- a/evaluation/benchmarks/agent_bench/scripts/run_infer.sh
+++ b/evaluation/benchmarks/agent_bench/scripts/run_infer.sh
@@ -20,10 +20,10 @@ if [ -z "$AGENT" ]; then
  AGENT="CodeActAgent"
 fi

-get_agent_version
+get_openhands_version

 echo "AGENT: $AGENT"
-echo "AGENT_VERSION: $AGENT_VERSION"
+echo "OPENHANDS_VERSION: $OPENHANDS_VERSION"
 echo "MODEL_CONFIG: $MODEL_CONFIG"

 COMMAND="export PYTHONPATH=evaluation/benchmarks/agent_bench:\$PYTHONPATH && poetry run python evaluation/benchmarks/agent_bench/run_infer.py \
@@ -31,7 +31,7 @@ COMMAND="export PYTHONPATH=evaluation/benchmarks/agent_bench:\$PYTHONPATH && poe
  --llm-config $MODEL_CONFIG \
  --max-iterations 30 \
  --eval-num-workers $NUM_WORKERS \
-  --eval-note $AGENT_VERSION"
+  --eval-note $OPENHANDS_VERSION"

 if [ -n "$EVAL_LIMIT" ]; then
  echo "EVAL_LIMIT: $EVAL_LIMIT"
--- a/evaluation/benchmarks/aider_bench/README.md
+++ b/evaluation/benchmarks/aider_bench/README.md
@@ -10,7 +10,7 @@ Hugging Face dataset based on the

 ## Setup Environment and LLM Configuration

-Please follow instruction [here](../README.md#setup) to setup your local
+Please follow instruction [here](../../README.md#setup) to setup your local
 development environment and LLM.

 ## Start the evaluation
--- a/evaluation/benchmarks/aider_bench/run_infer.py
+++ b/evaluation/benchmarks/aider_bench/run_infer.py
@@ -279,6 +279,8 @@ if __name__ == '__main__':
    llm_config = None
    if args.llm_config:
        llm_config = get_llm_config_arg(args.llm_config)
+        # modify_params must be False for evaluation purpose, for reproducibility and accurancy of results
+        llm_config.modify_params = False

    if llm_config is None:
        raise ValueError(f'Could not find LLM config: --llm_config {args.llm_config}')
--- a/evaluation/benchmarks/aider_bench/scripts/run_infer.sh
+++ b/evaluation/benchmarks/aider_bench/scripts/run_infer.sh
@@ -21,13 +21,13 @@ if [ -z "$AGENT" ]; then
  AGENT="CodeActAgent"
 fi

-get_agent_version
+get_openhands_version

 echo "AGENT: $AGENT"
-echo "AGENT_VERSION: $AGENT_VERSION"
+echo "OPENHANDS_VERSION: $OPENHANDS_VERSION"
 echo "MODEL_CONFIG: $MODEL_CONFIG"

-EVAL_NOTE=$AGENT_VERSION
+EVAL_NOTE=$OPENHANDS_VERSION

 # Default to NOT use unit tests.
 if [ -z "$USE_UNIT_TESTS" ]; then
--- a/evaluation/benchmarks/biocoder/README.md
+++ b/evaluation/benchmarks/biocoder/README.md
@@ -4,13 +4,14 @@ Implements evaluation of agents on BioCoder from the BioCoder benchmark introduc

 ## Setup Environment and LLM Configuration

-Please follow instruction [here](../README.md#setup) to setup your local development environment and LLM.
+Please follow instruction [here](../../README.md#setup) to setup your local development environment and LLM.

 ## BioCoder Docker Image

 In the openhands branch of the Biocoder repository, we have slightly modified our original Docker image to work with the OpenHands environment. In the Docker image are testing scripts (`/testing/start_test_openhands.py` and aux files in `/testing_files/`) to assist with evaluation. Additionally, we have installed all dependencies, including OpenJDK, mamba (with Python 3.6), and many system libraries. Notably, we have **not** packaged all repositories into the image, so they are downloaded at runtime.

 **Before first execution, pull our Docker image with the following command**
+
 ```bash
 docker pull public.ecr.aws/i5g0m1f6/eval_biocoder:v1.0
 ```
@@ -19,7 +20,6 @@ To reproduce this image, please see the Dockerfile_Openopenhands in the `biocode

 ## Start the evaluation

-
 ```bash
 ./evaluation/benchmarks/biocoder/scripts/run_infer.sh [model_config] [git-version] [agent] [eval_limit]
 ```
@@ -47,7 +47,8 @@ with current OpenHands version, then your command would be:
 ```

 ## Reference
-```
+
+```bibtex
@misc{tang2024biocoder,
      title={BioCoder: A Benchmark for Bioinformatics Code Generation with Large Language Models},
      author={Xiangru Tang and Bill Qian and Rick Gao and Jiakang Chen and Xinyun Chen and Mark Gerstein},
--- a/evaluation/benchmarks/biocoder/run_infer.py
+++ b/evaluation/benchmarks/biocoder/run_infer.py
@@ -328,6 +328,8 @@ if __name__ == '__main__':
    llm_config = None
    if args.llm_config:
        llm_config = get_llm_config_arg(args.llm_config)
+        # modify_params must be False for evaluation purpose, for reproducibility and accurancy of results
+        llm_config.modify_params = False

    if llm_config is None:
        raise ValueError(f'Could not find LLM config: --llm_config {args.llm_config}')
--- a/evaluation/benchmarks/biocoder/scripts/run_infer.sh
+++ b/evaluation/benchmarks/biocoder/scripts/run_infer.sh
@@ -21,10 +21,10 @@ if [ -z "$AGENT" ]; then
  AGENT="CodeActAgent"
 fi

-get_agent_version
+get_openhands_version

 echo "AGENT: $AGENT"
-echo "AGENT_VERSION: $AGENT_VERSION"
+echo "OPENHANDS_VERSION: $OPENHANDS_VERSION"
 echo "MODEL_CONFIG: $MODEL_CONFIG"
 echo "DATASET: $DATASET"

@@ -33,7 +33,7 @@ COMMAND="poetry run python evaluation/benchmarks/biocoder/run_infer.py \
  --llm-config $MODEL_CONFIG \
  --max-iterations 10 \
  --eval-num-workers $NUM_WORKERS \
-  --eval-note ${AGENT_VERSION}_${DATASET}"
+  --eval-note ${OPENHANDS_VERSION}_${DATASET}"

 if [ -n "$EVAL_LIMIT" ]; then
  echo "EVAL_LIMIT: $EVAL_LIMIT"
--- a/evaluation/benchmarks/bird/README.md
+++ b/evaluation/benchmarks/bird/README.md
--- a/evaluation/benchmarks/bird/run_infer.py
+++ b/evaluation/benchmarks/bird/run_infer.py
@@ -456,6 +456,8 @@ if __name__ == '__main__':
    llm_config = None
    if args.llm_config:
        llm_config = get_llm_config_arg(args.llm_config)
+        # modify_params must be False for evaluation purpose, for reproducibility and accurancy of results
+        llm_config.modify_params = False
    if llm_config is None:
        raise ValueError(f'Could not find LLM config: --llm_config {args.llm_config}')

--- a/evaluation/benchmarks/bird/scripts/run_infer.sh
+++ b/evaluation/benchmarks/bird/scripts/run_infer.sh
@@ -20,10 +20,10 @@ if [ -z "$AGENT" ]; then
  AGENT="CodeActAgent"
 fi

-get_agent_version
+get_openhands_version

 echo "AGENT: $AGENT"
-echo "AGENT_VERSION: $AGENT_VERSION"
+echo "OPENHANDS_VERSION: $OPENHANDS_VERSION"
 echo "MODEL_CONFIG: $MODEL_CONFIG"

 COMMAND="poetry run python evaluation/benchmarks/bird/run_infer.py \
@@ -31,7 +31,7 @@ COMMAND="poetry run python evaluation/benchmarks/bird/run_infer.py \
  --llm-config $MODEL_CONFIG \
  --max-iterations 5 \
  --eval-num-workers $NUM_WORKERS \
-  --eval-note $AGENT_VERSION" \
+  --eval-note $OPENHANDS_VERSION" \

 if [ -n "$EVAL_LIMIT" ]; then
  echo "EVAL_LIMIT: $EVAL_LIMIT"
--- a/evaluation/benchmarks/browsing_delegation/README.md
+++ b/evaluation/benchmarks/browsing_delegation/README.md
@@ -7,7 +7,7 @@ If so, the browsing performance upper-bound of CodeActAgent will be the performa

 ## Setup Environment and LLM Configuration

-Please follow instruction [here](../README.md#setup) to setup your local development environment and LLM.
+Please follow instruction [here](../../README.md#setup) to setup your local development environment and LLM.

 ## Run Inference

--- a/evaluation/benchmarks/browsing_delegation/run_infer.py
+++ b/evaluation/benchmarks/browsing_delegation/run_infer.py
@@ -142,6 +142,8 @@ if __name__ == '__main__':
    llm_config = None
    if args.llm_config:
        llm_config = get_llm_config_arg(args.llm_config)
+        # modify_params must be False for evaluation purpose, for reproducibility and accurancy of results
+        llm_config.modify_params = False

    if llm_config is None:
        raise ValueError(f'Could not find LLM config: --llm_config {args.llm_config}')
--- a/evaluation/benchmarks/browsing_delegation/scripts/run_infer.sh
+++ b/evaluation/benchmarks/browsing_delegation/scripts/run_infer.sh
@@ -20,13 +20,13 @@ if [ -z "$AGENT" ]; then
  AGENT="CodeActAgent"
 fi

-get_agent_version
+get_openhands_version

 echo "AGENT: $AGENT"
-echo "AGENT_VERSION: $AGENT_VERSION"
+echo "OPENHANDS_VERSION: $OPENHANDS_VERSION"
 echo "MODEL_CONFIG: $MODEL_CONFIG"

-EVAL_NOTE="$AGENT_VERSION"
+EVAL_NOTE="$OPENHANDS_VERSION"

 COMMAND="poetry run python evaluation/benchmarks/browsing_delegation/run_infer.py \
  --agent-cls $AGENT \
--- a/evaluation/benchmarks/commit0_bench/README.md
+++ b/evaluation/benchmarks/commit0_bench/README.md
@@ -4,19 +4,18 @@ This folder contains the evaluation harness that we built on top of the original

 The evaluation consists of three steps:

-1. Environment setup: [install python environment](../README.md#development-environment), [configure LLM config](../README.md#configure-openhands-and-your-llm).
+1. Environment setup: [install python environment](../../README.md#development-environment), [configure LLM config](../../README.md#configure-openhands-and-your-llm).
 2. [Run Evaluation](#run-inference-on-commit0-instances): Generate a edit patch for each Commit0 Repo, and get the evaluation results

 ## Setup Environment and LLM Configuration

-Please follow instruction [here](../README.md#setup) to setup your local development environment and LLM.
+Please follow instruction [here](../../README.md#setup) to setup your local development environment and LLM.

 ## OpenHands Commit0 Instance-level Docker Support

 OpenHands supports using the Commit0 Docker for **[inference](#run-inference-on-commit0-instances).
 This is now the default behavior.

-
 ## Run Inference on Commit0 Instances

 Make sure your Docker daemon is running, and you have ample disk space (at least 200-500GB, depends on the Commit0 set you are running on) for the [instance-level docker image](#openhands-commit0-instance-level-docker-support).
--- a/evaluation/benchmarks/commit0_bench/run_infer.py
+++ b/evaluation/benchmarks/commit0_bench/run_infer.py
@@ -571,6 +571,8 @@ if __name__ == '__main__':
    llm_config = None
    if args.llm_config:
        llm_config = get_llm_config_arg(args.llm_config)
+        # modify_params must be False for evaluation purpose, for reproducibility and accurancy of results
+        llm_config.modify_params = False
        llm_config.log_completions = True

    if llm_config is None:
--- a/evaluation/benchmarks/commit0_bench/scripts/run_infer.sh
+++ b/evaluation/benchmarks/commit0_bench/scripts/run_infer.sh
@@ -61,10 +61,10 @@ echo "USE_INSTANCE_IMAGE: $USE_INSTANCE_IMAGE"
 export RUN_WITH_BROWSING=$RUN_WITH_BROWSING
 echo "RUN_WITH_BROWSING: $RUN_WITH_BROWSING"

-get_agent_version
+get_openhands_version

 echo "AGENT: $AGENT"
-echo "AGENT_VERSION: $AGENT_VERSION"
+echo "OPENHANDS_VERSION: $OPENHANDS_VERSION"
 echo "MODEL_CONFIG: $MODEL_CONFIG"
 echo "DATASET: $DATASET"
 echo "HF SPLIT: $SPLIT"
@@ -75,7 +75,7 @@ if [ -z "$USE_HINT_TEXT" ]; then
  export USE_HINT_TEXT=false
 fi
 echo "USE_HINT_TEXT: $USE_HINT_TEXT"
-EVAL_NOTE="$AGENT_VERSION"
+EVAL_NOTE="$OPENHANDS_VERSION"
 # if not using Hint, add -no-hint to the eval note
 if [ "$USE_HINT_TEXT" = false ]; then
  EVAL_NOTE="$EVAL_NOTE-no-hint"
--- a/evaluation/benchmarks/discoverybench/run_infer.py
+++ b/evaluation/benchmarks/discoverybench/run_infer.py
@@ -466,6 +466,8 @@ if __name__ == '__main__':
    llm_config = None
    if args.llm_config:
        llm_config = get_llm_config_arg(args.llm_config)
+        # modify_params must be False for evaluation purpose, for reproducibility and accurancy of results
+        llm_config.modify_params = False
    if llm_config is None:
        raise ValueError(f'Could not find LLM config: --llm_config {args.llm_config}')

--- a/evaluation/benchmarks/discoverybench/scripts/run_infer.sh
+++ b/evaluation/benchmarks/discoverybench/scripts/run_infer.sh
@@ -23,10 +23,10 @@ if [ -z "$AGENT" ]; then
  AGENT="CodeActAgent"
 fi

-get_agent_version
+get_openhands_version

 echo "AGENT: $AGENT"
-echo "AGENT_VERSION: $AGENT_VERSION"
+echo "OPENHANDS_VERSION: $OPENHANDS_VERSION"
 echo "MODEL_CONFIG: $MODEL_CONFIG"

 COMMAND="poetry run python evaluation/benchmarks/discoverybench/run_infer.py \
@@ -35,7 +35,7 @@ COMMAND="poetry run python evaluation/benchmarks/discoverybench/run_infer.py \
  --max-iterations 10 \
  --max-chars 10000000 \
  --eval-num-workers $NUM_WORKERS \
-  --eval-note $AGENT_VERSION"
+  --eval-note $OPENHANDS_VERSION"

 if [ -n "$EVAL_LIMIT" ]; then
  echo "EVAL_LIMIT: $EVAL_LIMIT"
--- a/evaluation/benchmarks/gaia/README.md
+++ b/evaluation/benchmarks/gaia/README.md
@@ -4,9 +4,10 @@ This folder contains evaluation harness for evaluating agents on the [GAIA bench

 ## Setup Environment and LLM Configuration

-Please follow instruction [here](../README.md#setup) to setup your local development environment and LLM.
+Please follow instruction [here](../../README.md#setup) to setup your local development environment and LLM.

 ## Run the evaluation
+
 We are using the GAIA dataset hosted on [Hugging Face](https://huggingface.co/datasets/gaia-benchmark/GAIA).
 Please accept the terms and make sure to have logged in on your computer by `huggingface-cli login` before running the evaluation.

@@ -41,6 +42,7 @@ For example,
 ## Get score

 Then you can get stats by running the following command:
+
 ```bash
 python ./evaluation/benchmarks/gaia/get_score.py \
 --file <path_to/output.json>
--- a/evaluation/benchmarks/gaia/run_infer.py
+++ b/evaluation/benchmarks/gaia/run_infer.py
@@ -238,6 +238,9 @@ if __name__ == '__main__':
    llm_config = None
    if args.llm_config:
        llm_config = get_llm_config_arg(args.llm_config)
+        # modify_params must be False for evaluation purpose, for reproducibility and accurancy of results
+        llm_config.modify_params = False
+
    if llm_config is None:
        raise ValueError(f'Could not find LLM config: --llm_config {args.llm_config}')

--- a/evaluation/benchmarks/gaia/scripts/run_infer.sh
+++ b/evaluation/benchmarks/gaia/scripts/run_infer.sh
@@ -21,17 +21,17 @@ if [ -z "$AGENT" ]; then
  AGENT="CodeActAgent"
 fi

-get_agent_version
+get_openhands_version

 if [ -z "$LEVELS" ]; then
  LEVELS="2023_level1"
  echo "Levels not specified, use default $LEVELS"
 fi

-get_agent_version
+get_openhands_version

 echo "AGENT: $AGENT"
-echo "AGENT_VERSION: $AGENT_VERSION"
+echo "OPENHANDS_VERSION: $OPENHANDS_VERSION"
 echo "MODEL_CONFIG: $MODEL_CONFIG"
 echo "LEVELS: $LEVELS"

@@ -42,7 +42,7 @@ COMMAND="poetry run python ./evaluation/benchmarks/gaia/run_infer.py \
  --level $LEVELS \
  --data-split validation \
  --eval-num-workers $NUM_WORKERS \
-  --eval-note ${AGENT_VERSION}_${LEVELS}"
+  --eval-note ${OPENHANDS_VERSION}_${LEVELS}"

 if [ -n "$EVAL_LIMIT" ]; then
  echo "EVAL_LIMIT: $EVAL_LIMIT"
--- a/evaluation/benchmarks/gorilla/README.md
+++ b/evaluation/benchmarks/gorilla/README.md
@@ -4,7 +4,7 @@ This folder contains evaluation harness we built on top of the original [Gorilla

 ## Setup Environment and LLM Configuration

-Please follow instruction [here](../README.md#setup) to setup your local development environment and LLM.
+Please follow instruction [here](../../README.md#setup) to setup your local development environment and LLM.

 ## Run Inference on APIBench Instances

--- a/evaluation/benchmarks/gorilla/run_infer.py
+++ b/evaluation/benchmarks/gorilla/run_infer.py
@@ -146,6 +146,8 @@ if __name__ == '__main__':
    llm_config = None
    if args.llm_config:
        llm_config = get_llm_config_arg(args.llm_config)
+        # modify_params must be False for evaluation purpose, for reproducibility and accurancy of results
+        llm_config.modify_params = False
    if llm_config is None:
        raise ValueError(f'Could not find LLM config: --llm_config {args.llm_config}')

--- a/evaluation/benchmarks/gorilla/scripts/run_infer.sh
+++ b/evaluation/benchmarks/gorilla/scripts/run_infer.sh
@@ -21,7 +21,7 @@ if [ -z "$AGENT" ]; then
  AGENT="CodeActAgent"
 fi

-get_agent_version
+get_openhands_version

 if [ -z "$HUBS" ]; then
  HUBS="hf,torch,tf"
@@ -29,7 +29,7 @@ if [ -z "$HUBS" ]; then
 fi

 echo "AGENT: $AGENT"
-echo "AGENT_VERSION: $AGENT_VERSION"
+echo "OPENHANDS_VERSION: $OPENHANDS_VERSION"
 echo "MODEL_CONFIG: $MODEL_CONFIG"
 echo "HUBS: $HUBS"

@@ -40,7 +40,7 @@ COMMAND="poetry run python evaluation/benchmarks/gorilla/run_infer.py \
  --hubs $HUBS \
  --data-split validation \
  --eval-num-workers $NUM_WORKERS \
-  --eval-note ${AGENT_VERSION}_${LEVELS}"
+  --eval-note ${OPENHANDS_VERSION}_${LEVELS}"

 if [ -n "$EVAL_LIMIT" ]; then
  echo "EVAL_LIMIT: $EVAL_LIMIT"
--- a/evaluation/benchmarks/gpqa/README.md
+++ b/evaluation/benchmarks/gpqa/README.md
@@ -3,6 +3,7 @@
 Implements the evaluation of agents on the GPQA benchmark introduced in [GPQA: A Graduate-Level Google-Proof Q&A Benchmark](https://arxiv.org/abs/2308.07124).

 This code implements the evaluation of agents on the GPQA Benchmark with Open Book setting.
+
 - The benchmark consists of 448 high-quality and extremely difficult multiple-choice questions in the domains of biology, physics, and chemistry. The questions are intentionally designed to be "Google-proof," meaning that even highly skilled non-expert validators achieve only 34% accuracy despite unrestricted access to the web.
 - Even experts in the corresponding domains achieve only 65% accuracy.
 - State-of-the-art AI systems achieve only 39% accuracy on this challenging dataset.
@@ -11,20 +12,24 @@ This code implements the evaluation of agents on the GPQA Benchmark with Open Bo
 Accurate solving of above graduate level questions would require both tool use (e.g., python for calculations) and web-search for finding related facts as information required for the questions might not be part of the LLM knowledge / training data.

 Further references:
- https://arxiv.org/pdf/2311.12022
- https://paperswithcode.com/dataset/gpqa
- https://github.com/idavidrein/gpqa
+
+- <https://arxiv.org/pdf/2311.12022>
+- <https://paperswithcode.com/dataset/gpqa>
+- <https://github.com/idavidrein/gpqa>

 ## Setup Environment and LLM Configuration

-Please follow instruction [here](../README.md#setup) to setup your local development environment and LLM.
+Please follow instruction [here](../../README.md#setup) to setup your local development environment and LLM.

 ## Run Inference on GPQA Benchmark
+
 'gpqa_main', 'gqpa_diamond', 'gpqa_experts', 'gpqa_extended' -- data split options
 From the root of the OpenHands repo, run the following command:
+
 ```bash
 ./evaluation/benchmarks/gpqa/scripts/run_infer.sh [model_config_name] [git-version] [num_samples_eval] [data_split] [AgentClass]
 ```
+
 You can replace `model_config_name` with any model you set up in `config.toml`.

 - `model_config_name`: The model configuration name from `config.toml` that you want to evaluate.
--- a/evaluation/benchmarks/gpqa/run_infer.py
+++ b/evaluation/benchmarks/gpqa/run_infer.py
@@ -326,6 +326,9 @@ if __name__ == '__main__':
    llm_config = None
    if args.llm_config:
        llm_config = get_llm_config_arg(args.llm_config)
+        # modify_params must be False for evaluation purpose, for reproducibility and accurancy of results
+        llm_config.modify_params = False
+
    if llm_config is None:
        raise ValueError(f'Could not find LLM config: --llm_config {args.llm_config}')

--- a/evaluation/benchmarks/gpqa/scripts/run_infer.sh
+++ b/evaluation/benchmarks/gpqa/scripts/run_infer.sh
@@ -27,10 +27,10 @@ if [ -z "$DATA_SPLIT" ]; then
  DATA_SPLIT="gpqa_diamond"
 fi

-get_agent_version
+get_openhands_version

 echo "AGENT: $AGENT"
-echo "AGENT_VERSION: $AGENT_VERSION"
+echo "OPENHANDS_VERSION: $OPENHANDS_VERSION"
 echo "MODEL_CONFIG: $MODEL_CONFIG"

 COMMAND="poetry run python evaluation/benchmarks/gpqa/run_infer.py \
@@ -39,7 +39,7 @@ COMMAND="poetry run python evaluation/benchmarks/gpqa/run_infer.py \
  --max-iterations 10 \
  --eval-num-workers $NUM_WORKERS \
  --data-split $DATA_SPLIT \
-  --eval-note $AGENT_VERSION"
+  --eval-note $OPENHANDS_VERSION"

 if [ -n "$EVAL_LIMIT" ]; then
  echo "EVAL_LIMIT: $EVAL_LIMIT"
--- a/evaluation/benchmarks/humanevalfix/README.md
+++ b/evaluation/benchmarks/humanevalfix/README.md
@@ -4,7 +4,7 @@ Implements evaluation of agents on HumanEvalFix from the HumanEvalPack benchmark

 ## Setup Environment and LLM Configuration

-Please follow instruction [here](../README.md#setup) to setup your local development environment and LLM.
+Please follow instruction [here](../../README.md#setup) to setup your local development environment and LLM.

 ## Run Inference on HumanEvalFix

@@ -14,13 +14,11 @@ Please follow instruction [here](../README.md#setup) to setup your local develop

 You can replace `eval_gpt4_1106_preview` with any model you set up in `config.toml`.

-
 ## Examples

 For each problem, OpenHands is given a set number of iterations to fix the failing code. The history field shows each iteration's response to correct its code that fails any test case.

-
-```
+```json
 {
    "task_id": "Python/2",
    "instruction": "Please fix the function in Python__2.py such that all test cases pass.\nEnvironment has been set up for you to start working. You may assume all necessary tools are installed.\n\n# Problem Statement\ndef truncate_number(number: float) -> float:\n    return number % 1.0 + 1.0\n\n\n\n\n\n\ndef check(truncate_number):\n    assert truncate_number(3.5) == 0.5\n    assert abs(truncate_number(1.33) - 0.33) < 1e-6\n    assert abs(truncate_number(123.456) - 0.456) < 1e-6\n\ncheck(truncate_number)\n\nIMPORTANT: You should ONLY interact with the environment provided to you AND NEVER ASK FOR HUMAN HELP.\nYou should NOT modify any existing test case files. If needed, you can add new test cases in a NEW file to reproduce the issue.\nYou SHOULD INCLUDE PROPER INDENTATION in your edit commands.\nWhen you think you have fixed the issue through code changes, please finish the interaction using the "finish" tool.\n",
--- a/evaluation/benchmarks/humanevalfix/run_infer.py
+++ b/evaluation/benchmarks/humanevalfix/run_infer.py
@@ -285,6 +285,8 @@ if __name__ == '__main__':
    llm_config = None
    if args.llm_config:
        llm_config = get_llm_config_arg(args.llm_config)
+        # modify_params must be False for evaluation purpose, for reproducibility and accurancy of results
+        llm_config.modify_params = False
    if llm_config is None:
        raise ValueError(f'Could not find LLM config: --llm_config {args.llm_config}')

--- a/evaluation/benchmarks/humanevalfix/scripts/run_infer.sh
+++ b/evaluation/benchmarks/humanevalfix/scripts/run_infer.sh
@@ -58,10 +58,10 @@ if [ -z "$AGENT" ]; then
  AGENT="CodeActAgent"
 fi

-get_agent_version
+get_openhands_version

 echo "AGENT: $AGENT"
-echo "AGENT_VERSION: $AGENT_VERSION"
+echo "OPENHANDS_VERSION: $OPENHANDS_VERSION"
 echo "MODEL_CONFIG: $MODEL_CONFIG"

 COMMAND="poetry run python evaluation/benchmarks/humanevalfix/run_infer.py \
@@ -69,7 +69,7 @@ COMMAND="poetry run python evaluation/benchmarks/humanevalfix/run_infer.py \
  --llm-config $MODEL_CONFIG \
  --max-iterations 10 \
  --eval-num-workers $NUM_WORKERS \
-  --eval-note $AGENT_VERSION"
+  --eval-note $OPENHANDS_VERSION"

 if [ -n "$EVAL_LIMIT" ]; then
  echo "EVAL_LIMIT: $EVAL_LIMIT"
--- a/evaluation/benchmarks/logic_reasoning/README.md
+++ b/evaluation/benchmarks/logic_reasoning/README.md
@@ -4,9 +4,10 @@ This folder contains evaluation harness for evaluating agents on the logic reaso

 ## Setup Environment and LLM Configuration

-Please follow instruction [here](../README.md#setup) to setup your local development environment and LLM.
+Please follow instruction [here](../../README.md#setup) to setup your local development environment and LLM.

 ## Run Inference on logic_reasoning
+
 The following code will run inference on the first example of the ProofWriter dataset,

 ```bash
--- a/evaluation/benchmarks/logic_reasoning/run_infer.py
+++ b/evaluation/benchmarks/logic_reasoning/run_infer.py
@@ -272,7 +272,7 @@ if __name__ == '__main__':
        default='ProofWriter',
    )
    parser.add_argument(
-        '--data_split',
+        '--data-split',
        type=str,
        help='data split to evaluate on {validation}',  # right now we only support validation split
        default='validation',
@@ -288,6 +288,8 @@ if __name__ == '__main__':
    llm_config = None
    if args.llm_config:
        llm_config = get_llm_config_arg(args.llm_config)
+        # modify_params must be False for evaluation purpose, for reproducibility and accurancy of results
+        llm_config.modify_params = False
    if llm_config is None:
        raise ValueError(f'Could not find LLM config: --llm_config {args.llm_config}')

--- a/evaluation/benchmarks/logic_reasoning/scripts/run_infer.sh
+++ b/evaluation/benchmarks/logic_reasoning/scripts/run_infer.sh
@@ -28,10 +28,10 @@ if [ -z "$DATASET" ]; then
  DATASET="ProofWriter"
 fi

-get_agent_version
+get_openhands_version

 echo "AGENT: $AGENT"
-echo "AGENT_VERSION: $AGENT_VERSION"
+echo "OPENHANDS_VERSION: $OPENHANDS_VERSION"
 echo "MODEL_CONFIG: $MODEL_CONFIG"

 COMMAND="poetry run python evaluation/benchmarks/logic_reasoning/run_infer.py \
@@ -40,7 +40,7 @@ COMMAND="poetry run python evaluation/benchmarks/logic_reasoning/run_infer.py \
  --dataset $DATASET \
  --max-iterations 10 \
  --eval-num-workers $NUM_WORKERS \
-  --eval-note $AGENT_VERSION"
+  --eval-note $OPENHANDS_VERSION"

 if [ -n "$EVAL_LIMIT" ]; then
  echo "EVAL_LIMIT: $EVAL_LIMIT"
--- a/evaluation/benchmarks/miniwob/README.md
+++ b/evaluation/benchmarks/miniwob/README.md
@@ -4,7 +4,7 @@ This folder contains evaluation for [MiniWoB++](https://miniwob.farama.org/) ben

 ## Setup Environment and LLM Configuration

-Please follow instruction [here](../README.md#setup) to setup your local development environment and LLM.
+Please follow instruction [here](../../README.md#setup) to setup your local development environment and LLM.

 ## Test if your environment works

@@ -42,7 +42,6 @@ poetry run python evaluation/benchmarks/miniwob/get_success_rate.py evaluation/e

 You can start your own fork of [our huggingface evaluation outputs](https://huggingface.co/spaces/OpenHands/evaluation) and submit a PR of your evaluation results following the guide [here](https://huggingface.co/docs/hub/en/repositories-pull-requests-discussions#pull-requests-and-discussions).

-
 ## BrowsingAgent V1.0 result

 Tested on BrowsingAgent V1.0
--- a/evaluation/benchmarks/miniwob/run_infer.py
+++ b/evaluation/benchmarks/miniwob/run_infer.py
@@ -231,6 +231,8 @@ if __name__ == '__main__':
    llm_config = None
    if args.llm_config:
        llm_config = get_llm_config_arg(args.llm_config)
+        # modify_params must be False for evaluation purpose, for reproducibility and accurancy of results
+        llm_config.modify_params = False
    if llm_config is None:
        raise ValueError(f'Could not find LLM config: --llm_config {args.llm_config}')

--- a/evaluation/benchmarks/miniwob/scripts/run_infer.sh
+++ b/evaluation/benchmarks/miniwob/scripts/run_infer.sh
@@ -25,13 +25,13 @@ if [ -z "$AGENT" ]; then
  AGENT="BrowsingAgent"
 fi

-get_agent_version
+get_openhands_version

 echo "AGENT: $AGENT"
-echo "AGENT_VERSION: $AGENT_VERSION"
+echo "OPENHANDS_VERSION: $OPENHANDS_VERSION"
 echo "MODEL_CONFIG: $MODEL_CONFIG"

-EVAL_NOTE="${AGENT_VERSION}_${NOTE}"
+EVAL_NOTE="${OPENHANDS_VERSION}_${NOTE}"

 COMMAND="export PYTHONPATH=evaluation/benchmarks/miniwob:\$PYTHONPATH && poetry run python evaluation/benchmarks/miniwob/run_infer.py \
  --agent-cls $AGENT \
--- a/evaluation/benchmarks/mint/run_infer.py
+++ b/evaluation/benchmarks/mint/run_infer.py
@@ -279,6 +279,8 @@ if __name__ == '__main__':
    llm_config = None
    if args.llm_config:
        llm_config = get_llm_config_arg(args.llm_config)
+        # modify_params must be False for evaluation purpose, for reproducibility and accurancy of results
+        llm_config.modify_params = False
    if llm_config is None:
        raise ValueError(f'Could not find LLM config: --llm_config {args.llm_config}')

--- a/evaluation/benchmarks/mint/scripts/run_infer.sh
+++ b/evaluation/benchmarks/mint/scripts/run_infer.sh
@@ -18,10 +18,10 @@ checkout_eval_branch
 # Only 'CodeActAgent' is supported for MINT now
 AGENT="CodeActAgent"

-get_agent_version
+get_openhands_version

 echo "AGENT: $AGENT"
-echo "AGENT_VERSION: $AGENT_VERSION"
+echo "OPENHANDS_VERSION: $OPENHANDS_VERSION"

 export PYTHONPATH=$(pwd)

--- a/evaluation/benchmarks/ml_bench/README.md
+++ b/evaluation/benchmarks/ml_bench/README.md
@@ -12,7 +12,7 @@ For more details on the ML-Bench task and dataset, please refer to the paper: [M

 ## Setup Environment and LLM Configuration

-Please follow instruction [here](../README.md#setup) to setup your local development environment and LLM.
+Please follow instruction [here](../../README.md#setup) to setup your local development environment and LLM.

 ## Run Inference on ML-Bench

--- a/evaluation/benchmarks/ml_bench/run_analysis.py
+++ b/evaluation/benchmarks/ml_bench/run_analysis.py
@@ -124,6 +124,9 @@ if __name__ == '__main__':
    # for details of how to set `llm_config`
    if args.llm_config:
        specified_llm_config = get_llm_config_arg(args.llm_config)
+        # modify_params must be False for evaluation purpose, for reproducibility and accurancy of results
+        specified_llm_config.modify_params = False
+
        if specified_llm_config:
            config.llm = specified_llm_config
    logger.info(f'Config for evaluation: {config}')
--- a/evaluation/benchmarks/ml_bench/run_infer.py
+++ b/evaluation/benchmarks/ml_bench/run_infer.py
@@ -292,6 +292,8 @@ if __name__ == '__main__':
    llm_config = None
    if args.llm_config:
        llm_config = get_llm_config_arg(args.llm_config)
+        # modify_params must be False for evaluation purpose, for reproducibility and accurancy of results
+        llm_config.modify_params = False
    if llm_config is None:
        raise ValueError(f'Could not find LLM config: --llm_config {args.llm_config}')

--- a/evaluation/benchmarks/ml_bench/scripts/run_infer.sh
+++ b/evaluation/benchmarks/ml_bench/scripts/run_infer.sh
@@ -26,10 +26,10 @@ if [ -z "$AGENT" ]; then
  AGENT="CodeActAgent"
 fi

-get_agent_version
+get_openhands_version

 echo "AGENT: $AGENT"
-echo "AGENT_VERSION: $AGENT_VERSION"
+echo "OPENHANDS_VERSION: $OPENHANDS_VERSION"
 echo "MODEL_CONFIG: $MODEL_CONFIG"

 COMMAND="poetry run python evaluation/benchmarks/ml_bench/run_infer.py \
@@ -37,7 +37,7 @@ COMMAND="poetry run python evaluation/benchmarks/ml_bench/run_infer.py \
  --llm-config $MODEL_CONFIG \
  --max-iterations 10 \
  --eval-num-workers $NUM_WORKERS \
-  --eval-note $AGENT_VERSION"
+  --eval-note $OPENHANDS_VERSION"

 if [ -n "$EVAL_LIMIT" ]; then
  echo "EVAL_LIMIT: $EVAL_LIMIT"
--- a/evaluation/benchmarks/scienceagentbench/README.md
+++ b/evaluation/benchmarks/scienceagentbench/README.md
@@ -1,10 +1,10 @@
 # ScienceAgentBench Evaluation with OpenHands

-This folder contains the evaluation harness for [ScienceAgentBench](https://osu-nlp-group.github.io/ScienceAgentBench/) (paper: https://arxiv.org/abs/2410.05080).
+This folder contains the evaluation harness for [ScienceAgentBench](https://osu-nlp-group.github.io/ScienceAgentBench/) (paper: <https://arxiv.org/abs/2410.05080>).

 ## Setup Environment and LLM Configuration

-Please follow instruction [here](../README.md#setup) to setup your local development environment and LLM.
+Please follow instruction [here](../../README.md#setup) to setup your local development environment and LLM.

 ## Setup ScienceAgentBench

@@ -45,6 +45,7 @@ After the inference is completed, you may use the following command to extract n
 ```bash
 python post_proc.py [log_fname]
 ```
+
 - `log_fname`, e.g. `evaluation/.../output.jsonl`, is the automatically saved trajectory log of an OpenHands agent.

 Output will be write to e.g. `evaluation/.../output.converted.jsonl`
--- a/evaluation/benchmarks/scienceagentbench/run_infer.py
+++ b/evaluation/benchmarks/scienceagentbench/run_infer.py
@@ -251,7 +251,7 @@ If the program uses some packages that are incompatible, please figure out alter
 if __name__ == '__main__':
    parser = get_parser()
    parser.add_argument(
-        '--use_knowledge',
+        '--use-knowledge',
        type=str,
        default='false',
        choices=['true', 'false'],
@@ -272,6 +272,8 @@ if __name__ == '__main__':
    llm_config = None
    if args.llm_config:
        llm_config = get_llm_config_arg(args.llm_config)
+        # modify_params must be False for evaluation purpose, for reproducibility and accurancy of results
+        llm_config.modify_params = False
    if llm_config is None:
        raise ValueError(f'Could not find LLM config: --llm_config {args.llm_config}')

--- a/evaluation/benchmarks/scienceagentbench/scripts/run_infer.sh
+++ b/evaluation/benchmarks/scienceagentbench/scripts/run_infer.sh
@@ -26,19 +26,19 @@ if [ -z "$USE_KNOWLEDGE" ]; then
  USE_KNOWLEDGE=false
 fi

-get_agent_version
+get_openhands_version

 echo "AGENT: $AGENT"
-echo "AGENT_VERSION: $AGENT_VERSION"
+echo "OPENHANDS_VERSION: $OPENHANDS_VERSION"
 echo "MODEL_CONFIG: $MODEL_CONFIG"

 COMMAND="poetry run python evaluation/benchmarks/scienceagentbench/run_infer.py \
  --agent-cls $AGENT \
  --llm-config $MODEL_CONFIG \
-  --use_knowledge $USE_KNOWLEDGE \
+  --use-knowledge $USE_KNOWLEDGE \
  --max-iterations 30 \
  --eval-num-workers $NUM_WORKERS \
-  --eval-note $AGENT_VERSION" \
+  --eval-note $OPENHANDS_VERSION" \

 if [ -n "$EVAL_LIMIT" ]; then
  echo "EVAL_LIMIT: $EVAL_LIMIT"
--- a/evaluation/benchmarks/swe_bench/README.md
+++ b/evaluation/benchmarks/swe_bench/README.md
@@ -6,20 +6,19 @@ This folder contains the evaluation harness that we built on top of the original

 The evaluation consists of three steps:

-1. Environment setup: [install python environment](../README.md#development-environment), [configure LLM config](../README.md#configure-openhands-and-your-llm), and [pull docker](#openhands-swe-bench-instance-level-docker-support).
+1. Environment setup: [install python environment](../../README.md#development-environment), [configure LLM config](../../README.md#configure-openhands-and-your-llm), and [pull docker](#openhands-swe-bench-instance-level-docker-support).
 2. [Run inference](#run-inference-on-swe-bench-instances): Generate a edit patch for each Github issue
 3. [Evaluate patches using SWE-Bench docker](#evaluate-generated-patches)

 ## Setup Environment and LLM Configuration

-Please follow instruction [here](../README.md#setup) to setup your local development environment and LLM.
+Please follow instruction [here](../../README.md#setup) to setup your local development environment and LLM.

 ## OpenHands SWE-Bench Instance-level Docker Support

 OpenHands now support using the [official evaluation docker](https://github.com/princeton-nlp/SWE-bench/blob/main/docs/20240627_docker/README.md) for both **[inference](#run-inference-on-swe-bench-instances) and [evaluation](#evaluate-generated-patches)**.
 This is now the default behavior.

-
 ## Run Inference on SWE-Bench Instances

 Make sure your Docker daemon is running, and you have ample disk space (at least 200-500GB, depends on the SWE-Bench set you are running on) for the [instance-level docker image](#openhands-swe-bench-instance-level-docker-support).
@@ -52,7 +51,8 @@ default, it is set to 1.
 - `dataset_split`, split for the huggingface dataset. e.g., `test`, `dev`. Default to `test`.

 There are also two optional environment variables you can set.
-```
+
+```bash
 export USE_HINT_TEXT=true # if you want to use hint text in the evaluation. Default to false. Ignore this if you are not sure.
 export USE_INSTANCE_IMAGE=true # if you want to use instance-level docker images. Default to true
 ```
@@ -127,6 +127,7 @@ With `output.jsonl` file, you can run `eval_infer.sh` to evaluate generated patc
 **This evaluation is performed using the official dockerized evaluation announced [here](https://github.com/princeton-nlp/SWE-bench/blob/main/docs/20240627_docker/README.md).**

 > If you want to evaluate existing results, you should first run this to clone existing outputs
+>
 >```bash
 >git clone https://huggingface.co/spaces/OpenHands/evaluation evaluation/evaluation_outputs
 >```
@@ -143,6 +144,7 @@ Then you can run the following:
 ```

 The script now accepts optional arguments:
+
 - `instance_id`: Specify a single instance to evaluate (optional)
 - `dataset_name`: The name of the dataset to use (default: `"princeton-nlp/SWE-bench_Lite"`)
 - `split`: The split of the dataset to use (default: `"test"`)
@@ -179,7 +181,6 @@ To clean-up all existing runtimes that you've already started, run:
 ALLHANDS_API_KEY="YOUR-API-KEY" ./evaluation/benchmarks/swe_bench/scripts/cleanup_remote_runtime.sh
 ```

-
 ## Visualize Results

 First you need to clone `https://huggingface.co/spaces/OpenHands/evaluation` and add your own running results from openhands into the `outputs` of the cloned repo.
@@ -189,6 +190,7 @@ git clone https://huggingface.co/spaces/OpenHands/evaluation
 ```

 **(optional) setup streamlit environment with conda**:
+
 ```bash
 cd evaluation
 conda create -n streamlit python=3.10
--- a/evaluation/benchmarks/swe_bench/prompt.py
+++ b/evaluation/benchmarks/swe_bench/prompt.py
@@ -1,28 +0,0 @@
-CODEACT_SWE_PROMPT = """Now, you're going to solve this issue on your own. Your terminal session has started and you're in the repository's root directory. You can use any bash commands or the special interface to help you. Edit all the files you need to and run any checks or tests that you want.
-Remember, YOU CAN ONLY ENTER ONE COMMAND AT A TIME. You should always wait for feedback after every command.
-When you're satisfied with all of the changes you've made, you can use the "finish" tool to finish the interaction.
-Note however that you cannot use any interactive session commands (e.g. vim) in this environment, but you can write scripts and run them. E.g. you can write a python script and then run it with `python <script_name>.py`.
-
-NOTE ABOUT THE EDIT COMMAND: Indentation really matters! When editing a file, make sure to insert appropriate indentation before each line!
-
-IMPORTANT TIPS:
-1. Always start by trying to replicate the bug that the issues discusses.
-    If the issue includes code for reproducing the bug, we recommend that you re-implement that in your environment, and run it to make sure you can reproduce the bug.
-    Then start trying to fix it.
-    When you think you've fixed the bug, re-run the bug reproduction script to make sure that the bug has indeed been fixed.
-
-    If the bug reproduction script does not print anything when it successfully runs, we recommend adding a print("Script completed successfully, no errors.") command at the end of the file,
-    so that you can be sure that the script indeed ran fine all the way through.
-
-2. If you run a command and it doesn't work, try running a different command. A command that did not work once will not work the second time unless you modify it!
-
-3. If you open a file and need to get to an area around a specific line that is not in the first 100 lines, say line 583, don't just use the scroll_down command multiple times. Instead, use the goto 583 command. It's much quicker.
-
-4. If the bug reproduction script requires inputting/reading a specific file, such as buggy-input.png, and you'd like to understand how to input that file, conduct a search in the existing repo code, to see whether someone else has already done that. Do this by running the command: find_file("buggy-input.png") If that doesn't work, use the linux 'find' command.
-
-5. Always make sure to look at the currently open file and the current working directory (which appears right after the currently open file). The currently open file might be in a different directory than the working directory! Note that some commands, such as 'create', open files, so they might change the current  open file.
-
-6. When editing files, it is easy to accidentally specify a wrong line number or to write code with incorrect indentation. Always check the code after you issue an edit to make sure that it reflects what you wanted to accomplish. If it didn't, issue another command to fix it.
-
-[Current directory: /workspace/{workspace_dir_name}]
-"""
--- a/evaluation/benchmarks/swe_bench/run_infer.py
+++ b/evaluation/benchmarks/swe_bench/run_infer.py
@@ -9,13 +9,13 @@ import toml
 from datasets import load_dataset

 import openhands.agenthub
-from evaluation.benchmarks.swe_bench.prompt import CODEACT_SWE_PROMPT
 from evaluation.utils.shared import (
    EvalException,
    EvalMetadata,
    EvalOutput,
    assert_and_raise,
    codeact_user_response,
+    is_fatal_evaluation_error,
    make_metadata,
    prepare_dataset,
    reset_logger_for_multiprocessing,
@@ -45,7 +45,6 @@ RUN_WITH_BROWSING = os.environ.get('RUN_WITH_BROWSING', 'false').lower() == 'tru

 AGENT_CLS_TO_FAKE_USER_RESPONSE_FN = {
    'CodeActAgent': codeact_user_response,
-    'CodeActSWEAgent': codeact_user_response,
 }


@@ -56,40 +55,28 @@ def _get_swebench_workspace_dir_name(instance: pd.Series) -> str:
 def get_instruction(instance: pd.Series, metadata: EvalMetadata):
    workspace_dir_name = _get_swebench_workspace_dir_name(instance)
    # Prepare instruction
-    if metadata.agent_class == 'CodeActSWEAgent':
-        instruction = (
-            'We are currently solving the following issue within our repository. Here is the issue text:\n'
-            '--- BEGIN ISSUE ---\n'
-            f'{instance.problem_statement}\n'
-            '--- END ISSUE ---\n\n'
-        )
-        if USE_HINT_TEXT and instance.hints_text:
-            instruction += (
-                f'--- BEGIN HINTS ---\n{instance.hints_text}\n--- END HINTS ---\n'
-            )
-        instruction += CODEACT_SWE_PROMPT.format(workspace_dir_name=workspace_dir_name)
-    else:
-        # Instruction based on Anthropic's official trajectory
-        # https://github.com/eschluntz/swe-bench-experiments/tree/main/evaluation/verified/20241022_tools_claude-3-5-sonnet-updated/trajs
-        instruction = (
-            '<uploaded_files>\n'
-            f'/workspace/{workspace_dir_name}\n'
-            '</uploaded_files>\n'
-            f"I've uploaded a python code repository in the directory {workspace_dir_name}. Consider the following PR description:\n\n"
-            f'<pr_description>\n'
-            f'{instance.problem_statement}\n'
-            '</pr_description>\n\n'
-            'Can you help me implement the necessary changes to the repository so that the requirements specified in the <pr_description> are met?\n'
-            "I've already taken care of all changes to any of the test files described in the <pr_description>. This means you DON'T have to modify the testing logic or any of the tests in any way!\n"
-            'Your task is to make the minimal changes to non-tests files in the /workspace directory to ensure the <pr_description> is satisfied.\n'
-            'Follow these steps to resolve the issue:\n'
-            '1. As a first step, it might be a good idea to explore the repo to familiarize yourself with its structure.\n'
-            '2. Create a script to reproduce the error and execute it with `python <filename.py>` using the BashTool, to confirm the error\n'
-            '3. Edit the sourcecode of the repo to resolve the issue\n'
-            '4. Rerun your reproduce script and confirm that the error is fixed!\n'
-            '5. Think about edgecases and make sure your fix handles them as well\n'
-            "Your thinking should be thorough and so it's fine if it's very long.\n"
-        )
+
+    # Instruction based on Anthropic's official trajectory
+    # https://github.com/eschluntz/swe-bench-experiments/tree/main/evaluation/verified/20241022_tools_claude-3-5-sonnet-updated/trajs
+    instruction = (
+        '<uploaded_files>\n'
+        f'/workspace/{workspace_dir_name}\n'
+        '</uploaded_files>\n'
+        f"I've uploaded a python code repository in the directory {workspace_dir_name}. Consider the following PR description:\n\n"
+        f'<pr_description>\n'
+        f'{instance.problem_statement}\n'
+        '</pr_description>\n\n'
+        'Can you help me implement the necessary changes to the repository so that the requirements specified in the <pr_description> are met?\n'
+        "I've already taken care of all changes to any of the test files described in the <pr_description>. This means you DON'T have to modify the testing logic or any of the tests in any way!\n"
+        'Your task is to make the minimal changes to non-tests files in the /workspace directory to ensure the <pr_description> is satisfied.\n'
+        'Follow these steps to resolve the issue:\n'
+        '1. As a first step, it might be a good idea to explore the repo to familiarize yourself with its structure.\n'
+        '2. Create a script to reproduce the error and execute it with `python <filename.py>` using the BashTool, to confirm the error\n'
+        '3. Edit the sourcecode of the repo to resolve the issue\n'
+        '4. Rerun your reproduce script and confirm that the error is fixed!\n'
+        '5. Think about edgecases and make sure your fix handles them as well\n'
+        "Your thinking should be thorough and so it's fine if it's very long.\n"
+    )

    if RUN_WITH_BROWSING:
        instruction += (
@@ -383,6 +370,7 @@ def process_instance(
    instance: pd.Series,
    metadata: EvalMetadata,
    reset_logger: bool = True,
+    runtime_failure_count: int = 0,
 ) -> EvalOutput:
    config = get_config(instance, metadata)

@@ -393,6 +381,15 @@ def process_instance(
    else:
        logger.info(f'Starting evaluation for instance {instance.instance_id}.')

+    # Increase resource_factor with increasing attempt_id
+    if runtime_failure_count > 0:
+        config.sandbox.remote_runtime_resource_factor = min(
+            config.sandbox.remote_runtime_resource_factor * (2**runtime_failure_count),
+            2,  # hardcode maximum resource factor to 2
+        )
+        logger.warning(
+            f'This is the second attempt for instance {instance.instance_id}, setting resource factor to {config.sandbox.remote_runtime_resource_factor}'
+        )
    runtime = create_runtime(config)
    call_async_from_sync(runtime.connect)

@@ -414,11 +411,7 @@ def process_instance(
        )

        # if fatal error, throw EvalError to trigger re-run
-        if (
-            state.last_error
-            and 'fatal error during agent execution' in state.last_error
-            and 'stuck in a loop' not in state.last_error
-        ):
+        if is_fatal_evaluation_error(state.last_error):
            raise EvalException('Fatal error detected: ' + state.last_error)

        # ======= THIS IS SWE-Bench specific =======
@@ -504,6 +497,8 @@ if __name__ == '__main__':
    if args.llm_config:
        llm_config = get_llm_config_arg(args.llm_config)
        llm_config.log_completions = True
+        # modify_params must be False for evaluation purpose, for reproducibility and accurancy of results
+        llm_config.modify_params = False

    if llm_config is None:
        raise ValueError(f'Could not find LLM config: --llm_config {args.llm_config}')
--- a/evaluation/benchmarks/swe_bench/scripts/eval/summarize_outputs.py
+++ b/evaluation/benchmarks/swe_bench/scripts/eval/summarize_outputs.py
@@ -1,8 +1,14 @@
 #!/usr/bin/env python3
 import argparse
+import glob
 import json
+import os
 from collections import Counter

+import pandas as pd
+import random
+import numpy as np
+
 from openhands.events.serialization import event_from_dict
 from openhands.events.utils import get_pairs_from_events

@@ -10,25 +16,34 @@ ERROR_KEYWORDS = [
    'Agent encountered an error while processing the last action',
    'APIError',
    'Action execution failed',
+    'litellm.Timeout: APITimeoutError',
 ]

-if __name__ == '__main__':
-    parser = argparse.ArgumentParser()
-    parser.add_argument('output_file', type=str, help='The file to summarize')
-    args = parser.parse_args()

-    with open(args.output_file, 'r') as file:
+def get_bootstrap_accuracy_error_bars(values: float | int | bool, num_samples: int = 1000, p_value=0.05) -> tuple[float, float]:
+    sorted_vals = np.sort(
+        [
+            np.mean(random.sample(values, len(values) // 2))
+            for _ in range(num_samples)
+        ]
+    )
+    bottom_idx = int(num_samples * p_value / 2)
+    top_idx = int(num_samples * (1.0 - p_value / 2))
+    return (sorted_vals[bottom_idx], sorted_vals[top_idx])
+
+
+def process_file(file_path):
+    with open(file_path, 'r') as file:
        lines = file.readlines()

    num_lines = len(lines)
    num_error_lines = 0
    num_agent_stuck_in_loop = 0
-
    num_resolved = 0
+    resolved_arr = []
    num_empty_patch = 0
-
+    num_unfinished_runs = 0
    error_counter = Counter()
-
    main_agent_cost = []
    editor_cost = []
    num_turns = []
@@ -36,6 +51,11 @@ if __name__ == '__main__':
    for line in lines:
        _d = json.loads(line)

+        if 'metrics' not in _d or _d['metrics'] is None:
+            # this is a failed run
+            num_unfinished_runs += 1
+            continue
+
        # Cost
        costs = _d['metrics'].get('costs', [])
        _cur_main_agent_cost = 0
@@ -69,6 +89,9 @@ if __name__ == '__main__':
        resolved = report.get('resolved', False)
        if resolved:
            num_resolved += 1
+            resolved_arr.append(1)
+        else:
+            resolved_arr.append(0)

        # Error
        error = _d.get('error', None)
@@ -89,30 +112,188 @@ if __name__ == '__main__':
                num_error_lines += 1
                break

-    # print the error counter (with percentage)
-    print(
-        f'Number of resolved: {num_resolved} / {num_lines} ({num_resolved / num_lines * 100:.2f}%)'
-    )
-    print(
-        f'Number of empty patch: {num_empty_patch} / {num_lines} ({num_empty_patch / num_lines * 100:.2f}%)'
-    )
-    print(
-        f'Number of error lines: {num_error_lines} / {num_lines} ({num_error_lines / num_lines * 100:.2f}%)'
-    )
-    print(
-        f'Number of agent stuck in loop: {num_agent_stuck_in_loop} / {num_lines} ({num_agent_stuck_in_loop / num_lines * 100:.2f}%)'
-    )
-    assert len(num_turns) == num_lines
-    assert len(main_agent_cost) == num_lines
-    assert len(editor_cost) == num_lines
-    print('## Statistics')
-    print(f'Avg. num of turns per instance: {sum(num_turns) / num_lines:.2f}')
-    print(f'Avg. agent cost per instance: {sum(main_agent_cost) / num_lines:.2f} USD')
-    print(f'Avg. editor cost per instance: {sum(editor_cost) / num_lines:.2f} USD')
-    print(
-        f'Avg. total cost per instance: {(sum(main_agent_cost) + sum(editor_cost)) / num_lines:.2f} USD'
+    return {
+        'file_path': file_path,
+        'total_instances': num_lines,
+        'resolved': {
+            'count': num_resolved,
+            'percentage': (num_resolved / num_lines * 100) if num_lines > 0 else 0,
+            'ci': tuple(x * 100 for x in get_bootstrap_accuracy_error_bars(resolved_arr)),
+        },
+        'empty_patches': {
+            'count': num_empty_patch,
+            'percentage': (num_empty_patch / num_lines * 100) if num_lines > 0 else 0,
+        },
+        'unfinished_runs': {
+            'count': num_unfinished_runs,
+            'percentage': (num_unfinished_runs / num_lines * 100)
+            if num_lines > 0
+            else 0,
+        },
+        'errors': {
+            'total': num_error_lines,
+            'percentage': (num_error_lines / num_lines * 100) if num_lines > 0 else 0,
+            'stuck_in_loop': {
+                'count': num_agent_stuck_in_loop,
+                'percentage': (num_agent_stuck_in_loop / num_lines * 100)
+                if num_lines > 0
+                else 0,
+            },
+            'breakdown': {
+                str(error): {
+                    'count': count,
+                    'percentage': (count / num_lines * 100) if num_lines > 0 else 0,
+                }
+                for error, count in error_counter.items()
+            },
+        },
+        'costs': {
+            'main_agent': sum(main_agent_cost),
+            'editor': sum(editor_cost),
+            'total': sum(main_agent_cost) + sum(editor_cost),
+        },
+        'statistics': {
+            'avg_turns': sum(num_turns) / num_lines if num_lines > 0 else 0,
+            'costs': {
+                'main_agent': sum(main_agent_cost) / num_lines if num_lines > 0 else 0,
+                'editor': sum(editor_cost) / num_lines if num_lines > 0 else 0,
+                'total': (sum(main_agent_cost) + sum(editor_cost)) / num_lines
+                if num_lines > 0
+                else 0,
+            },
+        },
+    }
+
+
+def aggregate_directory(input_path) -> pd.DataFrame:
+    # Process all output.jsonl files in subdirectories
+    pattern = os.path.join(input_path, '**/output.jsonl')
+    files = glob.glob(pattern, recursive=True)
+    print(f'Processing {len(files)} files from directory {input_path}')
+
+    # Process each file silently and collect results
+    results = []
+    for file_path in files:
+        try:
+            result = process_file(file_path)
+            results.append(result)
+        except Exception as e:
+            print(f'Error processing {file_path}: {str(e)}')
+            import traceback
+
+            traceback.print_exc()
+            continue
+
+    # Convert results to pandas DataFrame and sort by resolve rate
+    df = pd.DataFrame(results)
+
+    # Extract directory name from file path
+    df['directory'] = df['file_path'].apply(
+        lambda x: os.path.basename(os.path.dirname(x))
    )

-    print('## Detailed error breakdown:')
-    for error, count in error_counter.items():
-        print(f'{error}: {count} ({count / num_lines * 100:.2f}%)')
+    df['resolve_rate'] = df['resolved'].apply(lambda x: x['percentage'])
+    df['resolve_rate_ci'] = df['resolved'].apply(lambda x: x['ci'])
+    df['empty_patch_rate'] = df['empty_patches'].apply(lambda x: x['percentage'])
+    df['unfinished_rate'] = df['unfinished_runs'].apply(lambda x: x['percentage'])
+    df['avg_turns'] = df['statistics'].apply(lambda x: x['avg_turns'])
+    df['error_rate'] = df['errors'].apply(lambda x: x['percentage'])
+    df['avg_cost'] = df['statistics'].apply(lambda x: x['costs']['total'])
+
+    df = df.sort_values('resolve_rate', ascending=False)
+
+    return df
+
+
+if __name__ == '__main__':
+    parser = argparse.ArgumentParser()
+    parser.add_argument(
+        'input_path', type=str, help='The file or directory to summarize'
+    )
+    parser.add_argument(
+        '--output',
+        type=str,
+        help='Output JSONL file for results',
+        default='summary_results.jsonl',
+    )
+    args = parser.parse_args()
+
+    if os.path.isdir(args.input_path):
+        df = aggregate_directory(args.input_path)
+        # Create the summary string
+        columns = [
+            'directory',
+            'resolve_rate',
+            'empty_patch_rate',
+            'unfinished_rate',
+            'error_rate',
+            'avg_turns',
+            'avg_cost',
+            'total_instances',
+        ]
+        summary_str = df[columns].to_string(
+            float_format=lambda x: '{:.2f}'.format(x),
+            formatters={
+                'directory': lambda x: x[:90]
+            },  # Truncate directory names to 20 chars
+            index=False,
+        )
+
+        # Print to console
+        print('\nResults summary (sorted by resolve rate):')
+        print(summary_str)
+
+        # Save to text file
+        txt_output = args.output.rsplit('.', 1)[0] + '.txt'
+        with open(txt_output, 'w') as f:
+            f.write('Results summary (sorted by resolve rate):\n')
+            f.write(summary_str)
+
+        # Save
+        df.to_json(args.output, lines=True, orient='records')
+        df[columns].to_csv(args.output.rsplit('.', 1)[0] + '.csv', index=False)
+    else:
+        # Process single file with detailed output
+        results = []
+        try:
+            result = process_file(args.input_path)
+            results.append(result)
+
+            # Print detailed results for single file
+            print(f'\nResults for {args.input_path}:')
+            print(
+                f"Number of resolved: {result['resolved']['count']} / {result['total_instances']} ({result['resolved']['percentage']:.2f}% [{result['resolved']['ci'][0]:.2f}%, {result['resolved']['ci'][1]:.2f}%])"
+            )
+            print(
+                f"Number of empty patch: {result['empty_patches']['count']} / {result['total_instances']} ({result['empty_patches']['percentage']:.2f}%)"
+            )
+            print(
+                f"Number of error lines: {result['errors']['total']} / {result['total_instances']} ({result['errors']['percentage']:.2f}%)"
+            )
+            print(
+                f"Number of agent stuck in loop: {result['errors']['stuck_in_loop']['count']} / {result['total_instances']} ({result['errors']['stuck_in_loop']['percentage']:.2f}%)"
+            )
+            print(
+                f"Number of unfinished runs: {result['unfinished_runs']['count']} / {result['total_instances']} ({result['unfinished_runs']['percentage']:.2f}%)"
+            )
+            print(f"Total cost: {result['costs']['total']:.2f} USD")
+            print('## Statistics')
+            print(
+                f"Avg. num of turns per instance: {result['statistics']['avg_turns']:.2f}"
+            )
+            print(
+                f"Avg. agent cost per instance: {result['statistics']['costs']['main_agent']:.2f} USD"
+            )
+            print(
+                f"Avg. editor cost per instance: {result['statistics']['costs']['editor']:.2f} USD"
+            )
+            print(
+                f"Avg. total cost per instance: {result['statistics']['costs']['total']:.2f} USD"
+            )
+
+            print('## Detailed error breakdown:')
+            for error, data in result['errors']['breakdown'].items():
+                print(f"{error}: {data['count']} ({data['percentage']:.2f}%)")
+
+        except Exception as e:
+            print(f'Error processing {args.input_path}: {str(e)}')
--- a/Show More
+++ b/Show More
				`@@ -1 +0,0 @@`
				`The files in this directory configure a development container for GitHub Codespaces.`