Fix issue #5219 : Feature: PR Review

2026-04-29 03:00:45 -04:00 · 2024-11-23 04:54:42 +00:00
569 changed files with 28574 additions and 12956 deletions
--- a/.github/dependabot.yml
+++ b/.github/dependabot.yml
@@ -16,9 +16,6 @@ updates:
      chromadb:
        patterns:
          - "chromadb"
-      browsergym:
-        patterns:
-          - "browsergym*"
      security-all:
        applies-to: "security-updates"
        patterns:
--- a/.github/workflows/eval-runner.yml
+++ b/.github/workflows/eval-runner.yml
@@ -1,8 +1,10 @@
-name: Run SWE-Bench Evaluation
+name: Run Evaluation

 on:
  pull_request:
    types: [labeled]
+  schedule:
+    - cron: "0 1 * * *" # Run daily at 1 AM UTC
  workflow_dispatch:
    inputs:
      reason:
@@ -58,6 +60,24 @@ jobs:
          echo "api_key = \"$DEEPSEEK_API_KEY\"" >> config.toml
          echo "temperature = 0.0" >> config.toml

+      - name: Run integration test evaluation
+        env:
+          ALLHANDS_API_KEY: ${{ secrets.ALLHANDS_EVAL_RUNTIME_API_KEY }}
+          RUNTIME: remote
+          SANDBOX_REMOTE_RUNTIME_API_URL: https://runtime.eval.all-hands.dev
+          EVAL_DOCKER_IMAGE_PREFIX: us-central1-docker.pkg.dev/evaluation-092424/swe-bench-images
+
+        run: |
+          poetry run ./evaluation/integration_tests/scripts/run_infer.sh llm.eval HEAD CodeActAgent '' $N_PROCESSES
+
+          # get evaluation report
+          REPORT_FILE=$(find evaluation/evaluation_outputs/outputs/integration_tests/CodeActAgent/deepseek-chat_maxiter_10_N* -name "report.md" -type f | head -n 1)
+          echo "REPORT_FILE: $REPORT_FILE"
+          echo "INTEGRATION_TEST_REPORT<<EOF" >> $GITHUB_ENV
+          cat $REPORT_FILE >> $GITHUB_ENV
+          echo >> $GITHUB_ENV
+          echo "EOF" >> $GITHUB_ENV
+
      - name: Run SWE-Bench evaluation
        env:
          ALLHANDS_API_KEY: ${{ secrets.ALLHANDS_EVAL_RUNTIME_API_KEY }}
@@ -66,12 +86,12 @@ jobs:
          EVAL_DOCKER_IMAGE_PREFIX: us-central1-docker.pkg.dev/evaluation-092424/swe-bench-images

        run: |
-          poetry run ./evaluation/benchmarks/swe_bench/scripts/run_infer.sh llm.eval HEAD CodeActAgent 300 30 $N_PROCESSES "princeton-nlp/SWE-bench_Lite" test
+          poetry run ./evaluation/swe_bench/scripts/run_infer.sh llm.eval HEAD CodeActAgent 300 30 $N_PROCESSES "princeton-nlp/SWE-bench_Lite" test
          OUTPUT_FOLDER=$(find evaluation/evaluation_outputs/outputs/princeton-nlp__SWE-bench_Lite-test/CodeActAgent -name "deepseek-chat_maxiter_50_N_*-no-hint-run_1" -type d | head -n 1)
          echo "OUTPUT_FOLDER for SWE-bench evaluation: $OUTPUT_FOLDER"
-          poetry run ./evaluation/benchmarks/swe_bench/scripts/eval_infer_remote.sh $OUTPUT_FOLDER/output.jsonl $N_PROCESSES "princeton-nlp/SWE-bench_Lite" test
+          poetry run ./evaluation/swe_bench/scripts/eval_infer_remote.sh $OUTPUT_FOLDER/output.jsonl $N_PROCESSES "princeton-nlp/SWE-bench_Lite" test

-          poetry run ./evaluation/benchmarks/swe_bench/scripts/eval/summarize_outputs.py $OUTPUT_FOLDER/output.jsonl > summarize_outputs.log 2>&1
+          poetry run ./evaluation/swe_bench/scripts/eval/summarize_outputs.py $OUTPUT_FOLDER/output.jsonl > summarize_outputs.log 2>&1
          echo "SWEBENCH_REPORT<<EOF" >> $GITHUB_ENV
          cat summarize_outputs.log >> $GITHUB_ENV
          echo "EOF" >> $GITHUB_ENV
@@ -125,6 +145,9 @@ jobs:
              **SWE-Bench Evaluation Report**
              ${{ env.SWEBENCH_REPORT }}
              ---
+              **Integration Tests Evaluation Report**
+              ${{ env.INTEGRATION_TEST_REPORT }}
+              ---
              You can download the full evaluation outputs [here](${{ env.ARTIFACT_URL }}).

      - name: Post to a Slack channel
--- a/.github/workflows/fe-unit-tests.yml
+++ b/.github/workflows/fe-unit-tests.yml
@@ -24,8 +24,7 @@ jobs:
    runs-on: ubuntu-latest
    strategy:
      matrix:
-        node-version: [20, 22]
-      fail-fast: true
+        node-version: [20]
    steps:
      - name: Checkout
        uses: actions/checkout@v4
@@ -36,9 +35,6 @@ jobs:
      - name: Install dependencies
        working-directory: ./frontend
        run: npm ci
-      - name: Run TypeScript compilation
-        working-directory: ./frontend
-        run: npm run make-i18n && tsc
      - name: Run tests and collect coverage
        working-directory: ./frontend
        run: npm run test:coverage
--- a/.github/workflows/ghcr-build.yml
+++ b/.github/workflows/ghcr-build.yml
@@ -291,7 +291,7 @@ jobs:
          SANDBOX_RUNTIME_CONTAINER_IMAGE=$image_name \
          TEST_IN_CI=true \
          RUN_AS_OPENHANDS=false \
-          poetry run pytest -n 3 -raRs --reruns 2 --reruns-delay 5 --cov=openhands --cov-report=xml -s ./tests/runtime --ignore=tests/runtime/test_browsergym_envs.py
+          poetry run pytest -n 3 -raRs --reruns 2 --reruns-delay 5 --cov=openhands --cov-report=xml -s ./tests/runtime
      - name: Upload coverage to Codecov
        uses: codecov/codecov-action@v4
        env:
@@ -368,7 +368,7 @@ jobs:
          SANDBOX_RUNTIME_CONTAINER_IMAGE=$image_name \
          TEST_IN_CI=true \
          RUN_AS_OPENHANDS=true \
-          poetry run pytest -n 3 -raRs --reruns 2 --reruns-delay 5 --cov=openhands --cov-report=xml -s ./tests/runtime --ignore=tests/runtime/test_browsergym_envs.py
+          poetry run pytest -n 3 -raRs --reruns 2 --reruns-delay 5 --cov=openhands --cov-report=xml -s ./tests/runtime
      - name: Upload coverage to Codecov
        uses: codecov/codecov-action@v4
        env:
--- a/.github/workflows/integration-runner.yml
+++ b/.github/workflows/integration-runner.yml
@@ -1,158 +0,0 @@
-name: Run Integration Tests
-
-on:
-  pull_request:
-    types: [labeled]
-  workflow_dispatch:
-    inputs:
-      reason:
-        description: 'Reason for manual trigger'
-        required: true
-        default: ''
-  schedule:
-    - cron: '30 22 * * *'  # Runs at 10:30pm UTC every day
-
-env:
-  N_PROCESSES: 10 # Global configuration for number of parallel processes for evaluation
-
-jobs:
-  run-integration-tests:
-    if: github.event.label.name == 'integration-test' || github.event_name == 'workflow_dispatch' || github.event_name == 'schedule'
-    runs-on: ubuntu-latest
-    permissions:
-      contents: "read"
-      id-token: "write"
-      pull-requests: "write"
-      issues: "write"
-    strategy:
-      matrix:
-        python-version: ["3.12"]
-    steps:
-      - name: Checkout repository
-        uses: actions/checkout@v4
-
-      - name: Install poetry via pipx
-        run: pipx install poetry
-
-      - name: Set up Python
-        uses: actions/setup-python@v5
-        with:
-          python-version: ${{ matrix.python-version }}
-          cache: "poetry"
-
-      - name: Comment on PR if 'integration-test' label is present
-        if: github.event_name == 'pull_request' && github.event.label.name == 'integration-test'
-        uses: KeisukeYamashita/create-comment@v1
-        with:
-          unique: false
-          comment: |
-            Hi! I started running the integration tests on your PR. You will receive a comment with the results shortly.
-
-      - name: Install Python dependencies using Poetry
-        run: poetry install --without evaluation,llama-index
-
-      - name: Configure config.toml for testing with Haiku
-        env:
-          LLM_MODEL: "litellm_proxy/claude-3-5-haiku-20241022"
-          LLM_API_KEY: ${{ secrets.LLM_API_KEY }}
-          LLM_BASE_URL: ${{ secrets.LLM_BASE_URL }}
-        run: |
-          echo "[llm.eval]" > config.toml
-          echo "model = \"$LLM_MODEL\"" >> config.toml
-          echo "api_key = \"$LLM_API_KEY\"" >> config.toml
-          echo "base_url = \"$LLM_BASE_URL\"" >> config.toml
-          echo "temperature = 0.0" >> config.toml
-
-      - name: Build environment
-        run: make build
-
-      - name: Run integration test evaluation for Haiku
-        env:
-          SANDBOX_FORCE_REBUILD_RUNTIME: True
-        run: |
-          poetry run ./evaluation/integration_tests/scripts/run_infer.sh llm.eval HEAD CodeActAgent '' $N_PROCESSES '' 'haiku_run'
-
-          # get integration tests report
-          REPORT_FILE_HAIKU=$(find evaluation/evaluation_outputs/outputs/integration_tests/CodeActAgent/*haiku*_maxiter_10_N* -name "report.md" -type f | head -n 1)
-          echo "REPORT_FILE: $REPORT_FILE_HAIKU"
-          echo "INTEGRATION_TEST_REPORT_HAIKU<<EOF" >> $GITHUB_ENV
-          cat $REPORT_FILE_HAIKU >> $GITHUB_ENV
-          echo >> $GITHUB_ENV
-          echo "EOF" >> $GITHUB_ENV
-
-      - name: Wait a little bit
-        run: sleep 10
-
-      - name: Configure config.toml for testing with DeepSeek
-        env:
-          LLM_MODEL: "litellm_proxy/deepseek-chat"
-          LLM_API_KEY: ${{ secrets.LLM_API_KEY }}
-          LLM_BASE_URL: ${{ secrets.LLM_BASE_URL }}
-        run: |
-          echo "[llm.eval]" > config.toml
-          echo "model = \"$LLM_MODEL\"" >> config.toml
-          echo "api_key = \"$LLM_API_KEY\"" >> config.toml
-          echo "base_url = \"$LLM_BASE_URL\"" >> config.toml
-          echo "temperature = 0.0" >> config.toml
-
-      - name: Run integration test evaluation for DeepSeek
-        env:
-          SANDBOX_FORCE_REBUILD_RUNTIME: True
-        run: |
-          poetry run ./evaluation/integration_tests/scripts/run_infer.sh llm.eval HEAD CodeActAgent '' $N_PROCESSES '' 'deepseek_run'
-
-          # get integration tests report
-          REPORT_FILE_DEEPSEEK=$(find evaluation/evaluation_outputs/outputs/integration_tests/CodeActAgent/deepseek*_maxiter_10_N* -name "report.md" -type f | head -n 1)
-          echo "REPORT_FILE: $REPORT_FILE_DEEPSEEK"
-          echo "INTEGRATION_TEST_REPORT_DEEPSEEK<<EOF" >> $GITHUB_ENV
-          cat $REPORT_FILE_DEEPSEEK >> $GITHUB_ENV
-          echo >> $GITHUB_ENV
-          echo "EOF" >> $GITHUB_ENV
-
-      - name: Create archive of evaluation outputs
-        run: |
-          TIMESTAMP=$(date +'%y-%m-%d-%H-%M')
-          cd evaluation/evaluation_outputs/outputs  # Change to the outputs directory
-          tar -czvf ../../../integration_tests_${TIMESTAMP}.tar.gz integration_tests/CodeActAgent/*  # Only include the actual result directories
-
-      - name: Upload evaluation results as artifact
-        uses: actions/upload-artifact@v4
-        id: upload_results_artifact
-        with:
-          name: integration-test-outputs-${{ github.run_id }}-${{ github.run_attempt }}
-          path: integration_tests_*.tar.gz
-
-      - name: Get artifact URLs
-        run: |
-          echo "ARTIFACT_URL=${{ steps.upload_results_artifact.outputs.artifact-url }}" >> $GITHUB_ENV
-
-      - name: Set timestamp and trigger reason
-        run: |
-          echo "TIMESTAMP=$(date +'%Y-%m-%d-%H-%M')" >> $GITHUB_ENV
-          if [[ "${{ github.event_name }}" == "pull_request" ]]; then
-            echo "TRIGGER_REASON=pr-${{ github.event.pull_request.number }}" >> $GITHUB_ENV
-          elif [[ "${{ github.event_name }}" == "workflow_dispatch" ]]; then
-            echo "TRIGGER_REASON=manual-${{ github.event.inputs.reason }}" >> $GITHUB_ENV
-          else
-            echo "TRIGGER_REASON=nightly-scheduled" >> $GITHUB_ENV
-          fi
-
-      - name: Comment with results and artifact link
-        id: create_comment
-        uses: KeisukeYamashita/create-comment@v1
-        with:
-          # if triggered by PR, use PR number, otherwise use 5318 as fallback issue number for manual triggers
-          number: ${{ github.event_name == 'pull_request' && github.event.pull_request.number || 5318 }}
-          unique: false
-          comment: |
-              Trigger by: ${{ github.event_name == 'pull_request' && format('Pull Request (integration-test label on PR #{0})', github.event.pull_request.number) || (github.event_name == 'workflow_dispatch' && format('Manual Trigger: {0}', github.event.inputs.reason)) || 'Nightly Scheduled Run' }}
-              Commit: ${{ github.sha }}
-              **Integration Tests Report (Haiku)**
-              Haiku LLM Test Results:
-              ${{ env.INTEGRATION_TEST_REPORT_HAIKU }}
-              ---
-              **Integration Tests Report (DeepSeek)**
-              DeepSeek LLM Test Results:
-              ${{ env.INTEGRATION_TEST_REPORT_DEEPSEEK }}
-              ---
-              Download testing outputs (includes both Haiku and DeepSeek results): [Download](${{ steps.upload_results_artifact.outputs.artifact-url }})
--- a/.github/workflows/lint-fix.yml
+++ b/.github/workflows/lint-fix.yml
@@ -5,10 +5,9 @@ on:
    types: [labeled]

 jobs:
-  # Frontend lint fixes
-  lint-fix-frontend:
+  lint-fix:
    if: github.event.label.name == 'lint-fix'
-    name: Fix frontend linting issues
+    name: Fix linting issues
    runs-on: ubuntu-latest
    permissions:
      contents: write
@@ -21,6 +20,7 @@ jobs:
          fetch-depth: 0
          token: ${{ secrets.GITHUB_TOKEN }}

+      # Frontend lint fixes
      - name: Install Node.js 20
        uses: actions/setup-node@v4
        with:
@@ -34,36 +34,7 @@ jobs:
          cd frontend
          npm run lint:fix

-      # Commit and push changes if any
-      - name: Check for changes
-        id: git-check
-        run: |
-          git diff --quiet || echo "changes=true" >> $GITHUB_OUTPUT
-      - name: Commit and push if there are changes
-        if: steps.git-check.outputs.changes == 'true'
-        run: |
-          git config --local user.email "openhands@all-hands.dev"
-          git config --local user.name "OpenHands Bot"
-          git add -A
-          git commit -m "🤖 Auto-fix frontend linting issues"
-          git push
-
-  # Python lint fixes
-  lint-fix-python:
-    if: github.event.label.name == 'lint-fix'
-    name: Fix Python linting issues
-    runs-on: ubuntu-latest
-    permissions:
-      contents: write
-      pull-requests: write
-    steps:
-      - uses: actions/checkout@v4
-        with:
-          ref: ${{ github.head_ref }}
-          repository: ${{ github.event.pull_request.head.repo.full_name }}
-          fetch-depth: 0
-          token: ${{ secrets.GITHUB_TOKEN }}
-
+      # Python lint fixes
      - name: Set up python
        uses: actions/setup-python@v5
        with:
@@ -87,5 +58,5 @@ jobs:
          git config --local user.email "openhands@all-hands.dev"
          git config --local user.name "OpenHands Bot"
          git add -A
-          git commit -m "🤖 Auto-fix Python linting issues"
+          git commit -m "🤖 Auto-fix linting issues"
          git push
--- a/.github/workflows/lint.yml
+++ b/.github/workflows/lint.yml
@@ -30,11 +30,10 @@ jobs:
        run: |
          cd frontend
          npm install --frozen-lockfile
-      - name: Lint and TypeScript compilation
+      - name: Lint
        run: |
          cd frontend
          npm run lint
-          npm run make-i18n && tsc

  # Run lint on the python code
  lint-python:
--- a/.github/workflows/openhands-resolver.yml
+++ b/.github/workflows/openhands-resolver.yml
@@ -16,26 +16,17 @@ on:
        type: string
        default: "main"
        description: "Target branch to pull and create PR against"
-      LLM_MODEL:
-        required: false
-        type: string
-        default: "anthropic/claude-3-5-sonnet-20241022"
-      base_container_image:
-        required: false
-        type: string
-        default: ""
-        description: "Custom sandbox env"
    secrets:
      LLM_MODEL:
-        required: false
+        required: true
      LLM_API_KEY:
        required: true
      LLM_BASE_URL:
        required: false
      PAT_TOKEN:
-        required: false
+        required: true
      PAT_USERNAME:
-        required: false
+        required: true

  issues:
    types: [labeled]
@@ -110,14 +101,13 @@ jobs:

      - name: Check required environment variables
        env:
-          LLM_MODEL: ${{ secrets.LLM_MODEL || inputs.LLM_MODEL }}
+          LLM_MODEL: ${{ secrets.LLM_MODEL }}
          LLM_API_KEY: ${{ secrets.LLM_API_KEY }}
          LLM_BASE_URL: ${{ secrets.LLM_BASE_URL }}
          PAT_TOKEN: ${{ secrets.PAT_TOKEN }}
          PAT_USERNAME: ${{ secrets.PAT_USERNAME }}
-          GITHUB_TOKEN: ${{ github.token }}
        run: |
-          required_vars=("LLM_MODEL" "LLM_API_KEY")
+          required_vars=("LLM_MODEL" "LLM_API_KEY" "PAT_TOKEN" "PAT_USERNAME")
          for var in "${required_vars[@]}"; do
            if [ -z "${!var}" ]; then
              echo "Error: Required environment variable $var is not set."
@@ -125,19 +115,6 @@ jobs:
            fi
          done

-          # Check optional variables and warn about fallbacks
-          if [ -z "$PAT_TOKEN" ]; then
-            echo "Warning: PAT_TOKEN is not set, falling back to GITHUB_TOKEN"
-          fi
-
-          if [ -z "$LLM_BASE_URL" ]; then
-            echo "Warning: LLM_BASE_URL is not set, will use default API endpoint"
-          fi
-
-          if [ -z "$PAT_USERNAME" ]; then
-            echo "Warning: PAT_USERNAME is not set, will use openhands-agent"
-          fi
-
      - name: Set environment variables
        run: |
          if [ -n "${{ github.event.review.body }}" ]; then
@@ -161,16 +138,13 @@ jobs:
          fi

          echo "MAX_ITERATIONS=${{ inputs.max_iterations || 50 }}" >> $GITHUB_ENV
-          echo "SANDBOX_ENV_GITHUB_TOKEN=${{ secrets.PAT_TOKEN || github.token }}" >> $GITHUB_ENV
-          echo "SANDBOX_ENV_BASE_CONTAINER_IMAGE=${{ inputs.base_container_image }}" >> $GITHUB_ENV
-
-          # Set branch variables
+          echo "SANDBOX_ENV_GITHUB_TOKEN=${{ secrets.GITHUB_TOKEN }}" >> $GITHUB_ENV
          echo "TARGET_BRANCH=${{ inputs.target_branch }}" >> $GITHUB_ENV

      - name: Comment on issue with start message
        uses: actions/github-script@v7
        with:
-          github-token: ${{ secrets.PAT_TOKEN || github.token }}
+          github-token: ${{secrets.GITHUB_TOKEN}}
          script: |
            const issueType = process.env.ISSUE_TYPE;
            github.rest.issues.createComment({
@@ -195,9 +169,9 @@ jobs:

      - name: Attempt to resolve issue
        env:
-          GITHUB_TOKEN: ${{ secrets.PAT_TOKEN || github.token }}
-          GITHUB_USERNAME: ${{ secrets.PAT_USERNAME || 'openhands-agent' }}
-          LLM_MODEL: ${{ secrets.LLM_MODEL || inputs.LLM_MODEL }}
+          GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
+          GITHUB_USERNAME: ${{ secrets.PAT_USERNAME }}
+          LLM_MODEL: ${{ secrets.LLM_MODEL }}
          LLM_API_KEY: ${{ secrets.LLM_API_KEY }}
          LLM_BASE_URL: ${{ secrets.LLM_BASE_URL }}
          PYTHONPATH: ""
@@ -229,9 +203,9 @@ jobs:
      - name: Create draft PR or push branch
        if: always() # Create PR or branch even if the previous steps fail
        env:
-          GITHUB_TOKEN: ${{ secrets.PAT_TOKEN || github.token }}
-          GITHUB_USERNAME: ${{ secrets.PAT_USERNAME || 'openhands-agent' }}
-          LLM_MODEL: ${{ secrets.LLM_MODEL || inputs.LLM_MODEL }}
+          GITHUB_TOKEN: ${{ secrets.PAT_TOKEN }}
+          GITHUB_USERNAME: ${{ secrets.PAT_USERNAME }}
+          LLM_MODEL: ${{ secrets.LLM_MODEL }}
          LLM_API_KEY: ${{ secrets.LLM_API_KEY }}
          LLM_BASE_URL: ${{ secrets.LLM_BASE_URL }}
          PYTHONPATH: ""
@@ -253,7 +227,314 @@ jobs:
        uses: actions/github-script@v7
        if: always() # Comment on issue even if the previous steps fail
        with:
-          github-token: ${{ secrets.PAT_TOKEN || github.token }}
+          github-token: ${{secrets.GITHUB_TOKEN}}
+          script: |
+            const fs = require('fs');
+            const issueNumber = ${{ env.ISSUE_NUMBER }};
+            const success = ${{ steps.check_result.outputs.RESOLUTION_SUCCESS }};
+
+            let prNumber = '';
+            let branchName = '';
+            let logContent = '';
+            const noChangesMessage = `No changes to commit for issue #${issueNumber}. Skipping commit.`;
+
+            try {
+              if (success){
+                logContent = fs.readFileSync('/tmp/pr_result.txt', 'utf8').trim();
+              } else {
+                logContent = fs.readFileSync('/tmp/branch_result.txt', 'utf8').trim();
+              }
+            } catch (error) {
+              console.error('Error reading results file:', error);
+            }
+
+            try {
+              if (success) {
+                prNumber = fs.readFileSync('/tmp/pr_number.txt', 'utf8').trim();
+              } else {
+                branchName = fs.readFileSync('/tmp/branch_name.txt', 'utf8').trim();
+              }
+            } catch (error) {
+              console.error('Error reading file:', error);
+            }
+
+            if (logContent.includes(noChangesMessage)) {
+              github.rest.issues.createComment({
+                issue_number: issueNumber,
+                owner: context.repo.owner,
+                repo: context.repo.repo,
+                body: `The workflow to fix this issue encountered an error. Openhands failed to create any code changes.`
+              });
+            } else if (success && prNumber) {
+              github.rest.issues.createComment({
+                issue_number: issueNumber,
+                owner: context.repo.owner,
+                repo: context.repo.repo,
+                body: `A potential fix has been generated and a draft PR #${prNumber} has been created. Please review the changes.`
+              });
+            } else if (!success && branchName) {
+              github.rest.issues.createComment({
+                issue_number: issueNumber,
+                owner: context.repo.owner,
+                repo: context.repo.repo,
+                body: `An attempt was made to automatically fix this issue, but it was unsuccessful. A branch named '${branchName}' has been created with the attempted changes. You can view the branch [here](https://github.com/${context.repo.owner}/${context.repo.repo}/tree/${branchName}). Manual intervention may be required.`
+              });
+            } else {
+              github.rest.issues.createComment({
+                issue_number: issueNumber,
+                owner: context.repo.owner,
+                repo: context.repo.repo,
+                body: `The workflow to fix this issue encountered an error. Please check the [workflow logs](https://github.com/${context.repo.owner}/${context.repo.repo}/actions/runs/${context.runId}) for more information.`
+              });
+            }
+
+  review-pr:
+    if: github.event.label.name == 'review-pr'
+    runs-on: ubuntu-latest
+    steps:
+      - name: Checkout repository
+        uses: actions/checkout@v4
+
+      - name: Set up Python
+        uses: actions/setup-python@v5
+        with:
+          python-version: "3.12"
+
+      - name: Get latest versions and create requirements.txt
+        run: |
+          python -m pip index versions openhands-ai > openhands_versions.txt
+          OPENHANDS_VERSION=$(head -n 1 openhands_versions.txt | awk '{print $2}' | tr -d '()')
+          echo "openhands-ai==${OPENHANDS_VERSION}" >> requirements.txt
+          cat requirements.txt
+
+      - name: Cache pip dependencies
+        uses: actions/cache@v3
+        with:
+          path: ${{ env.pythonLocation }}/lib/python3.12/site-packages/*
+          key: ${{ runner.os }}-pip-openhands-resolver-${{ hashFiles('requirements.txt') }}
+          restore-keys: |
+            ${{ runner.os }}-pip-openhands-resolver-${{ hashFiles('requirements.txt') }}
+
+      - name: Check required environment variables
+        env:
+          LLM_MODEL: ${{ secrets.LLM_MODEL }}
+          LLM_API_KEY: ${{ secrets.LLM_API_KEY }}
+          LLM_BASE_URL: ${{ secrets.LLM_BASE_URL }}
+          PAT_TOKEN: ${{ secrets.PAT_TOKEN }}
+          PAT_USERNAME: ${{ secrets.PAT_USERNAME }}
+        run: |
+          required_vars=("LLM_MODEL" "LLM_API_KEY" "PAT_TOKEN" "PAT_USERNAME")
+          for var in "${required_vars[@]}"; do
+            if [ -z "${!var}" ]; then
+              echo "Error: Required environment variable $var is not set."
+              exit 1
+            fi
+          done
+
+      - name: Set environment variables
+        run: |
+          echo "ISSUE_NUMBER=${{ github.event.pull_request.number }}" >> $GITHUB_ENV
+          echo "ISSUE_TYPE=pr" >> $GITHUB_ENV
+          echo "MAX_ITERATIONS=${{ inputs.max_iterations || 50 }}" >> $GITHUB_ENV
+          echo "SANDBOX_ENV_GITHUB_TOKEN=${{ secrets.GITHUB_TOKEN }}" >> $GITHUB_ENV
+          echo "TARGET_BRANCH=${{ inputs.target_branch }}" >> $GITHUB_ENV
+
+      - name: Comment on PR with start message
+        uses: actions/github-script@v7
+        with:
+          github-token: ${{secrets.GITHUB_TOKEN}}
+          script: |
+            github.rest.issues.createComment({
+              issue_number: ${{ env.ISSUE_NUMBER }},
+              owner: context.repo.owner,
+              repo: context.repo.repo,
+              body: `[OpenHands](https://github.com/All-Hands-AI/OpenHands) started reviewing the PR! You can monitor the progress [here](https://github.com/${context.repo.owner}/${context.repo.repo}/actions/runs/${context.runId}).`
+            });
+
+      - name: Install OpenHands
+        run: |
+          python -m pip install --upgrade -r requirements.txt
+
+      - name: Review PR
+        env:
+          GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
+          GITHUB_USERNAME: ${{ secrets.PAT_USERNAME }}
+          LLM_MODEL: ${{ secrets.LLM_MODEL }}
+          LLM_API_KEY: ${{ secrets.LLM_API_KEY }}
+          LLM_BASE_URL: ${{ secrets.LLM_BASE_URL }}
+          PYTHONPATH: ""
+        run: |
+          cd /tmp && python -m openhands.resolver.resolve_issue \
+            --repo ${{ github.repository }} \
+            --issue-number ${{ env.ISSUE_NUMBER }} \
+            --issue-type ${{ env.ISSUE_TYPE }} \
+            --max-iterations ${{ env.MAX_ITERATIONS }} \
+            --prompt-template pr-review \
+            --comment-id ${{ env.COMMENT_ID }}
+
+      - name: Set up Python
+        uses: actions/setup-python@v5
+        with:
+          python-version: "3.12"
+
+      - name: Get latest versions and create requirements.txt
+        run: |
+          python -m pip index versions openhands-ai > openhands_versions.txt
+          OPENHANDS_VERSION=$(head -n 1 openhands_versions.txt | awk '{print $2}' | tr -d '()')
+          echo "openhands-ai==${OPENHANDS_VERSION}" >> requirements.txt
+          cat requirements.txt
+
+      - name: Cache pip dependencies
+        if: |
+          !(
+            github.event.label.name == 'fix-me-experimental' ||
+            (
+              (github.event_name == 'issue_comment' || github.event_name == 'pull_request_review_comment') &&
+              contains(github.event.comment.body, '@openhands-agent-exp')
+            ) ||
+            (
+              github.event_name == 'pull_request_review' &&
+              contains(github.event.review.body, '@openhands-agent-exp')
+            )
+          )
+        uses: actions/cache@v3
+        with:
+          path: ${{ env.pythonLocation }}/lib/python3.12/site-packages/*
+          key: ${{ runner.os }}-pip-openhands-resolver-${{ hashFiles('requirements.txt') }}
+          restore-keys: |
+            ${{ runner.os }}-pip-openhands-resolver-${{ hashFiles('requirements.txt') }}
+
+      - name: Check required environment variables
+        env:
+          LLM_MODEL: ${{ secrets.LLM_MODEL }}
+          LLM_API_KEY: ${{ secrets.LLM_API_KEY }}
+          LLM_BASE_URL: ${{ secrets.LLM_BASE_URL }}
+          PAT_TOKEN: ${{ secrets.PAT_TOKEN }}
+          PAT_USERNAME: ${{ secrets.PAT_USERNAME }}
+        run: |
+          required_vars=("LLM_MODEL" "LLM_API_KEY" "PAT_TOKEN" "PAT_USERNAME")
+          for var in "${required_vars[@]}"; do
+            if [ -z "${!var}" ]; then
+              echo "Error: Required environment variable $var is not set."
+              exit 1
+            fi
+          done
+
+      - name: Set environment variables
+        run: |
+          if [ -n "${{ github.event.review.body }}" ]; then
+            echo "ISSUE_NUMBER=${{ github.event.pull_request.number }}" >> $GITHUB_ENV
+            echo "ISSUE_TYPE=pr" >> $GITHUB_ENV
+          elif [ -n "${{ github.event.issue.pull_request }}" ]; then
+            echo "ISSUE_NUMBER=${{ github.event.issue.number }}" >> $GITHUB_ENV
+            echo "ISSUE_TYPE=pr" >> $GITHUB_ENV
+          elif [ -n "${{ github.event.pull_request.number }}" ]; then
+            echo "ISSUE_NUMBER=${{ github.event.pull_request.number }}" >> $GITHUB_ENV
+            echo "ISSUE_TYPE=pr" >> $GITHUB_ENV
+          else
+            echo "ISSUE_NUMBER=${{ github.event.issue.number }}" >> $GITHUB_ENV
+            echo "ISSUE_TYPE=issue" >> $GITHUB_ENV
+          fi
+
+          if [ -n "${{ github.event.review.body }}" ]; then
+            echo "COMMENT_ID=${{ github.event.review.id || 'None' }}" >> $GITHUB_ENV
+          else
+            echo "COMMENT_ID=${{ github.event.comment.id || 'None' }}" >> $GITHUB_ENV
+          fi
+
+          echo "MAX_ITERATIONS=${{ inputs.max_iterations || 50 }}" >> $GITHUB_ENV
+          echo "SANDBOX_ENV_GITHUB_TOKEN=${{ secrets.GITHUB_TOKEN }}" >> $GITHUB_ENV
+
+          # Set branch variables
+          echo "TARGET_BRANCH=${{ inputs.target_branch }}" >> $GITHUB_ENV
+
+      - name: Comment on issue with start message
+        uses: actions/github-script@v7
+        with:
+          github-token: ${{secrets.GITHUB_TOKEN}}
+          script: |
+            const issueType = process.env.ISSUE_TYPE;
+            github.rest.issues.createComment({
+              issue_number: ${{ env.ISSUE_NUMBER }},
+              owner: context.repo.owner,
+              repo: context.repo.repo,
+              body: `[OpenHands](https://github.com/All-Hands-AI/OpenHands) started fixing the ${issueType}! You can monitor the progress [here](https://github.com/${context.repo.owner}/${context.repo.repo}/actions/runs/${context.runId}).`
+            });
+
+      - name: Install OpenHands
+        run: |
+          if [[ "${{ github.event.label.name }}" == "fix-me-experimental" ]] ||
+             ([[ "${{ github.event_name }}" == "issue_comment" || "${{ github.event_name }}" == "pull_request_review_comment" ]] &&
+              [[ "${{ github.event.comment.body }}" == "@openhands-agent-exp"* ]]) ||
+             ([[ "${{ github.event_name }}" == "pull_request_review" ]] &&
+              [[ "${{ github.event.review.body }}" == "@openhands-agent-exp"* ]]); then
+            python -m pip install --upgrade pip
+            pip install git+https://github.com/all-hands-ai/openhands.git
+          else
+            python -m pip install --upgrade -r requirements.txt
+          fi
+
+      - name: Attempt to resolve issue
+        env:
+          GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
+          GITHUB_USERNAME: ${{ secrets.PAT_USERNAME }}
+          LLM_MODEL: ${{ secrets.LLM_MODEL }}
+          LLM_API_KEY: ${{ secrets.LLM_API_KEY }}
+          LLM_BASE_URL: ${{ secrets.LLM_BASE_URL }}
+          PYTHONPATH: ""
+        run: |
+          cd /tmp && python -m openhands.resolver.resolve_issue \
+            --repo ${{ github.repository }} \
+            --issue-number ${{ env.ISSUE_NUMBER }} \
+            --issue-type ${{ env.ISSUE_TYPE }} \
+            --max-iterations ${{ env.MAX_ITERATIONS }} \
+            --comment-id ${{ env.COMMENT_ID }} \
+
+      - name: Check resolution result
+        id: check_result
+        run: |
+          if cd /tmp && grep -q '"success":true' output/output.jsonl; then
+            echo "RESOLUTION_SUCCESS=true" >> $GITHUB_OUTPUT
+          else
+            echo "RESOLUTION_SUCCESS=false" >> $GITHUB_OUTPUT
+          fi
+
+      - name: Upload output.jsonl as artifact
+        uses: actions/upload-artifact@v4
+        if: always() # Upload even if the previous steps fail
+        with:
+          name: resolver-output
+          path: /tmp/output/output.jsonl
+          retention-days: 30 # Keep the artifact for 30 days
+
+      - name: Create draft PR or push branch
+        if: always() # Create PR or branch even if the previous steps fail
+        env:
+          GITHUB_TOKEN: ${{ secrets.PAT_TOKEN }}
+          GITHUB_USERNAME: ${{ secrets.PAT_USERNAME }}
+          LLM_MODEL: ${{ secrets.LLM_MODEL }}
+          LLM_API_KEY: ${{ secrets.LLM_API_KEY }}
+          LLM_BASE_URL: ${{ secrets.LLM_BASE_URL }}
+          PYTHONPATH: ""
+        run: |
+          if [ "${{ steps.check_result.outputs.RESOLUTION_SUCCESS }}" == "true" ]; then
+            cd /tmp && python -m openhands.resolver.send_pull_request \
+              --issue-number ${{ env.ISSUE_NUMBER }} \
+              --pr-type draft | tee pr_result.txt && \
+              grep "draft created" pr_result.txt | sed 's/.*\///g' > pr_number.txt
+          else
+            cd /tmp && python -m openhands.resolver.send_pull_request \
+              --issue-number ${{ env.ISSUE_NUMBER }} \
+              --pr-type branch \
+              --send-on-failure | tee branch_result.txt && \
+              grep "branch created" branch_result.txt | sed 's/.*\///g; s/.expand=1//g' > branch_name.txt
+          fi
+
+      - name: Comment on issue
+        uses: actions/github-script@v7
+        if: always() # Comment on issue even if the previous steps fail
+        with:
+          github-token: ${{secrets.GITHUB_TOKEN}}
          script: |
            const fs = require('fs');
            const issueNumber = ${{ env.ISSUE_NUMBER }};
--- a/.github/workflows/review-pr.yml
+++ b/.github/workflows/review-pr.yml
@@ -0,0 +1,81 @@
+# Workflow that uses OpenHands to review a pull request. PR must be labeled 'review-this'
+name: Use OpenHands to Review Pull Request
+
+on:
+  pull_request:
+    types: [synchronize, labeled]
+
+permissions:
+  contents: write
+  pull-requests: write
+
+jobs:
+  dogfood:
+    if: contains(github.event.pull_request.labels.*.name, 'review-this')
+    runs-on: ubuntu-latest
+    steps:
+    - uses: actions/checkout@v4
+    - name: Set up Docker Buildx
+      id: buildx
+      uses: docker/setup-buildx-action@v3
+    - name: Set up Python
+      uses: actions/setup-python@v5
+      with:
+        python-version: '3.12'
+    - name: install git, github cli
+      run: |
+        sudo apt-get install -y git gh
+        git config --global --add safe.directory $PWD
+    - name: Checkout Repository
+      uses: actions/checkout@v4
+      with:
+        ref: ${{ github.event.pull_request.base.ref }} # check out the target branch
+    - name: Download Diff
+      run: |
+        curl -O "${{ github.event.pull_request.diff_url }}" -L
+    - name: Write Task File
+      run: |
+        echo "Your coworker wants to apply a pull request to this project." > task.txt
+        echo "Read and review ${{ github.event.pull_request.number }}.diff file. Create a review-${{ github.event.pull_request.number }}.txt and write your concise comments and suggestions there." >> task.txt
+        echo "Do not ask me for confirmation at any point." >> task.txt
+        echo "" >> task.txt
+        echo "Title" >> task.txt
+        echo "${{ github.event.pull_request.title }}" >> task.txt
+        echo "" >> task.txt
+        echo "Description" >> task.txt
+        echo "${{ github.event.pull_request.body }}" >> task.txt
+        echo "" >> task.txt
+        echo "Diff file is: ${{ github.event.pull_request.number }}.diff" >> task.txt
+    - name: Set up environment
+      run: |
+        curl -sSL https://install.python-poetry.org | python3 -
+        export PATH="/github/home/.local/bin:$PATH"
+        poetry install --without evaluation,llama-index
+        poetry run playwright install --with-deps chromium
+    - name: Run OpenHands
+      env:
+        LLM_API_KEY: ${{ secrets.LLM_API_KEY }}
+        LLM_MODEL: ${{ vars.LLM_MODEL }}
+      run: |
+        # Append path to launch poetry
+        export PATH="/github/home/.local/bin:$PATH"
+        # Append path to correctly import package, note: must set pwd at first
+        export PYTHONPATH=$(pwd):$PYTHONPATH
+        export WORKSPACE_MOUNT_PATH=$GITHUB_WORKSPACE
+        export WORKSPACE_BASE=$GITHUB_WORKSPACE
+        echo -e "/exit\n" | poetry run python openhands/core/main.py -i 50 -f task.txt
+        rm task.txt
+    - name: Check if review file is non-empty
+      id: check_file
+      run: |
+        ls -la
+        if [[ -s review-${{ github.event.pull_request.number }}.txt ]]; then
+          echo "non_empty=true" >> $GITHUB_OUTPUT
+        fi
+      shell: bash
+    - name: Create PR review if file is non-empty
+      env:
+        GH_TOKEN: ${{ github.token }}
+      if: steps.check_file.outputs.non_empty == 'true'
+      run: |
+        gh pr review ${{ github.event.pull_request.number }} --comment --body-file "review-${{ github.event.pull_request.number }}.txt"
--- a/CONTRIBUTING.md
+++ b/CONTRIBUTING.md
@@ -21,14 +21,14 @@ There are many ways that you can contribute:

 1. **Download and use** OpenHands, and send [issues](https://github.com/All-Hands-AI/OpenHands/issues) when you encounter something that isn't working or a feature that you'd like to see.
 2. **Send feedback** after each session by [clicking the thumbs-up thumbs-down buttons](https://docs.all-hands.dev/modules/usage/feedback), so we can see where things are working and failing, and also build an open dataset for training code agents.
-3. **Improve the Codebase** by sending [PRs](#sending-pull-requests-to-openhands) (see details below). In particular, we have some [good first issues](https://github.com/All-Hands-AI/OpenHands/labels/good%20first%20issue) that may be ones to start on.
+3. **Improve the Codebase** by sending PRs (see details below). In particular, we have some [good first issues](https://github.com/All-Hands-AI/OpenHands/labels/good%20first%20issue) that may be ones to start on.

 ## What can I build?
 Here are a few ways you can help improve the codebase.

 #### UI/UX
 We're always looking to improve the look and feel of the application. If you've got a small fix
-for something that's bugging you, feel free to open up a PR that changes the [`./frontend`](./frontend) directory.
+for something that's bugging you, feel free to open up a PR that changes the `./frontend` directory.

 If you're looking to make a bigger change, add a new UI element, or significantly alter the style
 of the application, please open an issue first, or better, join the #frontend channel in our Slack
@@ -46,7 +46,7 @@ We use the [SWE-bench](https://www.swebench.com/) benchmark to test our agent. Y
 channel in Slack to learn more.

 #### Adding a new agent
-You may want to experiment with building new types of agents. You can add an agent to [`openhands/agenthub`](./openhands/agenthub)
+You may want to experiment with building new types of agents. You can add an agent to `openhands/agenthub`
 to help expand the capabilities of OpenHands.

 #### Adding a new runtime
@@ -57,8 +57,8 @@ If you work for a company that provides a cloud-based runtime, you could help us
 by implementing the [interface specified here](https://github.com/All-Hands-AI/OpenHands/blob/main/openhands/runtime/base.py).

 #### Testing
-When you write code, it is also good to write tests. Please navigate to the [`./tests`](./tests) folder to see existing test suites.
-At the moment, we have two kinds of tests: [`unit`](./tests/unit) and [`integration`](./evaluation/integration_tests). Please refer to the README for each test suite. These tests also run on GitHub's continuous integration to ensure quality of the project.
+When you write code, it is also good to write tests. Please navigate to the `tests` folder to see existing test suites.
+At the moment, we have two kinds of tests: `unit` and `integration`. Please refer to the README for each test suite. These tests also run on GitHub's continuous integration to ensure quality of the project.

 ## Sending Pull Requests to OpenHands

@@ -103,7 +103,7 @@ Further, if you see an issue you like, please leave a "thumbs-up" or a comment,

 ### Making Pull Requests

-We're generally happy to consider all [PRs](https://github.com/All-Hands-AI/OpenHands/pulls), with the evaluation process varying based on the type of change:
+We're generally happy to consider all PRs, with the evaluation process varying based on the type of change:

 #### For Small Improvements

--- a/Development.md
+++ b/Development.md
@@ -100,7 +100,7 @@ poetry run pytest ./tests/unit/test_*.py
 To reduce build time (e.g., if no changes were made to the client-runtime component), you can use an existing Docker container image by
 setting the SANDBOX_RUNTIME_CONTAINER_IMAGE environment variable to the desired Docker image.

-Example: `export SANDBOX_RUNTIME_CONTAINER_IMAGE=ghcr.io/all-hands-ai/runtime:0.15-nikolaik`
+Example: `export SANDBOX_RUNTIME_CONTAINER_IMAGE=ghcr.io/all-hands-ai/runtime:0.14-nikolaik`

 ## Develop inside Docker container

--- a/README.md
+++ b/README.md
@@ -12,7 +12,7 @@
  <a href="https://codecov.io/github/All-Hands-AI/OpenHands?branch=main"><img alt="CodeCov" src="https://img.shields.io/codecov/c/github/All-Hands-AI/OpenHands?style=for-the-badge&color=blue"></a>
  <a href="https://github.com/All-Hands-AI/OpenHands/blob/main/LICENSE"><img src="https://img.shields.io/github/license/All-Hands-AI/OpenHands?style=for-the-badge&color=blue" alt="MIT License"></a>
  <br/>
-  <a href="https://join.slack.com/t/openhands-ai/shared_invite/zt-2vbfigwev-G03twSpXaErwzYVD4CFiBg"><img src="https://img.shields.io/badge/Slack-Join%20Us-red?logo=slack&logoColor=white&style=for-the-badge" alt="Join our Slack community"></a>
+  <a href="https://join.slack.com/t/openhands-ai/shared_invite/zt-2tom0er4l-JeNUGHt_AxpEfIBstbLPiw"><img src="https://img.shields.io/badge/Slack-Join%20Us-red?logo=slack&logoColor=white&style=for-the-badge" alt="Join our Slack community"></a>
  <a href="https://discord.gg/ESHStjSjD4"><img src="https://img.shields.io/badge/Discord-Join%20Us-purple?logo=discord&logoColor=white&style=for-the-badge" alt="Join our Discord community"></a>
  <a href="https://github.com/All-Hands-AI/OpenHands/blob/main/CREDITS.md"><img src="https://img.shields.io/badge/Project-Credits-blue?style=for-the-badge&color=FFE165&logo=github&logoColor=white" alt="Credits"></a>
  <br/>
@@ -38,16 +38,16 @@ See the [Installation](https://docs.all-hands.dev/modules/usage/installation) gu
 system requirements and more information.

 ```bash
-docker pull docker.all-hands.dev/all-hands-ai/runtime:0.15-nikolaik
+docker pull docker.all-hands.dev/all-hands-ai/runtime:0.14-nikolaik

 docker run -it --pull=always \
-    -e SANDBOX_RUNTIME_CONTAINER_IMAGE=docker.all-hands.dev/all-hands-ai/runtime:0.15-nikolaik \
+    -e SANDBOX_RUNTIME_CONTAINER_IMAGE=docker.all-hands.dev/all-hands-ai/runtime:0.14-nikolaik \
    -e LOG_ALL_EVENTS=true \
    -v /var/run/docker.sock:/var/run/docker.sock \
    -p 3000:3000 \
    --add-host host.docker.internal:host-gateway \
    --name openhands-app \
-    docker.all-hands.dev/all-hands-ai/openhands:0.15
+    docker.all-hands.dev/all-hands-ai/openhands:0.14
 ```

 You'll find OpenHands running at [http://localhost:3000](http://localhost:3000)!
@@ -82,7 +82,7 @@ troubleshooting resources, and advanced configuration options.
 OpenHands is a community-driven project, and we welcome contributions from everyone. We do most of our communication
 through Slack, so this is the best place to start, but we also are happy to have you contact us on Discord or Github:

- [Join our Slack workspace](https://join.slack.com/t/openhands-ai/shared_invite/zt-2vbfigwev-G03twSpXaErwzYVD4CFiBg) - Here we talk about research, architecture, and future development.
+- [Join our Slack workspace](https://join.slack.com/t/openhands-ai/shared_invite/zt-2tom0er4l-JeNUGHt_AxpEfIBstbLPiw) - Here we talk about research, architecture, and future development.
 - [Join our Discord server](https://discord.gg/ESHStjSjD4) - This is a community-run server for general discussion, questions, and feedback.
 - [Read or post Github Issues](https://github.com/All-Hands-AI/OpenHands/issues) - Check out the issues we're working on, or add your own ideas.

--- a/compose.yml
+++ b/compose.yml
@@ -7,7 +7,7 @@ services:
    image: openhands:latest
    container_name: openhands-app-${DATE:-}
    environment:
-      - SANDBOX_RUNTIME_CONTAINER_IMAGE=${SANDBOX_RUNTIME_CONTAINER_IMAGE:-ghcr.io/all-hands-ai/runtime:0.15-nikolaik}
+      - SANDBOX_RUNTIME_CONTAINER_IMAGE=${SANDBOX_RUNTIME_CONTAINER_IMAGE:-ghcr.io/all-hands-ai/runtime:0.14-nikolaik}
      - SANDBOX_USER_ID=${SANDBOX_USER_ID:-1234}
      - WORKSPACE_MOUNT_PATH=${WORKSPACE_BASE:-$PWD/workspace}
    ports:
--- a/config.template.toml
+++ b/config.template.toml
@@ -95,10 +95,10 @@ workspace_base = "./workspace"
 # AWS secret access key
 #aws_secret_access_key = ""

-# API key to use (For Headless / CLI only -  In Web this is overridden by Session Init)
+# API key to use
 api_key = "your-api-key"

-# API base URL (For Headless / CLI only -  In Web this is overridden by Session Init)
+# API base URL
 #base_url = ""

 # API version
@@ -131,7 +131,7 @@ embedding_model = "local"
 # Maximum number of output tokens
 #max_output_tokens = 0

-# Model to use. (For Headless / CLI only -  In Web this is overridden by Session Init)
+# Model to use
 model = "gpt-4o"

 # Number of retries to attempt when an operation fails with the LLM.
@@ -237,10 +237,10 @@ llm_config = 'gpt3'
 ##############################################################################
 [security]

-# Enable confirmation mode (For Headless / CLI only -  In Web this is overridden by Session Init)
+# Enable confirmation mode
 #confirmation_mode = false

-# The security analyzer to use (For Headless / CLI only -  In Web this is overridden by Session Init)
+# The security analyzer to use
 #security_analyzer = ""

 #################################### Eval ####################################
--- a/containers/dev/compose.yml
+++ b/containers/dev/compose.yml
@@ -11,7 +11,7 @@ services:
      - BACKEND_HOST=${BACKEND_HOST:-"0.0.0.0"}
      - SANDBOX_API_HOSTNAME=host.docker.internal
      #
-      - SANDBOX_RUNTIME_CONTAINER_IMAGE=${SANDBOX_RUNTIME_CONTAINER_IMAGE:-ghcr.io/all-hands-ai/runtime:0.15-nikolaik}
+      - SANDBOX_RUNTIME_CONTAINER_IMAGE=${SANDBOX_RUNTIME_CONTAINER_IMAGE:-ghcr.io/all-hands-ai/runtime:0.14-nikolaik}
      - SANDBOX_USER_ID=${SANDBOX_USER_ID:-1234}
      - WORKSPACE_MOUNT_PATH=${WORKSPACE_BASE:-$PWD/workspace}
    ports:
--- a/docs/i18n/fr/docusaurus-plugin-content-docs/current/usage/about.md
+++ b/docs/i18n/fr/docusaurus-plugin-content-docs/current/usage/about.md
@@ -27,7 +27,7 @@ Pour plus de détails, veuillez consulter [ce document](https://github.com/All-H

 Nous avons à la fois un espace de travail Slack pour la collaboration sur la construction d'OpenHands et un serveur Discord pour discuter de tout ce qui est lié, par exemple, à ce projet, LLM, agent, etc.

- [Espace de travail Slack](https://join.slack.com/t/openhands-ai/shared_invite/zt-2vbfigwev-G03twSpXaErwzYVD4CFiBg)
+- [Espace de travail Slack](https://join.slack.com/t/opendevin/shared_invite/zt-2oikve2hu-UDxHeo8nsE69y6T7yFX_BA)
 - [Serveur Discord](https://discord.gg/ESHStjSjD4)

 Si vous souhaitez contribuer, n'hésitez pas à rejoindre notre communauté. Simplifions ensemble l'ingénierie logicielle !
--- a/docs/i18n/fr/docusaurus-plugin-content-docs/current/usage/how-to/evaluation-harness.md
+++ b/docs/i18n/fr/docusaurus-plugin-content-docs/current/usage/how-to/evaluation-harness.md
@@ -76,7 +76,7 @@ La fonction `run_controller()` est le cœur de l'exécution d'OpenHands. Elle g

 ## Le moyen le plus simple de commencer : Explorer les benchmarks existants

-Nous vous encourageons à examiner les différents benchmarks d'évaluation disponibles dans le [répertoire `evaluation/benchmarks/`](https://github.com/All-Hands-AI/OpenHands/blob/main/evaluation/benchmarks) de notre dépôt.
+Nous vous encourageons à examiner les différents benchmarks d'évaluation disponibles dans le [répertoire `evaluation/`](https://github.com/All-Hands-AI/OpenHands/blob/main/evaluation) de notre dépôt.

 Pour intégrer votre propre benchmark, nous vous suggérons de commencer par celui qui ressemble le plus à vos besoins. Cette approche peut considérablement rationaliser votre processus d'intégration, vous permettant de vous appuyer sur les structures existantes et de les adapter à vos exigences spécifiques.

--- a/docs/i18n/zh-Hans/docusaurus-plugin-content-docs/current/usage/about.md
+++ b/docs/i18n/zh-Hans/docusaurus-plugin-content-docs/current/usage/about.md
@@ -27,7 +27,7 @@ OpenHands 是一个社区驱动的项目，我们欢迎每个人的贡献。无

 我们有 Slack 工作区用于协作构建 OpenHands，也有 Discord 服务器用于讨论任何相关的内容，例如此项目、大语言模型、代理等。

- [Slack 工作区](https://join.slack.com/t/openhands-ai/shared_invite/zt-2vbfigwev-G03twSpXaErwzYVD4CFiBg)
+- [Slack 工作区](https://join.slack.com/t/opendevin/shared_invite/zt-2oikve2hu-UDxHeo8nsE69y6T7yFX_BA)
 - [Discord 服务器](https://discord.gg/ESHStjSjD4)

 如果你想做出贡献，欢迎加入我们的社区。让我们一起简化软件工程！
--- a/docs/i18n/zh-Hans/docusaurus-plugin-content-docs/current/usage/how-to/evaluation-harness.md
+++ b/docs/i18n/zh-Hans/docusaurus-plugin-content-docs/current/usage/how-to/evaluation-harness.md
@@ -73,7 +73,7 @@ OpenHands 的主要入口点在 `openhands/core/main.py` 中。以下是它工

 ## 入门最简单的方法：探索现有基准

-我们鼓励您查看我们仓库的 [`evaluation/benchmarks/` 目录](https://github.com/All-Hands-AI/OpenHands/blob/main/evaluation/benchmarks)中提供的各种评估基准。
+我们鼓励您查看我们仓库的 [`evaluation/` 目录](https://github.com/All-Hands-AI/OpenHands/blob/main/evaluation)中提供的各种评估基准。

 要集成您自己的基准，我们建议从最接近您需求的基准开始。这种方法可以显著简化您的集成过程，允许您在现有结构的基础上进行构建并使其适应您的特定要求。

--- a/docs/modules/usage/configuration-options.md
+++ b/docs/modules/usage/configuration-options.md
@@ -1,465 +0,0 @@
-# Configuration Options
-
-This guide details all configuration options available for OpenHands, helping you customize its behavior and integrate it with other services.
-
-:::note
-If you are running in [GUI Mode](https://docs.all-hands.dev/modules/usage/how-to/gui-mode), the settings available in the Settings UI will always
-take precedence.
-:::
-
---
-
-# Table of Contents
-
-1. [Core Configuration](#core-configuration)
-   - [API Keys](#api-keys)
-   - [Workspace](#workspace)
-   - [Debugging and Logging](#debugging-and-logging)
-   - [Session Management](#session-management)
-   - [Trajectories](#trajectories)
-   - [File Store](#file-store)
-   - [Task Management](#task-management)
-   - [Sandbox Configuration](#sandbox-configuration)
-   - [Miscellaneous](#miscellaneous)
-2. [LLM Configuration](#llm-configuration)
-   - [AWS Credentials](#aws-credentials)
-   - [API Configuration](#api-configuration)
-   - [Custom LLM Provider](#custom-llm-provider)
-   - [Embeddings](#embeddings)
-   - [Message Handling](#message-handling)
-   - [Model Selection](#model-selection)
-   - [Retrying](#retrying)
-   - [Advanced Options](#advanced-options)
-3. [Agent Configuration](#agent-configuration)
-   - [Microagent Configuration](#microagent-configuration)
-   - [Memory Configuration](#memory-configuration)
-   - [LLM Configuration](#llm-configuration-2)
-   - [ActionSpace Configuration](#actionspace-configuration)
-   - [Microagent Usage](#microagent-usage)
-4. [Sandbox Configuration](#sandbox-configuration-2)
-   - [Execution](#execution)
-   - [Container Image](#container-image)
-   - [Networking](#networking)
-   - [Linting and Plugins](#linting-and-plugins)
-   - [Dependencies and Environment](#dependencies-and-environment)
-   - [Evaluation](#evaluation)
-5. [Security Configuration](#security-configuration)
-   - [Confirmation Mode](#confirmation-mode)
-   - [Security Analyzer](#security-analyzer)
-
---
-
-## Core Configuration
-
-The core configuration options are defined in the `[core]` section of the `config.toml` file.
-
-**API Keys**
- `e2b_api_key`
-  - Type: `str`
-  - Default: `""`
-  - Description: API key for E2B
-
- `modal_api_token_id`
-  - Type: `str`
-  - Default: `""`
-  - Description: API token ID for Modal
-
- `modal_api_token_secret`
-  - Type: `str`
-  - Default: `""`
-  - Description: API token secret for Modal
-
-**Workspace**
- `workspace_base`
-  - Type: `str`
-  - Default: `"./workspace"`
-  - Description: Base path for the workspace
-
- `cache_dir`
-  - Type: `str`
-  - Default: `"/tmp/cache"`
-  - Description: Cache directory path
-
-**Debugging and Logging**
- `debug`
-  - Type: `bool`
-  - Default: `false`
-  - Description: Enable debugging
-
- `disable_color`
-  - Type: `bool`
-  - Default: `false`
-  - Description: Disable color in terminal output
-
-**Trajectories**
- `trajectories_path`
-  - Type: `str`
-  - Default: `"./trajectories"`
-  - Description: Path to store trajectories (can be a folder or a file). If it's a folder, the trajectories will be saved in a file named with the session id name and .json extension, in that folder.
-
-**File Store**
- `file_store_path`
-  - Type: `str`
-  - Default: `"/tmp/file_store"`
-  - Description: File store path
-
- `file_store`
-  - Type: `str`
-  - Default: `"memory"`
-  - Description: File store type
-
- `file_uploads_allowed_extensions`
-  - Type: `list of str`
-  - Default: `[".*"]`
-  - Description: List of allowed file extensions for uploads
-
- `file_uploads_max_file_size_mb`
-  - Type: `int`
-  - Default: `0`
-  - Description: Maximum file size for uploads, in megabytes
-
- `file_uploads_restrict_file_types`
-  - Type: `bool`
-  - Default: `false`
-  - Description: Restrict file types for file uploads
-
- `file_uploads_allowed_extensions`
-  - Type: `list of str`
-  - Default: `[".*"]`
-  - Description: List of allowed file extensions for uploads
-
-**Task Management**
- `max_budget_per_task`
-  - Type: `float`
-  - Default: `0.0`
-  - Description: Maximum budget per task (0.0 means no limit)
-
- `max_iterations`
-  - Type: `int`
-  - Default: `100`
-  - Description: Maximum number of iterations
-
-**Sandbox Configuration**
- `workspace_mount_path_in_sandbox`
-  - Type: `str`
-  - Default: `"/workspace"`
-  - Description: Path to mount the workspace in the sandbox
-
- `workspace_mount_path`
-  - Type: `str`
-  - Default: `""`
-  - Description: Path to mount the workspace
-
- `workspace_mount_rewrite`
-  - Type: `str`
-  - Default: `""`
-  - Description: Path to rewrite the workspace mount path to. You can usually ignore this, it refers to special cases of running inside another container.
-
-**Miscellaneous**
- `run_as_openhands`
-  - Type: `bool`
-  - Default: `true`
-  - Description: Run as OpenHands
-
- `runtime`
-  - Type: `str`
-  - Default: `"eventstream"`
-  - Description: Runtime environment
-
- `default_agent`
-  - Type: `str`
-  - Default: `"CodeActAgent"`
-  - Description: Name of the default agent
-
- `jwt_secret`
-  - Type: `str`
-  - Default: `uuid.uuid4().hex`
-  - Description: JWT secret for authentication. Please set it to your own value.
-
-## LLM Configuration
-
-The LLM (Large Language Model) configuration options are defined in the `[llm]` section of the `config.toml` file.
-
-To use these with the docker command, pass in `-e LLM_<option>`. Example: `-e LLM_NUM_RETRIES`.
-
-**AWS Credentials**
- `aws_access_key_id`
-  - Type: `str`
-  - Default: `""`
-  - Description: AWS access key ID
-
- `aws_region_name`
-  - Type: `str`
-  - Default: `""`
-  - Description: AWS region name
-
- `aws_secret_access_key`
-  - Type: `str`
-  - Default: `""`
-  - Description: AWS secret access key
-
-**API Configuration**
- `api_key`
-  - Type: `str`
-  - Default: `None`
-  - Description: API key to use
-
- `base_url`
-  - Type: `str`
-  - Default: `""`
-  - Description: API base URL
-
- `api_version`
-  - Type: `str`
-  - Default: `""`
-  - Description: API version
-
- `input_cost_per_token`
-  - Type: `float`
-  - Default: `0.0`
-  - Description: Cost per input token
-
- `output_cost_per_token`
-  - Type: `float`
-  - Default: `0.0`
-  - Description: Cost per output token
-
-**Custom LLM Provider**
- `custom_llm_provider`
-  - Type: `str`
-  - Default: `""`
-  - Description: Custom LLM provider
-
-**Embeddings**
- `embedding_base_url`
-  - Type: `str`
-  - Default: `""`
-  - Description: Embedding API base URL
-
- `embedding_deployment_name`
-  - Type: `str`
-  - Default: `""`
-  - Description: Embedding deployment name
-
- `embedding_model`
-  - Type: `str`
-  - Default: `"local"`
-  - Description: Embedding model to use
-
-**Message Handling**
- `max_message_chars`
-  - Type: `int`
-  - Default: `30000`
-  - Description: The approximate maximum number of characters in the content of an event included in the prompt to the LLM. Larger observations are truncated.
-
- `max_input_tokens`
-  - Type: `int`
-  - Default: `0`
-  - Description: Maximum number of input tokens
-
- `max_output_tokens`
-  - Type: `int`
-  - Default: `0`
-  - Description: Maximum number of output tokens
-
-**Model Selection**
- `model`
-  - Type: `str`
-  - Default: `"claude-3-5-sonnet-20241022"`
-  - Description: Model to use
-
-**Retrying**
- `num_retries`
-  - Type: `int`
-  - Default: `8`
-  - Description: Number of retries to attempt
-
- `retry_max_wait`
-  - Type: `int`
-  - Default: `120`
-  - Description: Maximum wait time (in seconds) between retry attempts
-
- `retry_min_wait`
-  - Type: `int`
-  - Default: `15`
-  - Description: Minimum wait time (in seconds) between retry attempts
-
- `retry_multiplier`
-  - Type: `float`
-  - Default: `2.0`
-  - Description: Multiplier for exponential backoff calculation
-
-**Advanced Options**
- `drop_params`
-  - Type: `bool`
-  - Default: `false`
-  - Description: Drop any unmapped (unsupported) params without causing an exception
-
- `caching_prompt`
-  - Type: `bool`
-  - Default: `true`
-  - Description: Using the prompt caching feature if provided by the LLM and supported
-
- `ollama_base_url`
-  - Type: `str`
-  - Default: `""`
-  - Description: Base URL for the OLLAMA API
-
- `temperature`
-  - Type: `float`
-  - Default: `0.0`
-  - Description: Temperature for the API
-
- `timeout`
-  - Type: `int`
-  - Default: `0`
-  - Description: Timeout for the API
-
- `top_p`
-  - Type: `float`
-  - Default: `1.0`
-  - Description: Top p for the API
-
- `disable_vision`
-  - Type: `bool`
-  - Default: `None`
-  - Description: If model is vision capable, this option allows to disable image processing (useful for cost reduction)
-
-## Agent Configuration
-
-The agent configuration options are defined in the `[agent]` and `[agent.<agent_name>]` sections of the `config.toml` file.
-
-**Microagent Configuration**
- `micro_agent_name`
-  - Type: `str`
-  - Default: `""`
-  - Description: Name of the micro agent to use for this agent
-
-**Memory Configuration**
- `memory_enabled`
-  - Type: `bool`
-  - Default: `false`
-  - Description: Whether long-term memory (embeddings) is enabled
-
- `memory_max_threads`
-  - Type: `int`
-  - Default: `3`
-  - Description: The maximum number of threads indexing at the same time for embeddings
-
-**LLM Configuration**
- `llm_config`
-  - Type: `str`
-  - Default: `'your-llm-config-group'`
-  - Description: The name of the LLM config to use
-
-**ActionSpace Configuration**
- `function_calling`
-  - Type: `bool`
-  - Default: `true`
-  - Description: Whether function calling is enabled
-
- `codeact_enable_browsing`
-  - Type: `bool`
-  - Default: `false`
-  - Description: Whether browsing delegate is enabled in the action space (only works with function calling)
-
- `codeact_enable_llm_editor`
-  - Type: `bool`
-  - Default: `false`
-  - Description: Whether LLM editor is enabled in the action space (only works with function calling)
-
- `codeact_enable_jupyter`
-  - Type: `bool`
-  - Default: `false`
-  - Description: Whether Jupyter is enabled in the action space
-
-**Microagent Usage**
- `use_microagents`
-  - Type: `bool`
-  - Default: `true`
-  - Description: Whether to use microagents at all
-
- `disabled_microagents`
-  - Type: `list of str`
-  - Default: `None`
-  - Description: A list of microagents to disable
-
-## Sandbox Configuration
-
-The sandbox configuration options are defined in the `[sandbox]` section of the `config.toml` file.
-
-To use these with the docker command, pass in `-e SANDBOX_<option>`. Example: `-e SANDBOX_TIMEOUT`.
-
-**Execution**
- `timeout`
-  - Type: `int`
-  - Default: `120`
-  - Description: Sandbox timeout in seconds
-
- `user_id`
-  - Type: `int`
-  - Default: `1000`
-  - Description: Sandbox user ID
-
-**Container Image**
- `base_container_image`
-  - Type: `str`
-  - Default: `"nikolaik/python-nodejs:python3.12-nodejs22"`
-  - Description: Container image to use for the sandbox
-
-**Networking**
- `use_host_network`
-  - Type: `bool`
-  - Default: `false`
-  - Description: Use host network
-
-**Linting and Plugins**
- `enable_auto_lint`
-  - Type: `bool`
-  - Default: `false`
-  - Description: Enable auto linting after editing
-
- `initialize_plugins`
-  - Type: `bool`
-  - Default: `true`
-  - Description: Whether to initialize plugins
-
-**Dependencies and Environment**
- `runtime_extra_deps`
-  - Type: `str`
-  - Default: `""`
-  - Description: Extra dependencies to install in the runtime image
-
- `runtime_startup_env_vars`
-  - Type: `dict`
-  - Default: `{}`
-  - Description: Environment variables to set at the launch of the runtime
-
-**Evaluation**
- `browsergym_eval_env`
-  - Type: `str`
-  - Default: `""`
-  - Description: BrowserGym environment to use for evaluation
-
-## Security Configuration
-
-The security configuration options are defined in the `[security]` section of the `config.toml` file.
-
-To use these with the docker command, pass in `-e SECURITY_<option>`. Example: `-e SECURITY_CONFIRMATION_MODE`.
-
-**Confirmation Mode**
- `confirmation_mode`
-  - Type: `bool`
-  - Default: `false`
-  - Description: Enable confirmation mode
-
-**Security Analyzer**
- `security_analyzer`
-  - Type: `str`
-  - Default: `""`
-  - Description: The security analyzer to use
-
---
-
-> **Note**: Adjust configurations carefully, especially for memory, security, and network-related settings to ensure optimal performance and security.
-Please note that the configuration options may be subject to change in future versions of OpenHands. It's recommended to refer to the official documentation for the most up-to-date information.
--- a/docs/modules/usage/how-to/cli-mode.md
+++ b/docs/modules/usage/how-to/cli-mode.md
@@ -50,7 +50,7 @@ LLM_API_KEY="sk_test_12345"
 ```bash
 docker run -it \
    --pull=always \
-    -e SANDBOX_RUNTIME_CONTAINER_IMAGE=docker.all-hands.dev/all-hands-ai/runtime:0.15-nikolaik \
+    -e SANDBOX_RUNTIME_CONTAINER_IMAGE=docker.all-hands.dev/all-hands-ai/runtime:0.14-nikolaik \
    -e SANDBOX_USER_ID=$(id -u) \
    -e WORKSPACE_MOUNT_PATH=$WORKSPACE_BASE \
    -e LLM_API_KEY=$LLM_API_KEY \
@@ -59,7 +59,7 @@ docker run -it \
    -v /var/run/docker.sock:/var/run/docker.sock \
    --add-host host.docker.internal:host-gateway \
    --name openhands-app-$(date +%Y%m%d%H%M%S) \
-    docker.all-hands.dev/all-hands-ai/openhands:0.15 \
+    docker.all-hands.dev/all-hands-ai/openhands:0.14 \
    python -m openhands.core.cli
 ```

--- a/docs/modules/usage/how-to/evaluation-harness.md
+++ b/docs/modules/usage/how-to/evaluation-harness.md
@@ -73,7 +73,7 @@ The `run_controller()` function is the core of OpenHands's execution. It manages

 ## Easiest way to get started: Exploring Existing Benchmarks

-We encourage you to review the various evaluation benchmarks available in the [`evaluation/benchmarks/` directory](https://github.com/All-Hands-AI/OpenHands/blob/main/evaluation/benchmarks) of our repository.
+We encourage you to review the various evaluation benchmarks available in the [`evaluation/` directory](https://github.com/All-Hands-AI/OpenHands/blob/main/evaluation) of our repository.

 To integrate your own benchmark, we suggest starting with the one that most closely resembles your needs. This approach can significantly streamline your integration process, allowing you to build upon existing structures and adapt them to your specific requirements.

--- a/docs/modules/usage/how-to/github-action.md
+++ b/docs/modules/usage/how-to/github-action.md
@@ -37,15 +37,12 @@ the [README for the OpenHands Resolver](https://github.com/All-Hands-AI/OpenHand

 You can provide custom directions for OpenHands by following the [README for the resolver](https://github.com/All-Hands-AI/OpenHands/blob/main/openhands/resolver/README.md#providing-custom-instructions).

-### Custom configurations
+### Configure custom macro

-Github resolver will automatically check for valid [repository secrets](https://docs.github.com/en/actions/security-for-github-actions/security-guides/using-secrets-in-github-actions?tool=webui#creating-secrets-for-a-repository) or [repository variables](https://docs.github.com/en/actions/writing-workflows/choosing-what-your-workflow-does/store-information-in-variables#creating-configuration-variables-for-a-repository) to customize its behavior. The customization options you can set are:
+To customize the default macro (`@openhands-agent`):

-| **Attribute name**               | **Type** | **Purpose**                                                                                         | **Example**                                     |
-| -------------------------------- | -------- | --------------------------------------------------------------------------------------------------- | ----------------------------------------------- |
-| `OPENHANDS_MAX_ITER`             | Variable | Set max limit for agent iterations                                                                  | `OPENHANDS_MAX_ITER=10`                         |
-| `OPENHANDS_MACRO`                | Variable | Customize default macro for invoking the resolver                                                   | `OPENHANDS_MACRO=@resolveit`                    |
-| `OPENHANDS_BASE_CONTAINER_IMAGE` | Variable | Custom Sandbox ([learn more](https://docs.all-hands.dev/modules/usage/how-to/custom-sandbox-guide)) | `OPENHANDS_BASE_CONTAINER_IMAGE="custom_image"` |
+1. [Create a repository variable](https://docs.github.com/en/actions/writing-workflows/choosing-what-your-workflow-does/store-information-in-variables#creating-configuration-variables-for-a-repository) named `OPENHANDS_MACRO`
+2. Assign the variable a custom value

 ## Writing Effective .openhands_instructions Files

@@ -58,7 +55,6 @@ The `.openhands_instructions` file is a file that you can put in the root direct
 2. **Repository Structure**: Explain the key directories and their purposes, especially highlighting where different types of code (e.g., frontend, backend) are located.

 3. **Development Workflows**: Document the essential commands for:
-
   - Building and setting up the project
   - Running tests
   - Linting and code quality checks
@@ -73,29 +69,24 @@ The `.openhands_instructions` file is a file that you can put in the root direct

 ```markdown
 # Repository Overview
-
 [Brief description of the project]

 ## General Setup
-
 - Main build command
 - Development environment setup
 - Pre-commit checks

 ## Backend
-
 - Location and structure
 - Testing instructions
 - Environment requirements

 ## Frontend
-
 - Setup prerequisites
 - Build and test commands
 - Environment variables

 ## Additional Guidelines
-
 - Code style requirements
 - Special considerations
 - Common workflows
--- a/docs/modules/usage/how-to/gui-mode.md
+++ b/docs/modules/usage/how-to/gui-mode.md
@@ -23,75 +23,10 @@ OpenHands provides a user-friendly Graphical User Interface (GUI) mode for inter

 OpenHands automatically exports a `GITHUB_TOKEN` to the shell environment if it is available. This can happen in two ways:

-1. **Locally (OSS)**: The user directly inputs their GitHub token
-2. **Online (SaaS)**: The token is obtained through GitHub OAuth authentication
+1. Locally (OSS): The user directly inputs their GitHub token.
+2. Online (SaaS): The token is obtained through GitHub OAuth authentication.

-#### Setting Up a Local GitHub Token
-
-1. **Generate a Personal Access Token (PAT)**:
-   - Go to GitHub Settings > Developer Settings > Personal Access Tokens > Tokens (classic)
-   - Click "Generate new token (classic)"
-   - Required scopes:
-     - `repo` (Full control of private repositories)
-     - `workflow` (Update GitHub Action workflows)
-     - `read:org` (Read organization data)
-
-2. **Enter Token in OpenHands**:
-   - Click the Settings button (gear icon) in the top right
-   - Navigate to the "GitHub" section
-   - Paste your token in the "GitHub Token" field
-   - Click "Save" to apply the changes
-
-#### Organizational Token Policies
-
-If you're working with organizational repositories, additional setup may be required:
-
-1. **Check Organization Requirements**:
-   - Organization admins may enforce specific token policies
-   - Some organizations require tokens to be created with SSO enabled
-   - Review your organization's [token policy settings](https://docs.github.com/en/organizations/managing-programmatic-access-to-your-organization/setting-a-personal-access-token-policy-for-your-organization)
-
-2. **Verify Organization Access**:
-   - Go to your token settings on GitHub
-   - Look for the organization under "Organization access"
-   - If required, click "Enable SSO" next to your organization
-   - Complete the SSO authorization process
-
-#### OAuth Authentication (Online Mode)
-
-When using OpenHands in online mode, the GitHub OAuth flow:
-
-1. Requests the following permissions:
-   - Repository access (read/write)
-   - Workflow management
-   - Organization read access
-
-2. Authentication steps:
-   - Click "Sign in with GitHub" when prompted
-   - Review the requested permissions
-   - Authorize OpenHands to access your GitHub account
-   - If using an organization, authorize organization access if prompted
-
-#### Troubleshooting
-
-Common issues and solutions:
-
-1. **Token Not Recognized**:
-   - Ensure the token is properly saved in settings
-   - Check that the token hasn't expired
-   - Verify the token has the required scopes
-   - Try regenerating the token
-
-2. **Organization Access Denied**:
-   - Check if SSO is required but not enabled
-   - Verify organization membership
-   - Contact organization admin if token policies are blocking access
-
-3. **Verifying Token Works**:
-   - The app will show a green checkmark if the token is valid
-   - Try accessing a repository to confirm permissions
-   - Check the browser console for any error messages
-   - Use the "Test Connection" button in settings if available
+When you reach the `/app` route, the app checks if a token is present. If it finds one, it sets it in the environment for the agent to use.

 ### Advanced Settings

--- a/docs/modules/usage/how-to/headless-mode.md
+++ b/docs/modules/usage/how-to/headless-mode.md
@@ -44,7 +44,7 @@ LLM_API_KEY="sk_test_12345"
 ```bash
 docker run -it \
    --pull=always \
-    -e SANDBOX_RUNTIME_CONTAINER_IMAGE=docker.all-hands.dev/all-hands-ai/runtime:0.15-nikolaik \
+    -e SANDBOX_RUNTIME_CONTAINER_IMAGE=docker.all-hands.dev/all-hands-ai/runtime:0.14-nikolaik \
    -e SANDBOX_USER_ID=$(id -u) \
    -e WORKSPACE_MOUNT_PATH=$WORKSPACE_BASE \
    -e LLM_API_KEY=$LLM_API_KEY \
@@ -54,6 +54,6 @@ docker run -it \
    -v /var/run/docker.sock:/var/run/docker.sock \
    --add-host host.docker.internal:host-gateway \
    --name openhands-app-$(date +%Y%m%d%H%M%S) \
-    docker.all-hands.dev/all-hands-ai/openhands:0.15 \
+    docker.all-hands.dev/all-hands-ai/openhands:0.14 \
    python -m openhands.core.main -t "write a bash script that prints hi"
 ```
--- a/docs/modules/usage/installation.mdx
+++ b/docs/modules/usage/installation.mdx
@@ -11,16 +11,16 @@
 The easiest way to run OpenHands is in Docker.

 ```bash
-docker pull docker.all-hands.dev/all-hands-ai/runtime:0.15-nikolaik
+docker pull docker.all-hands.dev/all-hands-ai/runtime:0.14-nikolaik

 docker run -it --rm --pull=always \
-    -e SANDBOX_RUNTIME_CONTAINER_IMAGE=docker.all-hands.dev/all-hands-ai/runtime:0.15-nikolaik \
+    -e SANDBOX_RUNTIME_CONTAINER_IMAGE=docker.all-hands.dev/all-hands-ai/runtime:0.14-nikolaik \
    -e LOG_ALL_EVENTS=true \
    -v /var/run/docker.sock:/var/run/docker.sock \
    -p 3000:3000 \
    --add-host host.docker.internal:host-gateway \
    --name openhands-app \
-    docker.all-hands.dev/all-hands-ai/openhands:0.15
+    docker.all-hands.dev/all-hands-ai/openhands:0.14
 ```

 You can also run OpenHands in a scriptable [headless mode](https://docs.all-hands.dev/modules/usage/how-to/headless-mode), as an [interactive CLI](https://docs.all-hands.dev/modules/usage/how-to/cli-mode), or using the [OpenHands GitHub Action](https://docs.all-hands.dev/modules/usage/how-to/github-action).
--- a/docs/modules/usage/runtimes.md
+++ b/docs/modules/usage/runtimes.md
@@ -16,7 +16,7 @@ some flags being passed to `docker run` that make this possible:

 ```
 docker run # ...
-    -e SANDBOX_RUNTIME_CONTAINER_IMAGE=docker.all-hands.dev/all-hands-ai/runtime:0.15-nikolaik \
+    -e SANDBOX_RUNTIME_CONTAINER_IMAGE=docker.all-hands.dev/all-hands-ai/runtime:0.11-nikolaik \
    -v /var/run/docker.sock:/var/run/docker.sock \
    # ...
 ```
--- a/docs/modules/usage/troubleshooting/troubleshooting.md
+++ b/docs/modules/usage/troubleshooting/troubleshooting.md
@@ -17,7 +17,6 @@ Check out [Notes for WSL on Windows Users](troubleshooting/windows) for some tro
 * [`make build` getting stuck on package installations](#make-build-getting-stuck-on-package-installations)
 * [Sessions are not restored](#sessions-are-not-restored)
 * [Connection to host.docker.internal timed out](#connection-to-host-docker-internal-timed-out)
-* [Error building runtime docker image](#error-building-runtime-docker-image)

 ### Unable to connect to Docker

@@ -179,21 +178,3 @@ which OpenHands makes use of when the main server is running inside a docker con
 * [Install Docker Desktop](https://www.docker.com/products/docker-desktop/)
 * Run OpenHands in [Development Mode](https://github.com/All-Hands-AI/OpenHands/blob/main/Development.md),
  So that the main server is not run inside a container, but still creates dockerized runtime sandboxes.
-
---
-### Error building runtime docker image
-
-**Symptoms**
-Attempts to start a new session fail, and an errors with terms like the following appear in the logs:
-* `debian-security bookworm-security`
-* `InRelease At least one invalid signature was encountered.`
-
-This seems to happen when the hash of an existing external library changes and your local docker instance has
-cached a previous version. To work around this, please try the following:
-
-* Stop any containers where the name has the prefix `openhands-runtime-` :
-  `docker ps --filter name=openhands-runtime- --filter status=running -aq | xargs docker stop`
-* Remove any containers where the name has the prefix `openhands-runtime-` :
-  `docker rmi $(docker images --filter name=openhands-runtime- -q --no-trunc)`
-* Stop and Remove any containers / images where the name has the prefix `openhands-runtime-`
-* Prune containers / images : `docker container prune -f && docker image prune -f`
--- a/docs/package-lock.json
+++ b/docs/package-lock.json
@@ -8,23 +8,23 @@
      "name": "docs",
      "version": "0.0.0",
      "dependencies": {
-        "@docusaurus/core": "^3.6.3",
-        "@docusaurus/plugin-content-pages": "^3.6.3",
-        "@docusaurus/preset-classic": "^3.6.3",
-        "@docusaurus/theme-mermaid": "^3.6.3",
+        "@docusaurus/core": "^3.6.2",
+        "@docusaurus/plugin-content-pages": "^3.6.2",
+        "@docusaurus/preset-classic": "^3.6.2",
+        "@docusaurus/theme-mermaid": "^3.6.2",
        "@mdx-js/react": "^3.1.0",
        "clsx": "^2.0.0",
        "prism-react-renderer": "^2.4.0",
        "react": "^18.3.1",
        "react-dom": "^18.3.1",
-        "react-icons": "^5.4.0",
+        "react-icons": "^5.3.0",
        "react-use": "^17.5.1"
      },
      "devDependencies": {
        "@docusaurus/module-type-aliases": "^3.5.1",
-        "@docusaurus/tsconfig": "^3.6.3",
+        "@docusaurus/tsconfig": "^3.6.2",
        "@docusaurus/types": "^3.5.1",
-        "typescript": "~5.7.2"
+        "typescript": "~5.6.3"
      },
      "engines": {
        "node": ">=18.0"
@@ -3168,9 +3168,9 @@
      }
    },
    "node_modules/@docusaurus/babel": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/babel/-/babel-3.6.3.tgz",
-      "integrity": "sha512-7dW9Hat9EHYCVicFXYA4hjxBY38+hPuCURL8oRF9fySRm7vzNWuEOghA1TXcykuXZp0HLG2td4RhDxCvGG7tNw==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/babel/-/babel-3.6.2.tgz",
+      "integrity": "sha512-v8N8TWGXDsb5sxQC3Rcqb1CZr0LlU1OgqqVBUchN6cpIUr7EJuVJs5eHcIu5Ag8mwO/hWN3f7FE9uaHTMapAbg==",
      "dependencies": {
        "@babel/core": "^7.25.9",
        "@babel/generator": "^7.25.9",
@@ -3182,8 +3182,8 @@
        "@babel/runtime": "^7.25.9",
        "@babel/runtime-corejs3": "^7.25.9",
        "@babel/traverse": "^7.25.9",
-        "@docusaurus/logger": "3.6.3",
-        "@docusaurus/utils": "3.6.3",
+        "@docusaurus/logger": "3.6.2",
+        "@docusaurus/utils": "3.6.2",
        "babel-plugin-dynamic-import-node": "^2.3.3",
        "fs-extra": "^11.1.1",
        "tslib": "^2.6.0"
@@ -3193,16 +3193,16 @@
      }
    },
    "node_modules/@docusaurus/bundler": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/bundler/-/bundler-3.6.3.tgz",
-      "integrity": "sha512-47JLuc8D4wA+6VOvmMd5fUC9rFppBQpQOnxDYiVXffm/DeV/wmm3sbpNd5Y+O+G2+nevLTRnvCm/qyancv0Y3A==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/bundler/-/bundler-3.6.2.tgz",
+      "integrity": "sha512-YkEifEVs4lV931SrHBB4n6WqRowMw+aM/QPH3z8aU+5t1dWa+1p2OPqARS+tSbh3la9ns+L1zIfSbd8RHi2/PQ==",
      "dependencies": {
        "@babel/core": "^7.25.9",
-        "@docusaurus/babel": "3.6.3",
-        "@docusaurus/cssnano-preset": "3.6.3",
-        "@docusaurus/logger": "3.6.3",
-        "@docusaurus/types": "3.6.3",
-        "@docusaurus/utils": "3.6.3",
+        "@docusaurus/babel": "3.6.2",
+        "@docusaurus/cssnano-preset": "3.6.2",
+        "@docusaurus/logger": "3.6.2",
+        "@docusaurus/types": "3.6.2",
+        "@docusaurus/utils": "3.6.2",
        "babel-loader": "^9.2.1",
        "clean-css": "^5.3.2",
        "copy-webpack-plugin": "^11.0.0",
@@ -3236,17 +3236,17 @@
      }
    },
    "node_modules/@docusaurus/core": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/core/-/core-3.6.3.tgz",
-      "integrity": "sha512-xL7FRY9Jr5DWqB6pEnqgKqcMPJOX5V0pgWXi5lCiih11sUBmcFKM7c3+GyxcVeeWFxyYSDP3grLTWqJoP4P9Vw==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/core/-/core-3.6.2.tgz",
+      "integrity": "sha512-irMts/mGLZv8dWcy0WUtbY/U6b5qIfHgQd1/kXMyAxUJo99fL0wFSqhMI+tcxjk0HYy427MXerLMqFJj+Arg1w==",
      "dependencies": {
-        "@docusaurus/babel": "3.6.3",
-        "@docusaurus/bundler": "3.6.3",
-        "@docusaurus/logger": "3.6.3",
-        "@docusaurus/mdx-loader": "3.6.3",
-        "@docusaurus/utils": "3.6.3",
-        "@docusaurus/utils-common": "3.6.3",
-        "@docusaurus/utils-validation": "3.6.3",
+        "@docusaurus/babel": "3.6.2",
+        "@docusaurus/bundler": "3.6.2",
+        "@docusaurus/logger": "3.6.2",
+        "@docusaurus/mdx-loader": "3.6.2",
+        "@docusaurus/utils": "3.6.2",
+        "@docusaurus/utils-common": "3.6.2",
+        "@docusaurus/utils-validation": "3.6.2",
        "boxen": "^6.2.1",
        "chalk": "^4.1.2",
        "chokidar": "^3.5.3",
@@ -3310,9 +3310,9 @@
      }
    },
    "node_modules/@docusaurus/cssnano-preset": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/cssnano-preset/-/cssnano-preset-3.6.3.tgz",
-      "integrity": "sha512-qP7SXrwZ+23GFJdPN4aIHQrZW+oH/7tzwEuc/RNL0+BdZdmIjYQqUxdXsjE4lFxLNZjj0eUrSNYIS6xwfij+5Q==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/cssnano-preset/-/cssnano-preset-3.6.2.tgz",
+      "integrity": "sha512-mBkVa4QMHRwCFCVLYdBlOZuAT1iVVsS7GGSgliSVAeTOagP/AbtlBsCVrBs+keEuDuRF1w/6QEcqDoZe9fa5pw==",
      "dependencies": {
        "cssnano-preset-advanced": "^6.1.2",
        "postcss": "^8.4.38",
@@ -3324,9 +3324,9 @@
      }
    },
    "node_modules/@docusaurus/logger": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/logger/-/logger-3.6.3.tgz",
-      "integrity": "sha512-xSubJixcNyMV9wMV4q0s47CBz3Rlc5jbcCCuij8pfQP8qn/DIpt0ks8W6hQWzHAedg/J/EwxxUOUrnEoKzJo8g==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/logger/-/logger-3.6.2.tgz",
+      "integrity": "sha512-1p4IQhhgLyIfsey4UAdAIW69aUE1Ei6O91Nsw30ryZeDWSG5dh4o3zaRGOLxfAX69Ac/yDm6YCwJOafUxL6Vxg==",
      "dependencies": {
        "chalk": "^4.1.2",
        "tslib": "^2.6.0"
@@ -3336,13 +3336,13 @@
      }
    },
    "node_modules/@docusaurus/mdx-loader": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/mdx-loader/-/mdx-loader-3.6.3.tgz",
-      "integrity": "sha512-3iJdiDz9540ppBseeI93tWTDtUGVkxzh59nMq4ignylxMuXBLK8dFqVeaEor23v1vx6TrGKZ2FuLaTB+U7C0QQ==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/mdx-loader/-/mdx-loader-3.6.2.tgz",
+      "integrity": "sha512-7fbRmNgF3CR96Ja82Ya0/Cdu1OL9UJ/22llNMY8lr5gAbw718Y5ryXMVRIYn0JNLTiSxzgtvW4DIsUWEB8NMpw==",
      "dependencies": {
-        "@docusaurus/logger": "3.6.3",
-        "@docusaurus/utils": "3.6.3",
-        "@docusaurus/utils-validation": "3.6.3",
+        "@docusaurus/logger": "3.6.2",
+        "@docusaurus/utils": "3.6.2",
+        "@docusaurus/utils-validation": "3.6.2",
        "@mdx-js/mdx": "^3.0.0",
        "@slorber/remark-comment": "^1.0.0",
        "escape-html": "^1.0.3",
@@ -3374,11 +3374,11 @@
      }
    },
    "node_modules/@docusaurus/module-type-aliases": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/module-type-aliases/-/module-type-aliases-3.6.3.tgz",
-      "integrity": "sha512-MjaXX9PN/k5ugNvfRZdWyKWq4FsrhN4LEXaj0pEmMebJuBNlFeGyKQUa9DRhJHpadNaiMLrbo9m3U7Ig5YlsZg==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/module-type-aliases/-/module-type-aliases-3.6.2.tgz",
+      "integrity": "sha512-NrJkL2rLTCjHtWOqUvWzwqvJrsKLj0gVJeV6q5yeKdKKgItietcTf2fTRkM9LHKSUN8CBDXxwHABeQvTahvmXQ==",
      "dependencies": {
-        "@docusaurus/types": "3.6.3",
+        "@docusaurus/types": "3.6.2",
        "@types/history": "^4.7.11",
        "@types/react": "*",
        "@types/react-router-config": "*",
@@ -3392,18 +3392,18 @@
      }
    },
    "node_modules/@docusaurus/plugin-content-blog": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/plugin-content-blog/-/plugin-content-blog-3.6.3.tgz",
-      "integrity": "sha512-k0ogWwwJU3pFRFfvW1kRVHxzf2DutLGaaLjAnHVEU6ju+aRP0Z5ap/13DHyPOfHeE4WKpn/M0TqjdwZAcY3kAw==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/plugin-content-blog/-/plugin-content-blog-3.6.2.tgz",
+      "integrity": "sha512-6bJxr6Or4NslEVH3BJuPH30kUWiqUjDRdGPhvxpHmt9W/RY2/6u72WICG3bW3dLFxJ/2uDLBU92lHnatpvo7Ew==",
      "dependencies": {
-        "@docusaurus/core": "3.6.3",
-        "@docusaurus/logger": "3.6.3",
-        "@docusaurus/mdx-loader": "3.6.3",
-        "@docusaurus/theme-common": "3.6.3",
-        "@docusaurus/types": "3.6.3",
-        "@docusaurus/utils": "3.6.3",
-        "@docusaurus/utils-common": "3.6.3",
-        "@docusaurus/utils-validation": "3.6.3",
+        "@docusaurus/core": "3.6.2",
+        "@docusaurus/logger": "3.6.2",
+        "@docusaurus/mdx-loader": "3.6.2",
+        "@docusaurus/theme-common": "3.6.2",
+        "@docusaurus/types": "3.6.2",
+        "@docusaurus/utils": "3.6.2",
+        "@docusaurus/utils-common": "3.6.2",
+        "@docusaurus/utils-validation": "3.6.2",
        "cheerio": "1.0.0-rc.12",
        "feed": "^4.2.2",
        "fs-extra": "^11.1.1",
@@ -3425,19 +3425,19 @@
      }
    },
    "node_modules/@docusaurus/plugin-content-docs": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/plugin-content-docs/-/plugin-content-docs-3.6.3.tgz",
-      "integrity": "sha512-r2wS8y/fsaDcxkm20W5bbYJFPzdWdEaTWVYjNxlHlcmX086eqQR1Fomlg9BHTJ0dLXPzAlbC8EN4XqMr3QzNCQ==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/plugin-content-docs/-/plugin-content-docs-3.6.2.tgz",
+      "integrity": "sha512-e6WW1g10RIXXLN/rrtqTi/FyJ1Hj3X9Mmgz4V11/0pDCxIGGI8m4ocbAglUlLtgvbLD5viNLefl/NwbOW3JXiQ==",
      "dependencies": {
-        "@docusaurus/core": "3.6.3",
-        "@docusaurus/logger": "3.6.3",
-        "@docusaurus/mdx-loader": "3.6.3",
-        "@docusaurus/module-type-aliases": "3.6.3",
-        "@docusaurus/theme-common": "3.6.3",
-        "@docusaurus/types": "3.6.3",
-        "@docusaurus/utils": "3.6.3",
-        "@docusaurus/utils-common": "3.6.3",
-        "@docusaurus/utils-validation": "3.6.3",
+        "@docusaurus/core": "3.6.2",
+        "@docusaurus/logger": "3.6.2",
+        "@docusaurus/mdx-loader": "3.6.2",
+        "@docusaurus/module-type-aliases": "3.6.2",
+        "@docusaurus/theme-common": "3.6.2",
+        "@docusaurus/types": "3.6.2",
+        "@docusaurus/utils": "3.6.2",
+        "@docusaurus/utils-common": "3.6.2",
+        "@docusaurus/utils-validation": "3.6.2",
        "@types/react-router-config": "^5.0.7",
        "combine-promises": "^1.1.0",
        "fs-extra": "^11.1.1",
@@ -3456,15 +3456,15 @@
      }
    },
    "node_modules/@docusaurus/plugin-content-pages": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/plugin-content-pages/-/plugin-content-pages-3.6.3.tgz",
-      "integrity": "sha512-eHrmTgjgLZsuqfsYr5X2xEwyIcck0wseSofWrjTwT9FLOWp+KDmMAuVK+wRo7sFImWXZk3oV/xX/g9aZrhD7OA==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/plugin-content-pages/-/plugin-content-pages-3.6.2.tgz",
+      "integrity": "sha512-fo4NyGkw10lYHyHaTxE6TZLYnxNtCfRHeZkNK1N9pBYqe7TT2dBUNAEeVW2U3ed9m6YuB7JKSQsa++GGmcP+6g==",
      "dependencies": {
-        "@docusaurus/core": "3.6.3",
-        "@docusaurus/mdx-loader": "3.6.3",
-        "@docusaurus/types": "3.6.3",
-        "@docusaurus/utils": "3.6.3",
-        "@docusaurus/utils-validation": "3.6.3",
+        "@docusaurus/core": "3.6.2",
+        "@docusaurus/mdx-loader": "3.6.2",
+        "@docusaurus/types": "3.6.2",
+        "@docusaurus/utils": "3.6.2",
+        "@docusaurus/utils-validation": "3.6.2",
        "fs-extra": "^11.1.1",
        "tslib": "^2.6.0",
        "webpack": "^5.88.1"
@@ -3478,13 +3478,13 @@
      }
    },
    "node_modules/@docusaurus/plugin-debug": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/plugin-debug/-/plugin-debug-3.6.3.tgz",
-      "integrity": "sha512-zB9GXfIZNPRfzKnNjU6xGVrqn9bPXuGhpjgsuc/YtcTDjnjhasg38NdYd5LEqXex5G/zIorQgWB3n6x/Ut62vQ==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/plugin-debug/-/plugin-debug-3.6.2.tgz",
+      "integrity": "sha512-T/eS3VvHElpeV5S8uwp7Si4ujEynmgFtJLvA2CSa5pzQuOF1EEghF9nekAIj0cWtDHsqNUDZNr8hK1brivFXSg==",
      "dependencies": {
-        "@docusaurus/core": "3.6.3",
-        "@docusaurus/types": "3.6.3",
-        "@docusaurus/utils": "3.6.3",
+        "@docusaurus/core": "3.6.2",
+        "@docusaurus/types": "3.6.2",
+        "@docusaurus/utils": "3.6.2",
        "fs-extra": "^11.1.1",
        "react-json-view-lite": "^1.2.0",
        "tslib": "^2.6.0"
@@ -3498,13 +3498,13 @@
      }
    },
    "node_modules/@docusaurus/plugin-google-analytics": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/plugin-google-analytics/-/plugin-google-analytics-3.6.3.tgz",
-      "integrity": "sha512-rCDNy1QW8Dag7nZq67pcum0bpFLrwvxJhYuVprhFh8BMBDxV0bY+bAkGHbSf68P3Bk9C3hNOAXX1srGLIDvcTA==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/plugin-google-analytics/-/plugin-google-analytics-3.6.2.tgz",
+      "integrity": "sha512-B7ihrr3wz8e4XqW+dIAtq844u3Z83u5CeiL1xrCqzFH+vDCjUZHTamS3zKXNcgi6YVVe6hUQXPG15ltaqQaVPQ==",
      "dependencies": {
-        "@docusaurus/core": "3.6.3",
-        "@docusaurus/types": "3.6.3",
-        "@docusaurus/utils-validation": "3.6.3",
+        "@docusaurus/core": "3.6.2",
+        "@docusaurus/types": "3.6.2",
+        "@docusaurus/utils-validation": "3.6.2",
        "tslib": "^2.6.0"
      },
      "engines": {
@@ -3516,13 +3516,13 @@
      }
    },
    "node_modules/@docusaurus/plugin-google-gtag": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/plugin-google-gtag/-/plugin-google-gtag-3.6.3.tgz",
-      "integrity": "sha512-+OyDvhM6rqVkQOmLVkQWVJAizEEfkPzVWtIHXlWPOCFGK9X4/AWeBSrU0WG4iMg9Z4zD4YDRrU+lvI4s6DSC+w==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/plugin-google-gtag/-/plugin-google-gtag-3.6.2.tgz",
+      "integrity": "sha512-V8ijI6qddAAkJ0vd8sjZ7S/apRTLJn9dAwvj/rSMd93witGdKINemL+9TyfLkhcXKTxyqRT8zKdu8ewjPXqKHg==",
      "dependencies": {
-        "@docusaurus/core": "3.6.3",
-        "@docusaurus/types": "3.6.3",
-        "@docusaurus/utils-validation": "3.6.3",
+        "@docusaurus/core": "3.6.2",
+        "@docusaurus/types": "3.6.2",
+        "@docusaurus/utils-validation": "3.6.2",
        "@types/gtag.js": "^0.0.12",
        "tslib": "^2.6.0"
      },
@@ -3535,13 +3535,13 @@
      }
    },
    "node_modules/@docusaurus/plugin-google-tag-manager": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/plugin-google-tag-manager/-/plugin-google-tag-manager-3.6.3.tgz",
-      "integrity": "sha512-1M6UPB13gWUtN2UHX083/beTn85PlRI9ABItTl/JL1FJ5dJTWWFXXsHf9WW/6hrVwthwTeV/AGbGKvLKV+IlCA==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/plugin-google-tag-manager/-/plugin-google-tag-manager-3.6.2.tgz",
+      "integrity": "sha512-fnWQ5FdN9f8c8VTgjaQ98208Y+d/JjHhD506rWIIL9rt1cJOf29XElxvOeKpMJadfkgY5KLZSAiHkGt+4qgN4g==",
      "dependencies": {
-        "@docusaurus/core": "3.6.3",
-        "@docusaurus/types": "3.6.3",
-        "@docusaurus/utils-validation": "3.6.3",
+        "@docusaurus/core": "3.6.2",
+        "@docusaurus/types": "3.6.2",
+        "@docusaurus/utils-validation": "3.6.2",
        "tslib": "^2.6.0"
      },
      "engines": {
@@ -3553,16 +3553,16 @@
      }
    },
    "node_modules/@docusaurus/plugin-sitemap": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/plugin-sitemap/-/plugin-sitemap-3.6.3.tgz",
-      "integrity": "sha512-94qOO4M9Fwv9KfVQJsgbe91k+fPJ4byf1L3Ez8TUa6TAFPo/BrLwQ80zclHkENlL1824TuxkcMKv33u6eydQCg==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/plugin-sitemap/-/plugin-sitemap-3.6.2.tgz",
+      "integrity": "sha512-qcAQAP1Ot0dZpeRoJ0L/Zck5FVDkll2IleVZQLzxeRVDZIw1P9/TK7/Aw1w2pmH7dmw/Cwk/cLSVRvLAmp9k7A==",
      "dependencies": {
-        "@docusaurus/core": "3.6.3",
-        "@docusaurus/logger": "3.6.3",
-        "@docusaurus/types": "3.6.3",
-        "@docusaurus/utils": "3.6.3",
-        "@docusaurus/utils-common": "3.6.3",
-        "@docusaurus/utils-validation": "3.6.3",
+        "@docusaurus/core": "3.6.2",
+        "@docusaurus/logger": "3.6.2",
+        "@docusaurus/types": "3.6.2",
+        "@docusaurus/utils": "3.6.2",
+        "@docusaurus/utils-common": "3.6.2",
+        "@docusaurus/utils-validation": "3.6.2",
        "fs-extra": "^11.1.1",
        "sitemap": "^7.1.1",
        "tslib": "^2.6.0"
@@ -3576,23 +3576,23 @@
      }
    },
    "node_modules/@docusaurus/preset-classic": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/preset-classic/-/preset-classic-3.6.3.tgz",
-      "integrity": "sha512-VHSYWROT3flvNNI1SrnMOtW1EsjeHNK9dhU6s9eY5hryZe79lUqnZJyze/ymDe2LXAqzyj6y5oYvyBoZZk6ErA==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/preset-classic/-/preset-classic-3.6.2.tgz",
+      "integrity": "sha512-r2n5eHdhiNSrJGsrrYcw+WsyStmXxe0ZG3RdA9LVyK5+jBHM8blrUWJEDugnzCNbyhUzhdtcmgCC9fhdAvKuQw==",
      "dependencies": {
-        "@docusaurus/core": "3.6.3",
-        "@docusaurus/plugin-content-blog": "3.6.3",
-        "@docusaurus/plugin-content-docs": "3.6.3",
-        "@docusaurus/plugin-content-pages": "3.6.3",
-        "@docusaurus/plugin-debug": "3.6.3",
-        "@docusaurus/plugin-google-analytics": "3.6.3",
-        "@docusaurus/plugin-google-gtag": "3.6.3",
-        "@docusaurus/plugin-google-tag-manager": "3.6.3",
-        "@docusaurus/plugin-sitemap": "3.6.3",
-        "@docusaurus/theme-classic": "3.6.3",
-        "@docusaurus/theme-common": "3.6.3",
-        "@docusaurus/theme-search-algolia": "3.6.3",
-        "@docusaurus/types": "3.6.3"
+        "@docusaurus/core": "3.6.2",
+        "@docusaurus/plugin-content-blog": "3.6.2",
+        "@docusaurus/plugin-content-docs": "3.6.2",
+        "@docusaurus/plugin-content-pages": "3.6.2",
+        "@docusaurus/plugin-debug": "3.6.2",
+        "@docusaurus/plugin-google-analytics": "3.6.2",
+        "@docusaurus/plugin-google-gtag": "3.6.2",
+        "@docusaurus/plugin-google-tag-manager": "3.6.2",
+        "@docusaurus/plugin-sitemap": "3.6.2",
+        "@docusaurus/theme-classic": "3.6.2",
+        "@docusaurus/theme-common": "3.6.2",
+        "@docusaurus/theme-search-algolia": "3.6.2",
+        "@docusaurus/types": "3.6.2"
      },
      "engines": {
        "node": ">=18.0"
@@ -3603,23 +3603,23 @@
      }
    },
    "node_modules/@docusaurus/theme-classic": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/theme-classic/-/theme-classic-3.6.3.tgz",
-      "integrity": "sha512-1RRLK1tSArI2c00qugWYO3jRocjOZwGF1mBzPPylDVRwWCS/rnWWR91ChdbbaxIupRJ+hX8ZBYrwr5bbU0oztQ==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/theme-classic/-/theme-classic-3.6.2.tgz",
+      "integrity": "sha512-bCdOPqPNezhLx+hgNVO2Cf+8/1AHa9uHDOqTx/CKAx2I0J/jV9G+6JiMtpSRKGNfBoLT1O+56/7+WtkOf54xTw==",
      "dependencies": {
-        "@docusaurus/core": "3.6.3",
-        "@docusaurus/logger": "3.6.3",
-        "@docusaurus/mdx-loader": "3.6.3",
-        "@docusaurus/module-type-aliases": "3.6.3",
-        "@docusaurus/plugin-content-blog": "3.6.3",
-        "@docusaurus/plugin-content-docs": "3.6.3",
-        "@docusaurus/plugin-content-pages": "3.6.3",
-        "@docusaurus/theme-common": "3.6.3",
-        "@docusaurus/theme-translations": "3.6.3",
-        "@docusaurus/types": "3.6.3",
-        "@docusaurus/utils": "3.6.3",
-        "@docusaurus/utils-common": "3.6.3",
-        "@docusaurus/utils-validation": "3.6.3",
+        "@docusaurus/core": "3.6.2",
+        "@docusaurus/logger": "3.6.2",
+        "@docusaurus/mdx-loader": "3.6.2",
+        "@docusaurus/module-type-aliases": "3.6.2",
+        "@docusaurus/plugin-content-blog": "3.6.2",
+        "@docusaurus/plugin-content-docs": "3.6.2",
+        "@docusaurus/plugin-content-pages": "3.6.2",
+        "@docusaurus/theme-common": "3.6.2",
+        "@docusaurus/theme-translations": "3.6.2",
+        "@docusaurus/types": "3.6.2",
+        "@docusaurus/utils": "3.6.2",
+        "@docusaurus/utils-common": "3.6.2",
+        "@docusaurus/utils-validation": "3.6.2",
        "@mdx-js/react": "^3.0.0",
        "clsx": "^2.0.0",
        "copy-text-to-clipboard": "^3.2.0",
@@ -3643,14 +3643,14 @@
      }
    },
    "node_modules/@docusaurus/theme-common": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/theme-common/-/theme-common-3.6.3.tgz",
-      "integrity": "sha512-b8ZkhczXHDxWWyvz+YJy4t/PlPbEogTTbgnHoflYnH7rmRtyoodTsu8WVM12la5LmlMJBclBXFl29OH8kPE7gg==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/theme-common/-/theme-common-3.6.2.tgz",
+      "integrity": "sha512-lfgsL064KEHpCkgGUc0OYoUPCpYfzggp6Hof8sz59UuKiLvb/Z7raewE9/NfocrJ2HZI17rLgMX3SQlRDh/5gg==",
      "dependencies": {
-        "@docusaurus/mdx-loader": "3.6.3",
-        "@docusaurus/module-type-aliases": "3.6.3",
-        "@docusaurus/utils": "3.6.3",
-        "@docusaurus/utils-common": "3.6.3",
+        "@docusaurus/mdx-loader": "3.6.2",
+        "@docusaurus/module-type-aliases": "3.6.2",
+        "@docusaurus/utils": "3.6.2",
+        "@docusaurus/utils-common": "3.6.2",
        "@types/history": "^4.7.11",
        "@types/react": "*",
        "@types/react-router-config": "*",
@@ -3670,15 +3670,15 @@
      }
    },
    "node_modules/@docusaurus/theme-mermaid": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/theme-mermaid/-/theme-mermaid-3.6.3.tgz",
-      "integrity": "sha512-kIqpjNCP/9R2GGf8UmiDxD3CkOAEJuJIEFlaKMgQtjVxa/vH+9PLI1+DFbArGoG4+0ENTYUq8phHPW7SeL36uQ==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/theme-mermaid/-/theme-mermaid-3.6.2.tgz",
+      "integrity": "sha512-Ui+rBtqMPKj3RCOxNlY04i1tEjNg+fZg4URTvkHmYR07hcKaJw+vkw+wlaYjd0HFZk+3Er9vUAcwsCWuea4cVQ==",
      "dependencies": {
-        "@docusaurus/core": "3.6.3",
-        "@docusaurus/module-type-aliases": "3.6.3",
-        "@docusaurus/theme-common": "3.6.3",
-        "@docusaurus/types": "3.6.3",
-        "@docusaurus/utils-validation": "3.6.3",
+        "@docusaurus/core": "3.6.2",
+        "@docusaurus/module-type-aliases": "3.6.2",
+        "@docusaurus/theme-common": "3.6.2",
+        "@docusaurus/types": "3.6.2",
+        "@docusaurus/utils-validation": "3.6.2",
        "mermaid": ">=10.4",
        "tslib": "^2.6.0"
      },
@@ -3691,18 +3691,18 @@
      }
    },
    "node_modules/@docusaurus/theme-search-algolia": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/theme-search-algolia/-/theme-search-algolia-3.6.3.tgz",
-      "integrity": "sha512-rt+MGCCpYgPyWCGXtbxlwFbTSobu15jWBTPI2LHsHNa5B0zSmOISX6FWYAPt5X1rNDOqMGM0FATnh7TBHRohVA==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/theme-search-algolia/-/theme-search-algolia-3.6.2.tgz",
+      "integrity": "sha512-SFLS+Rq8Cg2yepnHucA9sRpIR97yHvZWlCgMzBLunV3KHbB6hD2h5HPhFV39wYHYCjJUAOH1lX9poJ1qKYuSvg==",
      "dependencies": {
        "@docsearch/react": "^3.5.2",
-        "@docusaurus/core": "3.6.3",
-        "@docusaurus/logger": "3.6.3",
-        "@docusaurus/plugin-content-docs": "3.6.3",
-        "@docusaurus/theme-common": "3.6.3",
-        "@docusaurus/theme-translations": "3.6.3",
-        "@docusaurus/utils": "3.6.3",
-        "@docusaurus/utils-validation": "3.6.3",
+        "@docusaurus/core": "3.6.2",
+        "@docusaurus/logger": "3.6.2",
+        "@docusaurus/plugin-content-docs": "3.6.2",
+        "@docusaurus/theme-common": "3.6.2",
+        "@docusaurus/theme-translations": "3.6.2",
+        "@docusaurus/utils": "3.6.2",
+        "@docusaurus/utils-validation": "3.6.2",
        "algoliasearch": "^4.18.0",
        "algoliasearch-helper": "^3.13.3",
        "clsx": "^2.0.0",
@@ -3721,9 +3721,9 @@
      }
    },
    "node_modules/@docusaurus/theme-translations": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/theme-translations/-/theme-translations-3.6.3.tgz",
-      "integrity": "sha512-Gb0regclToVlngSIIwUCtBMQBq48qVUaN1XQNKW4XwlsgUyk0vP01LULdqbem7czSwIeBAFXFoORJ0RPX7ht/w==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/theme-translations/-/theme-translations-3.6.2.tgz",
+      "integrity": "sha512-LIWrYoDUsOTKmb0c7IQzawiPUTAaczBs5IOx6srxOWoTHVUMLzJCkl5Y6whfuRrnul8G05qv2vk238bN5Ko62g==",
      "dependencies": {
        "fs-extra": "^11.1.1",
        "tslib": "^2.6.0"
@@ -3733,15 +3733,15 @@
      }
    },
    "node_modules/@docusaurus/tsconfig": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/tsconfig/-/tsconfig-3.6.3.tgz",
-      "integrity": "sha512-1pT/rTrRpMV15E4tJH95W5PrjboMn5JkKF+Ys8cTjMegetiXjs0gPFOSDA5hdTlberKQLDO50xPjMJHondLuzA==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/tsconfig/-/tsconfig-3.6.2.tgz",
+      "integrity": "sha512-TWLkyYHBYhIJNcXCEc3D1M9R8UFV4IZ82rGef5U9mE1ZrcgDUlZxYaYdoSuHrPrzPRIl3orjmpscO2FAk2gdZw==",
      "dev": true
    },
    "node_modules/@docusaurus/types": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/types/-/types-3.6.3.tgz",
-      "integrity": "sha512-xD9oTGDrouWzefkhe9ogB2fDV96/82cRpNGx2HIvI5L87JHNhQVIWimQ/3JIiiX/TEd5S9s+VO6FFguwKNRVow==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/types/-/types-3.6.2.tgz",
+      "integrity": "sha512-117Wsk6xXrWEAsCYCXS3TGJv5tkdIZDcd7T/V0UJvKYmY0gyVPPcEQChy8yTdjbIkbB2q4fa7Jpox72Qv86mqQ==",
      "dependencies": {
        "@mdx-js/mdx": "^3.0.0",
        "@types/history": "^4.7.11",
@@ -3759,13 +3759,13 @@
      }
    },
    "node_modules/@docusaurus/utils": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/utils/-/utils-3.6.3.tgz",
-      "integrity": "sha512-0R/FR3bKVl4yl8QwbL4TYFfR+OXBRpVUaTJdENapBGR3YMwfM6/JnhGilWQO8AOwPJGtGoDK7ib8+8UF9f3OZQ==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/utils/-/utils-3.6.2.tgz",
+      "integrity": "sha512-oxnpUcFZGE3uPCDoXr8GJriB3VWM9sFjPedFidX3Fsz87l1NZNc1wtbKPfQ7GYFDMYo2IGlAv5+47Me9RkM6lg==",
      "dependencies": {
-        "@docusaurus/logger": "3.6.3",
-        "@docusaurus/types": "3.6.3",
-        "@docusaurus/utils-common": "3.6.3",
+        "@docusaurus/logger": "3.6.2",
+        "@docusaurus/types": "3.6.2",
+        "@docusaurus/utils-common": "3.6.2",
        "@svgr/webpack": "^8.1.0",
        "escape-string-regexp": "^4.0.0",
        "file-loader": "^6.2.0",
@@ -3790,11 +3790,11 @@
      }
    },
    "node_modules/@docusaurus/utils-common": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/utils-common/-/utils-common-3.6.3.tgz",
-      "integrity": "sha512-v4nKDaANLgT3pMBewHYEMAl/ufY0LkXao1QkFWzI5huWFOmNQ2UFzv2BiKeHX5Ownis0/w6cAyoxPhVdDonlSQ==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/utils-common/-/utils-common-3.6.2.tgz",
+      "integrity": "sha512-dr5wK+OyU2QAWxG7S5siD2bPgS7+ZeqWHfgLNHZ5yalaZf8TbeNNLqydfngukAY56BGZN0NbMkX6jGIr7ZF0sA==",
      "dependencies": {
-        "@docusaurus/types": "3.6.3",
+        "@docusaurus/types": "3.6.2",
        "tslib": "^2.6.0"
      },
      "engines": {
@@ -3802,13 +3802,13 @@
      }
    },
    "node_modules/@docusaurus/utils-validation": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/utils-validation/-/utils-validation-3.6.3.tgz",
-      "integrity": "sha512-bhEGGiN5BE38h21vjqD70Gxg++j+PfYVddDUE5UFvLDup68QOcpD33CLr+2knPorlxRbEaNfz6HQDUMQ3HuqKw==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/utils-validation/-/utils-validation-3.6.2.tgz",
+      "integrity": "sha512-Y3EwblDz72KOcobb5t2zlhHSmrfE8EaHusPJ96Kx2JYtNXL2omqCoOb6FpaXWhES75wvjUpkFLYfiNqAqEov8g==",
      "dependencies": {
-        "@docusaurus/logger": "3.6.3",
-        "@docusaurus/utils": "3.6.3",
-        "@docusaurus/utils-common": "3.6.3",
+        "@docusaurus/logger": "3.6.2",
+        "@docusaurus/utils": "3.6.2",
+        "@docusaurus/utils-common": "3.6.2",
        "fs-extra": "^11.2.0",
        "joi": "^17.9.2",
        "js-yaml": "^4.1.0",
@@ -15155,10 +15155,9 @@
      }
    },
    "node_modules/react-icons": {
-      "version": "5.4.0",
-      "resolved": "https://registry.npmjs.org/react-icons/-/react-icons-5.4.0.tgz",
-      "integrity": "sha512-7eltJxgVt7X64oHh6wSWNwwbKTCtMfK35hcjvJS0yxEAhPM8oUKdS3+kqaW1vicIltw+kR2unHaa12S9pPALoQ==",
-      "license": "MIT",
+      "version": "5.3.0",
+      "resolved": "https://registry.npmjs.org/react-icons/-/react-icons-5.3.0.tgz",
+      "integrity": "sha512-DnUk8aFbTyQPSkCfF8dbX6kQjXA9DktMeJqfjrg6cK9vwQVMxmcA3BfP4QoiztVmEHtwlTgLFsPuH2NskKT6eg==",
      "peerDependencies": {
        "react": "*"
      }
@@ -16986,10 +16985,9 @@
      }
    },
    "node_modules/typescript": {
-      "version": "5.7.2",
-      "resolved": "https://registry.npmjs.org/typescript/-/typescript-5.7.2.tgz",
-      "integrity": "sha512-i5t66RHxDvVN40HfDd1PsEThGNnlMCMT3jMUuoh9/0TaqWevNontacunWyN02LA9/fIbEWlcHZcgTKb9QoaLfg==",
-      "license": "Apache-2.0",
+      "version": "5.6.3",
+      "resolved": "https://registry.npmjs.org/typescript/-/typescript-5.6.3.tgz",
+      "integrity": "sha512-hjcS1mhfuyi4WW8IWtjP7brDrG2cuDZukyrYrSauoXGNgx0S7zceP07adYkJycEr56BOUTNPzbInooiN3fn1qw==",
      "bin": {
        "tsc": "bin/tsc",
        "tsserver": "bin/tsserver"
--- a/docs/package.json
+++ b/docs/package.json
@@ -15,23 +15,23 @@
    "typecheck": "tsc"
  },
  "dependencies": {
-    "@docusaurus/core": "^3.6.3",
-    "@docusaurus/plugin-content-pages": "^3.6.3",
-    "@docusaurus/preset-classic": "^3.6.3",
-    "@docusaurus/theme-mermaid": "^3.6.3",
+    "@docusaurus/core": "^3.6.2",
+    "@docusaurus/plugin-content-pages": "^3.6.2",
+    "@docusaurus/preset-classic": "^3.6.2",
+    "@docusaurus/theme-mermaid": "^3.6.2",
    "@mdx-js/react": "^3.1.0",
    "clsx": "^2.0.0",
    "prism-react-renderer": "^2.4.0",
    "react": "^18.3.1",
    "react-dom": "^18.3.1",
-    "react-icons": "^5.4.0",
+    "react-icons": "^5.3.0",
    "react-use": "^17.5.1"
  },
  "devDependencies": {
    "@docusaurus/module-type-aliases": "^3.5.1",
-    "@docusaurus/tsconfig": "^3.6.3",
+    "@docusaurus/tsconfig": "^3.6.2",
    "@docusaurus/types": "^3.5.1",
-    "typescript": "~5.7.2"
+    "typescript": "~5.6.3"
  },
  "browserslist": {
    "production": [
--- a/docs/sidebars.ts
+++ b/docs/sidebars.ts
@@ -100,11 +100,6 @@ const sidebars: SidebarsConfig = {
          label: 'Runtime Configuration',
          id: 'usage/runtimes',
        },
-        {
-          type: 'doc',
-          label: 'Configuration Options',
-          id: 'usage/configuration-options',
-        },
        {
          type: 'doc',
          label: 'Custom Sandbox',
--- a/docs/src/components/CustomFooter.tsx
+++ b/docs/src/components/CustomFooter.tsx
@@ -8,7 +8,7 @@ function CustomFooter() {
    <footer className="custom-footer">
      <div className="footer-content">
        <div className="footer-icons">
-          <a href="https://join.slack.com/t/openhands-ai/shared_invite/zt-2vbfigwev-G03twSpXaErwzYVD4CFiBg" target="_blank" rel="noopener noreferrer">
+          <a href="https://join.slack.com/t/opendevin/shared_invite/zt-2oikve2hu-UDxHeo8nsE69y6T7yFX_BA" target="_blank" rel="noopener noreferrer">
            <FaSlack />
          </a>
          <a href="https://discord.gg/ESHStjSjD4" target="_blank" rel="noopener noreferrer">
--- a/docs/src/components/HomepageHeader/HomepageHeader.tsx
+++ b/docs/src/components/HomepageHeader/HomepageHeader.tsx
@@ -23,7 +23,7 @@ export function HomepageHeader() {
          <a href="https://codecov.io/github/All-Hands-AI/OpenHands?branch=main"><img alt="CodeCov" src="https://img.shields.io/codecov/c/github/All-Hands-AI/OpenHands?style=for-the-badge&color=blue" /></a>
          <a href="https://github.com/All-Hands-AI/OpenHands/blob/main/LICENSE"><img src="https://img.shields.io/github/license/All-Hands-AI/OpenHands?style=for-the-badge&color=blue" alt="MIT License" /></a>
          <br/>
-          <a href="https://join.slack.com/t/openhands-ai/shared_invite/zt-2vbfigwev-G03twSpXaErwzYVD4CFiBg"><img src="https://img.shields.io/badge/Slack-Join%20Us-red?logo=slack&logoColor=white&style=for-the-badge" alt="Join our Slack community" /></a>
+          <a href="https://join.slack.com/t/opendevin/shared_invite/zt-2oikve2hu-UDxHeo8nsE69y6T7yFX_BA"><img src="https://img.shields.io/badge/Slack-Join%20Us-red?logo=slack&logoColor=white&style=for-the-badge" alt="Join our Slack community" /></a>
          <a href="https://discord.gg/ESHStjSjD4"><img src="https://img.shields.io/badge/Discord-Join%20Us-purple?logo=discord&logoColor=white&style=for-the-badge" alt="Join our Discord community" /></a>
          <a href="https://github.com/All-Hands-AI/OpenHands/blob/main/CREDITS.md"><img src="https://img.shields.io/badge/Project-Credits-blue?style=for-the-badge&color=FFE165&logo=github&logoColor=white" alt="Credits" /></a>
          <br/>
--- a/docs/static/img/teaser.mp4
+++ b/docs/static/img/teaser.mp4
--- a/docs/yarn.lock
+++ b/docs/yarn.lock
--- a/evaluation/benchmarks/EDA/README.md
+++ b/evaluation/benchmarks/EDA/README.md
@@ -4,13 +4,15 @@ This folder contains evaluation harness for evaluating agents on the Entity-dedu

 ## Setup Environment and LLM Configuration

-Please follow instruction [here](../../README.md#setup) to setup your local development environment and LLM.
+Please follow instruction [here](../README.md#setup) to setup your local development environment and LLM.
+

 ## Start the evaluation

+
 ```bash
 export OPENAI_API_KEY="sk-XXX"; # This is required for evaluation (to simulate another party of conversation)
-./evaluation/benchmarks/EDA/scripts/run_infer.sh [model_config] [git-version] [agent] [dataset] [eval_limit]
+./evaluation/EDA/scripts/run_infer.sh [model_config] [git-version] [agent] [dataset] [eval_limit]
 ```

 where `model_config` is mandatory, while `git-version`, `agent`, `dataset` and `eval_limit` are optional.
@@ -31,12 +33,11 @@ to `CodeActAgent`.
 For example,

 ```bash
-./evaluation/benchmarks/EDA/scripts/run_infer.sh eval_gpt4o_2024_05_13 0.6.2 CodeActAgent things
+./evaluation/EDA/scripts/run_infer.sh eval_gpt4o_2024_05_13 0.6.2 CodeActAgent things
 ```

 ## Reference
-
-```bibtex
+```
@inproceedings{zhang2023entity,
  title={Probing the Multi-turn Planning Capabilities of LLMs via 20 Question Games},
  author={Zhang, Yizhe and Lu, Jiarui and Jaitly, Navdeep},
--- a/evaluation/benchmarks/EDA/game.py
+++ b/evaluation/benchmarks/EDA/game.py
--- a/evaluation/benchmarks/EDA/run_infer.py
+++ b/evaluation/benchmarks/EDA/run_infer.py
@@ -4,7 +4,7 @@ import os
 import pandas as pd
 from datasets import load_dataset

-from evaluation.benchmarks.EDA.game import Q20Game, Q20GameCelebrity
+from evaluation.EDA.game import Q20Game, Q20GameCelebrity
 from evaluation.utils.shared import (
    EvalMetadata,
    EvalOutput,
--- a/evaluation/benchmarks/EDA/scripts/run_infer.sh
+++ b/evaluation/benchmarks/EDA/scripts/run_infer.sh
@@ -21,7 +21,7 @@ if [ -z "$AGENT" ]; then
  AGENT="CodeActAgent"
 fi

-get_openhands_version
+get_agent_version

 if [ -z "$DATASET" ]; then
  echo "Dataset not specified, use default 'things'"
@@ -34,13 +34,16 @@ if [ -z "$OPENAI_API_KEY" ]; then
  exit 1
 fi

+# IMPORTANT: Because Agent's prompt changes fairly often in the rapidly evolving codebase of OpenHands
+# We need to track the version of Agent in the evaluation to make sure results are comparable
+AGENT_VERSION=v$(poetry run python -c "import openhands.agenthub; from openhands.controller.agent import Agent; print(Agent.get_cls('$AGENT').VERSION)")

 echo "AGENT: $AGENT"
-echo "OPENHANDS_VERSION: $OPENHANDS_VERSION"
+echo "AGENT_VERSION: $AGENT_VERSION"
 echo "MODEL_CONFIG: $MODEL_CONFIG"
 echo "DATASET: $DATASET"

-COMMAND="poetry run python evaluation/benchmarks/EDA/run_infer.py \
+COMMAND="poetry run python evaluation/EDA/run_infer.py \
  --agent-cls $AGENT \
  --llm-config $MODEL_CONFIG \
  --dataset $DATASET \
@@ -48,7 +51,7 @@ COMMAND="poetry run python evaluation/benchmarks/EDA/run_infer.py \
  --max-iterations 20 \
  --OPENAI_API_KEY $OPENAI_API_KEY \
  --eval-num-workers $NUM_WORKERS \
-  --eval-note ${OPENHANDS_VERSION}_${DATASET}"
+  --eval-note ${AGENT_VERSION}_${DATASET}"

 if [ -n "$EVAL_LIMIT" ]; then
  echo "EVAL_LIMIT: $EVAL_LIMIT"
--- a/evaluation/README.md
+++ b/evaluation/README.md
@@ -6,9 +6,9 @@ This folder contains code and resources to run experiments and evaluations.

 ### Setup

-Before starting evaluation, follow the instructions [here](https://github.com/All-Hands-AI/OpenHands/blob/main/Development.md) to setup your local development environment and LLM.
+Before starting evaluation, follow the instructions here [here](https://github.com/All-Hands-AI/OpenHands/blob/main/Development.md) to setup your local development environment and LLM.

-Once you are done with setup, you can follow the benchmark-specific instructions in each subdirectory of the [evaluation directory](#supported-benchmarks).
+Once you are done with setup, you can follow the benchmark-specific instructions in each subdirectory of the evaluation directory.
 Generally these will involve running `run_infer.py` to perform inference with the agents.

 ### Implementing and Evaluating an Agent
@@ -42,36 +42,32 @@ temperature = 0.0

 ## Supported Benchmarks

-The OpenHands evaluation harness supports a wide variety of benchmarks across [software engineering](#software-engineering), [web browsing](#web-browsing), and [miscellaneous assistance](#misc-assistance) tasks.
+The OpenHands evaluation harness supports a wide variety of benchmarks across software engineering, web browsing, and miscellaneous assistance tasks.

 ### Software Engineering

- SWE-Bench: [`evaluation/benchmarks/swe_bench`](./benchmarks/swe_bench)
- HumanEvalFix: [`evaluation/benchmarks/humanevalfix`](./benchmarks/humanevalfix)
- BIRD: [`evaluation/benchmarks/bird`](./benchmarks/bird)
- BioCoder: [`evaluation/benchmarks/ml_bench`](./benchmarks/ml_bench)
- ML-Bench: [`evaluation/benchmarks/ml_bench`](./benchmarks/ml_bench)
- APIBench: [`evaluation/benchmarks/gorilla`](./benchmarks/gorilla/)
- ToolQA: [`evaluation/benchmarks/toolqa`](./benchmarks/toolqa/)
- AiderBench: [`evaluation/benchmarks/aider_bench`](./benchmarks/aider_bench/)
- Commit0: [`evaluation/benchmarks/commit0_bench`](./benchmarks/commit0_bench/)
- DiscoveryBench: [`evaluation/benchmarks/discoverybench`](./benchmarks/discoverybench/)
+- SWE-Bench: [`evaluation/swe_bench`](./swe_bench)
+- HumanEvalFix: [`evaluation/humanevalfix`](./humanevalfix)
+- BIRD: [`evaluation/bird`](./bird)
+- BioCoder: [`evaluation/ml_bench`](./ml_bench)
+- ML-Bench: [`evaluation/ml_bench`](./ml_bench)
+- APIBench: [`evaluation/gorilla`](./gorilla/)
+- ToolQA: [`evaluation/toolqa`](./toolqa/)
+- AiderBench: [`evaluation/aider_bench`](./aider_bench/)

 ### Web Browsing

- WebArena: [`evaluation/benchmarks/webarena`](./benchmarks/webarena/)
- MiniWob++: [`evaluation/benchmarks/miniwob`](./benchmarks/miniwob/)
- Browsing Delegation: [`evaluation/benchmarks/browsing_delegation`](./benchmarks/browsing_delegation/)
+- WebArena: [`evaluation/webarena`](./webarena/)
+- MiniWob++: [`evaluation/miniwob`](./miniwob/)

 ### Misc. Assistance

- GAIA: [`evaluation/benchmarks/gaia`](./benchmarks/gaia)
- GPQA: [`evaluation/benchmarks/gpqa`](./benchmarks/gpqa)
- AgentBench: [`evaluation/benchmarks/agent_bench`](./benchmarks/agent_bench)
- MINT: [`evaluation/benchmarks/mint`](./benchmarks/mint)
- Entity deduction Arena (EDA): [`evaluation/benchmarks/EDA`](./benchmarks/EDA)
- ProofWriter: [`evaluation/benchmarks/logic_reasoning`](./benchmarks/logic_reasoning)
- ScienceAgentBench: [`evaluation/benchmarks/scienceagentbench`](./benchmarks/scienceagentbench)
+- GAIA: [`evaluation/gaia`](./gaia)
+- GPQA: [`evaluation/gpqa`](./gpqa)
+- AgentBench: [`evaluation/agent_bench`](./agent_bench)
+- MINT: [`evaluation/mint`](./mint)
+- Entity deduction Arena (EDA): [`evaluation/EDA`](./EDA)
+- ProofWriter: [`evaluation/logic_reasoning`](./logic_reasoning)

 ## Result Visualization

@@ -83,7 +79,7 @@ You can start your own fork of [our huggingface evaluation outputs](https://hugg

 To learn more about how to integrate your benchmark into OpenHands, check out [tutorial here](https://docs.all-hands.dev/modules/usage/how-to/evaluation-harness). Briefly,

- Each subfolder contains a specific benchmark or experiment. For example, [`evaluation/benchmarks/swe_bench`](./benchmarks/swe_bench) should contain
+- Each subfolder contains a specific benchmark or experiment. For example, `evaluation/swe_bench` should contain
 all the preprocessing/evaluation/analysis scripts.
 - Raw data and experimental records should not be stored within this repo.
 - For model outputs, they should be stored at [this huggingface space](https://huggingface.co/spaces/OpenHands/evaluation) for visualization.
--- a/evaluation/benchmarks/agent_bench/README.md
+++ b/evaluation/benchmarks/agent_bench/README.md
@@ -4,12 +4,12 @@ This folder contains evaluation harness for evaluating agents on the [AgentBench

 ## Setup Environment and LLM Configuration

-Please follow instruction [here](../../README.md#setup) to setup your local development environment and LLM.
+Please follow instruction [here](../README.md#setup) to setup your local development environment and LLM.

 ## Start the evaluation

 ```bash
-./evaluation/benchmarks/agent_bench/scripts/run_infer.sh [model_config] [git-version] [agent] [eval_limit]
+./evaluation/agent_bench/scripts/run_infer.sh [model_config] [git-version] [agent] [eval_limit]
 ```

 - `model_config`, e.g. `eval_gpt4_1106_preview`, is the config group name for your
@@ -25,7 +25,7 @@ in order to use `eval_limit`, you must also set `agent`.

 Following is the basic command to start the evaluation.

-You can update the arguments in the script `evaluation/benchmarks/agent_bench/scripts/run_infer.sh`, such as `--max-iterations`, `--eval-num-workers` and so on.
+You can update the arguments in the script `evaluation/agent_bench/scripts/run_infer.sh`, such as `--max-iterations`, `--eval-num-workers` and so on.

 - `--agent-cls`, the agent to use. For example, `CodeActAgent`.
 - `--llm-config`: the LLM configuration to use. For example, `eval_gpt4_1106_preview`.
@@ -34,23 +34,5 @@ You can update the arguments in the script `evaluation/benchmarks/agent_bench/sc
 - `--eval-n-limit`: the number of examples to evaluate. For example, `100`.

 ```bash
-./evaluation/benchmarks/agent_bench/scripts/run_infer.sh eval_gpt35_turbo HEAD CodeActAgent 1
+./evaluation/agent_bench/scripts/run_infer.sh eval_gpt35_turbo HEAD CodeActAgent 1
 ```
-
-## Run with Remote Runtime (experimental)
-
-You can run the evaluation using a remote runtime instead of a local Docker container. This is useful when you want to run the evaluation in a cloud environment or when you don't have Docker installed locally.
-
-To use the remote runtime, set the following environment variables:
-
-```bash
-# Required environment variables
-export ALLHANDS_API_KEY="your-api-key"  # Contact the team to get an API key
-export RUNTIME=remote
-export SANDBOX_REMOTE_RUNTIME_API_URL="https://runtime.eval.all-hands.dev"
-
-# Run the evaluation
-./evaluation/benchmarks/agent_bench/scripts/run_infer.sh llm.eval_gpt4_1106_preview HEAD CodeActAgent 1
-```
-
-The remote runtime will build a container image and run the evaluation in a cloud environment. The results will be saved locally in the same way as when running with a local runtime.
--- a/evaluation/benchmarks/agent_bench/init.py
+++ b/evaluation/benchmarks/agent_bench/init.py
--- a/evaluation/benchmarks/agent_bench/helper.py
+++ b/evaluation/benchmarks/agent_bench/helper.py
--- a/evaluation/benchmarks/agent_bench/run_infer.py
+++ b/evaluation/benchmarks/agent_bench/run_infer.py
@@ -7,7 +7,7 @@ from typing import Any
 import pandas as pd
 from datasets import load_dataset

-from evaluation.benchmarks.agent_bench.helper import (
+from evaluation.agent_bench.helper import (
    FAKE_RESPONSES,
    INST_SUFFIXES,
    compare_results,
@@ -43,16 +43,12 @@ def get_config(
    config = AppConfig(
        default_agent=metadata.agent_class,
        run_as_openhands=False,
-        runtime=os.environ.get('RUNTIME', 'eventstream'),
+        runtime='eventstream',
        max_iterations=metadata.max_iterations,
        sandbox=SandboxConfig(
-            base_container_image='python:3.12-slim',
+            base_container_image='python:3.12-bookworm',
            enable_auto_lint=True,
            use_host_network=False,
-            api_key=os.environ.get('ALLHANDS_API_KEY', None),
-            remote_runtime_api_url=os.environ.get('SANDBOX_REMOTE_RUNTIME_API_URL'),
-            keep_runtime_alive=False,
-            remote_runtime_init_timeout=3600,
        ),
        # do not mount workspace
        workspace_base=None,
--- a/evaluation/benchmarks/agent_bench/scripts/run_infer.sh
+++ b/evaluation/benchmarks/agent_bench/scripts/run_infer.sh
@@ -20,18 +20,18 @@ if [ -z "$AGENT" ]; then
  AGENT="CodeActAgent"
 fi

-get_openhands_version
+get_agent_version

 echo "AGENT: $AGENT"
-echo "OPENHANDS_VERSION: $OPENHANDS_VERSION"
+echo "AGENT_VERSION: $AGENT_VERSION"
 echo "MODEL_CONFIG: $MODEL_CONFIG"

-COMMAND="export PYTHONPATH=evaluation/benchmarks/agent_bench:\$PYTHONPATH && poetry run python evaluation/benchmarks/agent_bench/run_infer.py \
+COMMAND="export PYTHONPATH=evaluation/agent_bench:\$PYTHONPATH && poetry run python evaluation/agent_bench/run_infer.py \
  --agent-cls $AGENT \
  --llm-config $MODEL_CONFIG \
  --max-iterations 30 \
  --eval-num-workers $NUM_WORKERS \
-  --eval-note $OPENHANDS_VERSION"
+  --eval-note $AGENT_VERSION"

 if [ -n "$EVAL_LIMIT" ]; then
  echo "EVAL_LIMIT: $EVAL_LIMIT"
--- a/evaluation/benchmarks/agent_bench/scripts/summarise_results.py
+++ b/evaluation/benchmarks/agent_bench/scripts/summarise_results.py
--- a/evaluation/benchmarks/aider_bench/README.md
+++ b/evaluation/benchmarks/aider_bench/README.md
@@ -10,13 +10,13 @@ Hugging Face dataset based on the

 ## Setup Environment and LLM Configuration

-Please follow instruction [here](../../README.md#setup) to setup your local
+Please follow instruction [here](../README.md#setup) to setup your local
 development environment and LLM.

 ## Start the evaluation

 ```bash
-./evaluation/benchmarks/aider_bench/scripts/run_infer.sh [model_config] [git-version] [agent] [eval_limit] [eval-num-workers] [eval_ids]
+./evaluation/aider_bench/scripts/run_infer.sh [model_config] [git-version] [agent] [eval_limit] [eval-num-workers] [eval_ids]
 ```

 - `model_config`, e.g. `eval_gpt4_1106_preview`, is the config group name for
@@ -42,7 +42,7 @@ export SKIP_NUM=12 # skip the first 12 instances from the dataset
 Following is the basic command to start the evaluation.

 You can update the arguments in the script
-`evaluation/benchmarks/aider_bench/scripts/run_infer.sh`, such as `--max-iterations`,
+`evaluation/aider_bench/scripts/run_infer.sh`, such as `--max-iterations`,
 `--eval-num-workers` and so on:

 - `--agent-cls`, the agent to use. For example, `CodeActAgent`.
@@ -53,7 +53,7 @@ You can update the arguments in the script
 - `--eval-ids`: the IDs of the examples to evaluate (comma separated). For example, `"1,3,10"`.

 ```bash
-./evaluation/benchmarks/aider_bench/scripts/run_infer.sh eval_gpt35_turbo HEAD CodeActAgent 100 1 "1,3,10"
+./evaluation/aider_bench/scripts/run_infer.sh eval_gpt35_turbo HEAD CodeActAgent 100 1 "1,3,10"
 ```

 ### Run Inference on `RemoteRuntime` (experimental)
@@ -61,25 +61,25 @@ You can update the arguments in the script
 This is in limited beta. Contact Xingyao over slack if you want to try this out!

 ```bash
-./evaluation/benchmarks/aider_bench/scripts/run_infer.sh [model_config] [git-version] [agent] [eval_limit] [eval-num-workers] [eval_ids]
+./evaluation/aider_bench/scripts/run_infer.sh [model_config] [git-version] [agent] [eval_limit] [eval-num-workers] [eval_ids]

 # Example - This runs evaluation on CodeActAgent for 133 instances on aider_bench test set, with 2 workers running in parallel
 export ALLHANDS_API_KEY="YOUR-API-KEY"
 export RUNTIME=remote
 export SANDBOX_REMOTE_RUNTIME_API_URL="https://runtime.eval.all-hands.dev"
-./evaluation/benchmarks/aider_bench/scripts/run_infer.sh llm.eval HEAD CodeActAgent 133 2
+./evaluation/aider_bench/scripts/run_infer.sh llm.eval HEAD CodeActAgent 133 2
 ```

 ## Summarize Results

 ```bash
-poetry run python ./evaluation/benchmarks/aider_bench/scripts/summarize_results.py [path_to_output_jsonl_file]
+poetry run python ./evaluation/aider_bench/scripts/summarize_results.py [path_to_output_jsonl_file]
 ```

 Full example:

 ```bash
-poetry run python ./evaluation/benchmarks/aider_bench/scripts/summarize_results.py evaluation/evaluation_outputs/outputs/AiderBench/CodeActAgent/claude-3-5-sonnet@20240620_maxiter_30_N_v1.9/output.jsonl
+poetry run python ./evaluation/aider_bench/scripts/summarize_results.py evaluation/evaluation_outputs/outputs/AiderBench/CodeActAgent/claude-3-5-sonnet@20240620_maxiter_30_N_v1.9/output.jsonl
 ```

 This will list the instances that passed and the instances that failed. For each
--- a/evaluation/benchmarks/aider_bench/create_dataset.py
+++ b/evaluation/benchmarks/aider_bench/create_dataset.py
--- a/evaluation/benchmarks/aider_bench/helper.py
+++ b/evaluation/benchmarks/aider_bench/helper.py
--- a/evaluation/benchmarks/aider_bench/run_infer.py
+++ b/evaluation/benchmarks/aider_bench/run_infer.py
@@ -7,7 +7,7 @@ from typing import Any
 import pandas as pd
 from datasets import load_dataset

-from evaluation.benchmarks.aider_bench.helper import (
+from evaluation.aider_bench.helper import (
    FAKE_RESPONSES,
    INST_SUFFIXES,
    INSTRUCTIONS_ADDENDUM,
--- a/evaluation/benchmarks/aider_bench/scripts/run_infer.sh
+++ b/evaluation/benchmarks/aider_bench/scripts/run_infer.sh
@@ -21,13 +21,13 @@ if [ -z "$AGENT" ]; then
  AGENT="CodeActAgent"
 fi

-get_openhands_version
+get_agent_version

 echo "AGENT: $AGENT"
-echo "OPENHANDS_VERSION: $OPENHANDS_VERSION"
+echo "AGENT_VERSION: $AGENT_VERSION"
 echo "MODEL_CONFIG: $MODEL_CONFIG"

-EVAL_NOTE=$OPENHANDS_VERSION
+EVAL_NOTE=$AGENT_VERSION

 # Default to NOT use unit tests.
 if [ -z "$USE_UNIT_TESTS" ]; then
@@ -39,7 +39,7 @@ if [ "$USE_UNIT_TESTS" = true ]; then
  EVAL_NOTE=$EVAL_NOTE-w-test
 fi

-COMMAND="export PYTHONPATH=evaluation/benchmarks/aider_bench:\$PYTHONPATH && poetry run python evaluation/benchmarks/aider_bench/run_infer.py \
+COMMAND="export PYTHONPATH=evaluation/aider_bench:\$PYTHONPATH && poetry run python evaluation/aider_bench/run_infer.py \
  --agent-cls $AGENT \
  --llm-config $MODEL_CONFIG \
  --max-iterations 30 \
--- a/evaluation/benchmarks/aider_bench/scripts/summarize_results.py
+++ b/evaluation/benchmarks/aider_bench/scripts/summarize_results.py
--- a/evaluation/benchmarks/swe_bench/scripts/eval/summarize_outputs.py
+++ b/evaluation/benchmarks/swe_bench/scripts/eval/summarize_outputs.py
@@ -1,279 +0,0 @@
-#!/usr/bin/env python3
-import argparse
-import glob
-import json
-import os
-from collections import Counter
-
-import pandas as pd
-
-from openhands.events.serialization import event_from_dict
-from openhands.events.utils import get_pairs_from_events
-
-ERROR_KEYWORDS = [
-    'Agent encountered an error while processing the last action',
-    'APIError',
-    'Action execution failed',
-    'litellm.Timeout: APITimeoutError',
-]
-
-
-def process_file(file_path):
-    with open(file_path, 'r') as file:
-        lines = file.readlines()
-
-    num_lines = len(lines)
-    num_error_lines = 0
-    num_agent_stuck_in_loop = 0
-    num_resolved = 0
-    num_empty_patch = 0
-    num_unfinished_runs = 0
-    error_counter = Counter()
-    main_agent_cost = []
-    editor_cost = []
-    num_turns = []
-
-    for line in lines:
-        _d = json.loads(line)
-
-        if 'metrics' not in _d or _d['metrics'] is None:
-            # this is a failed run
-            num_unfinished_runs += 1
-            continue
-
-        # Cost
-        costs = _d['metrics'].get('costs', [])
-        _cur_main_agent_cost = 0
-        _cur_editor_cost = 0
-        for cost in costs:
-            if isinstance(cost, float):
-                # backward compatible
-                _cur_main_agent_cost += cost
-            else:
-                if 'draft_editor' in cost['model']:
-                    _cur_editor_cost += cost['cost']
-                else:
-                    _cur_main_agent_cost += cost['cost']
-
-        main_agent_cost.append(_cur_main_agent_cost)
-        editor_cost.append(_cur_editor_cost)
-
-        # Turn status
-        history = _d.get('history', [])
-        events = [event_from_dict(event) for event in history]
-        pairs = get_pairs_from_events(events)
-        num_turns.append(len(pairs))
-
-        # Patch & resolve status
-        patch = _d.get('test_result', {}).get('git_patch', '')
-        if patch == '':
-            num_empty_patch += 1
-            continue
-
-        report = _d.get('report', {}) or {}
-        resolved = report.get('resolved', False)
-        if resolved:
-            num_resolved += 1
-
-        # Error
-        error = _d.get('error', None)
-
-        if error is not None and isinstance(error, str):
-            agent_stuck_in_loop = 'Agent got stuck in a loop' in error
-            contains_error = bool(error) and not agent_stuck_in_loop
-            if agent_stuck_in_loop:
-                error_counter['Agent got stuck in a loop'] += 1
-                num_agent_stuck_in_loop += 1
-            elif contains_error:
-                error_counter[error] += 1
-            continue
-
-        for keyword in ERROR_KEYWORDS:
-            if keyword in line:
-                error_counter[keyword] += 1
-                num_error_lines += 1
-                break
-
-    return {
-        'file_path': file_path,
-        'total_instances': num_lines,
-        'resolved': {
-            'count': num_resolved,
-            'percentage': (num_resolved / num_lines * 100) if num_lines > 0 else 0,
-        },
-        'empty_patches': {
-            'count': num_empty_patch,
-            'percentage': (num_empty_patch / num_lines * 100) if num_lines > 0 else 0,
-        },
-        'unfinished_runs': {
-            'count': num_unfinished_runs,
-            'percentage': (num_unfinished_runs / num_lines * 100)
-            if num_lines > 0
-            else 0,
-        },
-        'errors': {
-            'total': num_error_lines,
-            'percentage': (num_error_lines / num_lines * 100) if num_lines > 0 else 0,
-            'stuck_in_loop': {
-                'count': num_agent_stuck_in_loop,
-                'percentage': (num_agent_stuck_in_loop / num_lines * 100)
-                if num_lines > 0
-                else 0,
-            },
-            'breakdown': {
-                str(error): {
-                    'count': count,
-                    'percentage': (count / num_lines * 100) if num_lines > 0 else 0,
-                }
-                for error, count in error_counter.items()
-            },
-        },
-        'costs': {
-            'main_agent': sum(main_agent_cost),
-            'editor': sum(editor_cost),
-            'total': sum(main_agent_cost) + sum(editor_cost),
-        },
-        'statistics': {
-            'avg_turns': sum(num_turns) / num_lines if num_lines > 0 else 0,
-            'costs': {
-                'main_agent': sum(main_agent_cost) / num_lines if num_lines > 0 else 0,
-                'editor': sum(editor_cost) / num_lines if num_lines > 0 else 0,
-                'total': (sum(main_agent_cost) + sum(editor_cost)) / num_lines
-                if num_lines > 0
-                else 0,
-            },
-        },
-    }
-
-
-def aggregate_directory(input_path) -> pd.DataFrame:
-    # Process all output.jsonl files in subdirectories
-    pattern = os.path.join(input_path, '**/output.jsonl')
-    files = glob.glob(pattern, recursive=True)
-    print(f'Processing {len(files)} files from directory {input_path}')
-
-    # Process each file silently and collect results
-    results = []
-    for file_path in files:
-        try:
-            result = process_file(file_path)
-            results.append(result)
-        except Exception as e:
-            print(f'Error processing {file_path}: {str(e)}')
-            import traceback
-
-            traceback.print_exc()
-            continue
-
-    # Convert results to pandas DataFrame and sort by resolve rate
-    df = pd.DataFrame(results)
-
-    # Extract directory name from file path
-    df['directory'] = df['file_path'].apply(
-        lambda x: os.path.basename(os.path.dirname(x))
-    )
-
-    df['resolve_rate'] = df['resolved'].apply(lambda x: x['percentage'])
-    df['empty_patch_rate'] = df['empty_patches'].apply(lambda x: x['percentage'])
-    df['unfinished_rate'] = df['unfinished_runs'].apply(lambda x: x['percentage'])
-    df['avg_turns'] = df['statistics'].apply(lambda x: x['avg_turns'])
-    df['error_rate'] = df['errors'].apply(lambda x: x['percentage'])
-    df['avg_cost'] = df['statistics'].apply(lambda x: x['costs']['total'])
-
-    df = df.sort_values('resolve_rate', ascending=False)
-
-    return df
-
-
-if __name__ == '__main__':
-    parser = argparse.ArgumentParser()
-    parser.add_argument(
-        'input_path', type=str, help='The file or directory to summarize'
-    )
-    parser.add_argument(
-        '--output',
-        type=str,
-        help='Output JSONL file for results',
-        default='summary_results.jsonl',
-    )
-    args = parser.parse_args()
-
-    if os.path.isdir(args.input_path):
-        df = aggregate_directory(args.input_path)
-        # Create the summary string
-        columns = [
-            'directory',
-            'resolve_rate',
-            'empty_patch_rate',
-            'unfinished_rate',
-            'error_rate',
-            'avg_turns',
-            'avg_cost',
-            'total_instances',
-        ]
-        summary_str = df[columns].to_string(
-            float_format=lambda x: '{:.2f}'.format(x),
-            formatters={
-                'directory': lambda x: x[:90]
-            },  # Truncate directory names to 20 chars
-            index=False,
-        )
-
-        # Print to console
-        print('\nResults summary (sorted by resolve rate):')
-        print(summary_str)
-
-        # Save to text file
-        txt_output = args.output.rsplit('.', 1)[0] + '.txt'
-        with open(txt_output, 'w') as f:
-            f.write('Results summary (sorted by resolve rate):\n')
-            f.write(summary_str)
-
-        # Save
-        df.to_json(args.output, lines=True, orient='records')
-        df[columns].to_csv(args.output.rsplit('.', 1)[0] + '.csv', index=False)
-    else:
-        # Process single file with detailed output
-        results = []
-        try:
-            result = process_file(args.input_path)
-            results.append(result)
-
-            # Print detailed results for single file
-            print(f'\nResults for {args.input_path}:')
-            print(
-                f"Number of resolved: {result['resolved']['count']} / {result['total_instances']} ({result['resolved']['percentage']:.2f}%)"
-            )
-            print(
-                f"Number of empty patch: {result['empty_patches']['count']} / {result['total_instances']} ({result['empty_patches']['percentage']:.2f}%)"
-            )
-            print(
-                f"Number of error lines: {result['errors']['total']} / {result['total_instances']} ({result['errors']['percentage']:.2f}%)"
-            )
-            print(
-                f"Number of agent stuck in loop: {result['errors']['stuck_in_loop']['count']} / {result['total_instances']} ({result['errors']['stuck_in_loop']['percentage']:.2f}%)"
-            )
-            print(
-                f"Number of unfinished runs: {result['unfinished_runs']['count']} / {result['total_instances']} ({result['unfinished_runs']['percentage']:.2f}%)"
-            )
-            print(f"Total cost: {result['costs']['total']:.2f} USD")
-            print('## Statistics')
-            print(
-                f"Avg. num of turns per instance: {result['statistics']['avg_turns']:.2f}"
-            )
-            print(
-                f"Avg. agent cost per instance: {result['statistics']['costs']['main_agent']:.2f} USD"
-            )
-            print(
-                f"Avg. editor cost per instance: {result['statistics']['costs']['editor']:.2f} USD"
-            )
-            print(
-                f"Avg. total cost per instance: {result['statistics']['costs']['total']:.2f} USD"
-            )
-
-            print('## Detailed error breakdown:')
-            for error, data in result['errors']['breakdown'].items():
-                print(f"{error}: {data['count']} ({data['percentage']:.2f}%)")
-
-        except Exception as e:
-            print(f'Error processing {args.input_path}: {str(e)}')
--- a/evaluation/benchmarks/swe_bench/scripts/eval/verify_costs.py
+++ b/evaluation/benchmarks/swe_bench/scripts/eval/verify_costs.py
@@ -1,104 +0,0 @@
-import argparse
-
-import pandas as pd
-
-from openhands.core.logger import openhands_logger as logger
-
-
-def verify_instance_costs(row: pd.Series) -> float:
-    """
-    Verifies that the accumulated_cost matches the sum of individual costs in metrics.
-    Also checks for duplicate consecutive costs which might indicate buggy counting.
-    If the consecutive costs are identical, the file is affected by this bug:
-    https://github.com/All-Hands-AI/OpenHands/issues/5383
-
-    Args:
-        row: DataFrame row containing instance data with metrics
-    Returns:
-        float: The verified total cost for this instance (corrected if needed)
-    """
-    try:
-        metrics = row.get('metrics')
-        if not metrics:
-            logger.warning(f"Instance {row['instance_id']}: No metrics found")
-            return 0.0
-
-        accumulated = metrics.get('accumulated_cost')
-        costs = metrics.get('costs', [])
-
-        if accumulated is None:
-            logger.warning(
-                f"Instance {row['instance_id']}: No accumulated_cost in metrics"
-            )
-            return 0.0
-
-        # Check for duplicate consecutive costs and systematic even-odd pairs
-        has_duplicate = False
-        all_pairs_match = True
-
-        # Check each even-odd pair (0-1, 2-3, etc.)
-        for i in range(0, len(costs) - 1, 2):
-            if abs(costs[i]['cost'] - costs[i + 1]['cost']) < 1e-6:
-                has_duplicate = True
-                logger.debug(
-                    f"Instance {row['instance_id']}: Possible buggy double-counting detected! "
-                    f"Steps {i} and {i+1} have identical costs: {costs[i]['cost']:.2f}"
-                )
-            else:
-                all_pairs_match = False
-                break
-
-        # Calculate total cost, accounting for buggy double counting if detected
-        if len(costs) >= 2 and has_duplicate and all_pairs_match:
-            paired_steps_cost = sum(
-                cost_entry['cost']
-                for cost_entry in costs[: -1 if len(costs) % 2 else None]
-            )
-            real_paired_cost = paired_steps_cost / 2
-
-            unpaired_cost = costs[-1]['cost'] if len(costs) % 2 else 0
-            total_cost = real_paired_cost + unpaired_cost
-
-        else:
-            total_cost = sum(cost_entry['cost'] for cost_entry in costs)
-
-        if not abs(total_cost - accumulated) < 1e-6:
-            logger.warning(
-                f"Instance {row['instance_id']}: Cost mismatch: "
-                f"accumulated: {accumulated:.2f}, sum of costs: {total_cost:.2f}, "
-            )
-
-        return total_cost
-
-    except Exception as e:
-        logger.error(
-            f"Error verifying costs for instance {row.get('instance_id', 'UNKNOWN')}: {e}"
-        )
-        return 0.0
-
-
-def main():
-    parser = argparse.ArgumentParser(
-        description='Verify costs in SWE-bench output file'
-    )
-    parser.add_argument(
-        'input_filepath', type=str, help='Path to the output.jsonl file'
-    )
-    args = parser.parse_args()
-
-    try:
-        # Load and verify the JSONL file
-        df = pd.read_json(args.input_filepath, lines=True)
-        logger.info(f'Loaded {len(df)} instances from {args.input_filepath}')
-
-        # Verify costs for each instance and sum up total
-        total_cost = df.apply(verify_instance_costs, axis=1).sum()
-        logger.info(f'Total verified cost across all instances: ${total_cost:.2f}')
-
-    except Exception as e:
-        logger.error(f'Failed to process file: {e}')
-        raise
-
-
-if __name__ == '__main__':
-    main()
--- a/evaluation/benchmarks/biocoder/README.md
+++ b/evaluation/benchmarks/biocoder/README.md
@@ -4,14 +4,13 @@ Implements evaluation of agents on BioCoder from the BioCoder benchmark introduc

 ## Setup Environment and LLM Configuration

-Please follow instruction [here](../../README.md#setup) to setup your local development environment and LLM.
+Please follow instruction [here](../README.md#setup) to setup your local development environment and LLM.

 ## BioCoder Docker Image

 In the openhands branch of the Biocoder repository, we have slightly modified our original Docker image to work with the OpenHands environment. In the Docker image are testing scripts (`/testing/start_test_openhands.py` and aux files in `/testing_files/`) to assist with evaluation. Additionally, we have installed all dependencies, including OpenJDK, mamba (with Python 3.6), and many system libraries. Notably, we have **not** packaged all repositories into the image, so they are downloaded at runtime.

 **Before first execution, pull our Docker image with the following command**
-
 ```bash
 docker pull public.ecr.aws/i5g0m1f6/eval_biocoder:v1.0
 ```
@@ -20,8 +19,9 @@ To reproduce this image, please see the Dockerfile_Openopenhands in the `biocode

 ## Start the evaluation

+
 ```bash
-./evaluation/benchmarks/biocoder/scripts/run_infer.sh [model_config] [git-version] [agent] [eval_limit]
+./evaluation/biocoder/scripts/run_infer.sh [model_config] [git-version] [agent] [eval_limit]
 ```

 where `model_config` is mandatory, while `git-version`, `agent`, `dataset` and `eval_limit` are optional.
@@ -43,12 +43,11 @@ with current OpenHands version, then your command would be:
 ## Examples

 ```bash
-./evaluation/benchmarks/biocoder/scripts/run_infer.sh eval_gpt4o_2024_05_13 HEAD CodeActAgent 1
+./evaluation/biocoder/scripts/run_infer.sh eval_gpt4o_2024_05_13 HEAD CodeActAgent 1
 ```

 ## Reference
-
-```bibtex
+```
@misc{tang2024biocoder,
      title={BioCoder: A Benchmark for Bioinformatics Code Generation with Large Language Models},
      author={Xiangru Tang and Bill Qian and Rick Gao and Jiakang Chen and Xinyun Chen and Mark Gerstein},
--- a/evaluation/benchmarks/biocoder/run_infer.py
+++ b/evaluation/benchmarks/biocoder/run_infer.py
@@ -8,7 +8,7 @@ from typing import Any
 import pandas as pd
 from datasets import load_dataset

-from evaluation.benchmarks.biocoder.utils import BiocoderData
+from evaluation.biocoder.utils import BiocoderData
 from evaluation.utils.shared import (
    EvalMetadata,
    EvalOutput,
--- a/evaluation/benchmarks/biocoder/scripts/run_infer.sh
+++ b/evaluation/benchmarks/biocoder/scripts/run_infer.sh
@@ -21,19 +21,19 @@ if [ -z "$AGENT" ]; then
  AGENT="CodeActAgent"
 fi

-get_openhands_version
+get_agent_version

 echo "AGENT: $AGENT"
-echo "OPENHANDS_VERSION: $OPENHANDS_VERSION"
+echo "AGENT_VERSION: $AGENT_VERSION"
 echo "MODEL_CONFIG: $MODEL_CONFIG"
 echo "DATASET: $DATASET"

-COMMAND="poetry run python evaluation/benchmarks/biocoder/run_infer.py \
+COMMAND="poetry run python evaluation/biocoder/run_infer.py \
  --agent-cls $AGENT \
  --llm-config $MODEL_CONFIG \
  --max-iterations 10 \
  --eval-num-workers $NUM_WORKERS \
-  --eval-note ${OPENHANDS_VERSION}_${DATASET}"
+  --eval-note ${AGENT_VERSION}_${DATASET}"

 if [ -n "$EVAL_LIMIT" ]; then
  echo "EVAL_LIMIT: $EVAL_LIMIT"
--- a/evaluation/benchmarks/biocoder/scripts/setup/copy_changed_code.py
+++ b/evaluation/benchmarks/biocoder/scripts/setup/copy_changed_code.py
--- a/evaluation/benchmarks/biocoder/scripts/setup/remove_code.py
+++ b/evaluation/benchmarks/biocoder/scripts/setup/remove_code.py
--- a/evaluation/benchmarks/biocoder/utils.py
+++ b/evaluation/benchmarks/biocoder/utils.py
--- a/evaluation/benchmarks/bird/README.md
+++ b/evaluation/benchmarks/bird/README.md
--- a/evaluation/benchmarks/bird/init.py
+++ b/evaluation/benchmarks/bird/init.py
--- a/evaluation/benchmarks/bird/run_infer.py
+++ b/evaluation/benchmarks/bird/run_infer.py
--- a/evaluation/benchmarks/bird/scripts/run_infer.sh
+++ b/evaluation/benchmarks/bird/scripts/run_infer.sh
@@ -20,18 +20,18 @@ if [ -z "$AGENT" ]; then
  AGENT="CodeActAgent"
 fi

-get_openhands_version
+get_agent_version

 echo "AGENT: $AGENT"
-echo "OPENHANDS_VERSION: $OPENHANDS_VERSION"
+echo "AGENT_VERSION: $AGENT_VERSION"
 echo "MODEL_CONFIG: $MODEL_CONFIG"

-COMMAND="poetry run python evaluation/benchmarks/bird/run_infer.py \
+COMMAND="poetry run python evaluation/bird/run_infer.py \
  --agent-cls $AGENT \
  --llm-config $MODEL_CONFIG \
  --max-iterations 5 \
  --eval-num-workers $NUM_WORKERS \
-  --eval-note $OPENHANDS_VERSION" \
+  --eval-note $AGENT_VERSION" \

 if [ -n "$EVAL_LIMIT" ]; then
  echo "EVAL_LIMIT: $EVAL_LIMIT"
--- a/evaluation/benchmarks/browsing_delegation/README.md
+++ b/evaluation/benchmarks/browsing_delegation/README.md
@@ -7,12 +7,12 @@ If so, the browsing performance upper-bound of CodeActAgent will be the performa

 ## Setup Environment and LLM Configuration

-Please follow instruction [here](../../README.md#setup) to setup your local development environment and LLM.
+Please follow instruction [here](../README.md#setup) to setup your local development environment and LLM.

 ## Run Inference

 ```bash
-./evaluation/benchmarks/browsing_delegation/scripts/run_infer.sh [model_config] [git-version] [agent] [eval_limit]
+./evaluation/browsing_delegation/scripts/run_infer.sh [model_config] [git-version] [agent] [eval_limit]
 # e.g., ./evaluation/swe_bench/scripts/run_infer.sh llm.eval_gpt4_1106_preview_llm HEAD CodeActAgent 300
 ```

--- a/evaluation/benchmarks/browsing_delegation/run_infer.py
+++ b/evaluation/benchmarks/browsing_delegation/run_infer.py
--- a/evaluation/benchmarks/browsing_delegation/scripts/run_infer.sh
+++ b/evaluation/benchmarks/browsing_delegation/scripts/run_infer.sh
@@ -20,15 +20,15 @@ if [ -z "$AGENT" ]; then
  AGENT="CodeActAgent"
 fi

-get_openhands_version
+get_agent_version

 echo "AGENT: $AGENT"
-echo "OPENHANDS_VERSION: $OPENHANDS_VERSION"
+echo "AGENT_VERSION: $AGENT_VERSION"
 echo "MODEL_CONFIG: $MODEL_CONFIG"

-EVAL_NOTE="$OPENHANDS_VERSION"
+EVAL_NOTE="$AGENT_VERSION"

-COMMAND="poetry run python evaluation/benchmarks/browsing_delegation/run_infer.py \
+COMMAND="poetry run python evaluation/browsing_delegation/run_infer.py \
  --agent-cls $AGENT \
  --llm-config $MODEL_CONFIG \
  --max-iterations 1 \
--- a/evaluation/benchmarks/commit0_bench/README.md
+++ b/evaluation/benchmarks/commit0_bench/README.md
@@ -4,18 +4,19 @@ This folder contains the evaluation harness that we built on top of the original

 The evaluation consists of three steps:

-1. Environment setup: [install python environment](../../README.md#development-environment), [configure LLM config](../../README.md#configure-openhands-and-your-llm).
+1. Environment setup: [install python environment](../README.md#development-environment), [configure LLM config](../README.md#configure-openhands-and-your-llm).
 2. [Run Evaluation](#run-inference-on-commit0-instances): Generate a edit patch for each Commit0 Repo, and get the evaluation results

 ## Setup Environment and LLM Configuration

-Please follow instruction [here](../../README.md#setup) to setup your local development environment and LLM.
+Please follow instruction [here](../README.md#setup) to setup your local development environment and LLM.

 ## OpenHands Commit0 Instance-level Docker Support

 OpenHands supports using the Commit0 Docker for **[inference](#run-inference-on-commit0-instances).
 This is now the default behavior.

+
 ## Run Inference on Commit0 Instances

 Make sure your Docker daemon is running, and you have ample disk space (at least 200-500GB, depends on the Commit0 set you are running on) for the [instance-level docker image](#openhands-commit0-instance-level-docker-support).
@@ -23,10 +24,10 @@ Make sure your Docker daemon is running, and you have ample disk space (at least
 When the `run_infer.sh` script is started, it will automatically pull the `lite` split in Commit0. For example, for instance ID `commit-0/minitorch`, it will try to pull our pre-build docker image `wentingzhao/minitorch` from DockerHub. This image will be used create an OpenHands runtime image where the agent will operate on.

 ```bash
-./evaluation/benchmarks/commit0_bench/scripts/run_infer.sh [repo_split] [model_config] [git-version] [agent] [eval_limit] [max_iter] [num_workers] [dataset] [dataset_split]
+./evaluation/commit0_bench/scripts/run_infer.sh [repo_split] [model_config] [git-version] [agent] [eval_limit] [max_iter] [num_workers] [dataset] [dataset_split]

 # Example
-./evaluation/benchmarks/commit0_bench/scripts/run_infer.sh lite llm.eval_sonnet HEAD CodeActAgent 16 100 8 wentingzhao/commit0_combined test
+./evaluation/commit0_bench/scripts/run_infer.sh lite llm.eval_sonnet HEAD CodeActAgent 16 100 8 wentingzhao/commit0_combined test
 ```

 where `model_config` is mandatory, and the rest are optional.
@@ -55,7 +56,7 @@ Let's say you'd like to run 10 instances using `llm.eval_sonnet` and CodeActAgen
 then your command would be:

 ```bash
-./evaluation/benchmarks/commit0_bench/scripts/run_infer.sh lite llm.eval_sonnet HEAD CodeActAgent 10 30 1 wentingzhao/commit0_combined test
+./evaluation/commit0_bench/scripts/run_infer.sh lite llm.eval_sonnet HEAD CodeActAgent 10 30 1 wentingzhao/commit0_combined test
 ```

 ### Run Inference on `RemoteRuntime` (experimental)
@@ -63,17 +64,17 @@ then your command would be:
 This is in limited beta. Contact Xingyao over slack if you want to try this out!

 ```bash
-./evaluation/benchmarks/commit0_bench/scripts/run_infer.sh [repo_split] [model_config] [git-version] [agent] [eval_limit] [max_iter] [num_workers] [dataset] [dataset_split]
+./evaluation/commit0_bench/scripts/run_infer.sh [repo_split] [model_config] [git-version] [agent] [eval_limit] [max_iter] [num_workers] [dataset] [dataset_split]

 # Example - This runs evaluation on CodeActAgent for 10 instances on "wentingzhao/commit0_combined"'s test set, with max 30 iteration per instances, with 1 number of workers running in parallel
 ALLHANDS_API_KEY="YOUR-API-KEY" RUNTIME=remote SANDBOX_REMOTE_RUNTIME_API_URL="https://runtime.eval.all-hands.dev" EVAL_DOCKER_IMAGE_PREFIX="docker.io/wentingzhao" \
-./evaluation/benchmarks/commit0_bench/scripts/run_infer.sh lite llm.eval_sonnet HEAD CodeActAgent 10 30 1 wentingzhao/commit0_combined test
+./evaluation/commit0_bench/scripts/run_infer.sh lite llm.eval_sonnet HEAD CodeActAgent 10 30 1 wentingzhao/commit0_combined test
 ```

 To clean-up all existing runtime you've already started, run:

 ```bash
-ALLHANDS_API_KEY="YOUR-API-KEY" ./evaluation/benchmarks/commit0_bench/scripts/cleanup_remote_runtime.sh
+ALLHANDS_API_KEY="YOUR-API-KEY" ./evaluation/commit0_bench/scripts/cleanup_remote_runtime.sh
 ```

 ### Specify a subset of tasks to run infer
--- a/evaluation/benchmarks/commit0_bench/run_infer.py
+++ b/evaluation/benchmarks/commit0_bench/run_infer.py
--- a/evaluation/benchmarks/commit0_bench/scripts/cleanup_remote_runtime.sh
+++ b/evaluation/benchmarks/commit0_bench/scripts/cleanup_remote_runtime.sh
--- a/evaluation/benchmarks/commit0_bench/scripts/run_infer.sh
+++ b/evaluation/benchmarks/commit0_bench/scripts/run_infer.sh
@@ -61,10 +61,10 @@ echo "USE_INSTANCE_IMAGE: $USE_INSTANCE_IMAGE"
 export RUN_WITH_BROWSING=$RUN_WITH_BROWSING
 echo "RUN_WITH_BROWSING: $RUN_WITH_BROWSING"

-get_openhands_version
+get_agent_version

 echo "AGENT: $AGENT"
-echo "OPENHANDS_VERSION: $OPENHANDS_VERSION"
+echo "AGENT_VERSION: $AGENT_VERSION"
 echo "MODEL_CONFIG: $MODEL_CONFIG"
 echo "DATASET: $DATASET"
 echo "HF SPLIT: $SPLIT"
@@ -75,7 +75,7 @@ if [ -z "$USE_HINT_TEXT" ]; then
  export USE_HINT_TEXT=false
 fi
 echo "USE_HINT_TEXT: $USE_HINT_TEXT"
-EVAL_NOTE="$OPENHANDS_VERSION"
+EVAL_NOTE="$AGENT_VERSION"
 # if not using Hint, add -no-hint to the eval note
 if [ "$USE_HINT_TEXT" = false ]; then
  EVAL_NOTE="$EVAL_NOTE-no-hint"
@@ -91,7 +91,7 @@ fi

 function run_eval() {
  local eval_note=$1
-  COMMAND="poetry run python evaluation/benchmarks/commit0_bench/run_infer.py \
+  COMMAND="poetry run python evaluation/commit0_bench/run_infer.py \
    --agent-cls $AGENT \
    --llm-config $MODEL_CONFIG \
    --max-iterations $MAX_ITER \
--- a/evaluation/benchmarks/discoverybench/README.md
+++ b/evaluation/benchmarks/discoverybench/README.md
@@ -16,7 +16,7 @@
 2. Execute the bash script to start DiscoveryBench Evaluation

 ```
-./evaluation/benchmarks/discoverybench/scripts/run_infer.sh [YOUR MODEL CONFIG]
+./evaluation/discoverybench/scripts/run_infer.sh [YOUR MODEL CONFIG]
 ```
 Replace `[YOUR MODEL CONFIG]` with any model the model that you have set up in `config.toml`

@@ -27,7 +27,7 @@ When the `run_infer.sh` script is started, it will automatically pull the latest


 ```
-./evaluation/benchmarks/discoverybench/scripts/run_infer.sh [MODEL_CONFIG] [GIT_COMMIT] [AGENT] [EVAL_LIMIT] [NUM_WORKERS]
+./evaluation/discoverybench/scripts/run_infer.sh [MODEL_CONFIG] [GIT_COMMIT] [AGENT] [EVAL_LIMIT] [NUM_WORKERS]
 ```

 - `MODEL_CONFIG`: Name of the model you want to evaluate with
--- a/evaluation/benchmarks/discoverybench/eval_utils/README.md
+++ b/evaluation/benchmarks/discoverybench/eval_utils/README.md
--- a/evaluation/benchmarks/discoverybench/eval_utils/init.py
+++ b/evaluation/benchmarks/discoverybench/eval_utils/init.py
--- a/evaluation/benchmarks/discoverybench/eval_utils/eval_w_subhypo_gen.py
+++ b/evaluation/benchmarks/discoverybench/eval_utils/eval_w_subhypo_gen.py
--- a/evaluation/benchmarks/discoverybench/eval_utils/lm_utils.py
+++ b/evaluation/benchmarks/discoverybench/eval_utils/lm_utils.py
--- a/evaluation/benchmarks/discoverybench/eval_utils/openai_helpers.py
+++ b/evaluation/benchmarks/discoverybench/eval_utils/openai_helpers.py
--- a/evaluation/benchmarks/discoverybench/eval_utils/openai_semantic_gen_prompts.py
+++ b/evaluation/benchmarks/discoverybench/eval_utils/openai_semantic_gen_prompts.py
--- a/evaluation/benchmarks/discoverybench/eval_utils/response_parser.py
+++ b/evaluation/benchmarks/discoverybench/eval_utils/response_parser.py
--- a/evaluation/benchmarks/discoverybench/run_infer.py
+++ b/evaluation/benchmarks/discoverybench/run_infer.py
@@ -5,10 +5,10 @@ import os
 import git
 import pandas as pd

-from evaluation.benchmarks.discoverybench.eval_utils.eval_w_subhypo_gen import (
+from evaluation.discoverybench.eval_utils.eval_w_subhypo_gen import (
    run_eval_gold_vs_gen_NL_hypo_workflow,
 )
-from evaluation.benchmarks.discoverybench.eval_utils.response_parser import (
+from evaluation.discoverybench.eval_utils.response_parser import (
    extract_gen_hypo_from_logs,
 )
 from evaluation.utils.shared import (
--- a/evaluation/benchmarks/discoverybench/scripts/run_infer.sh
+++ b/evaluation/benchmarks/discoverybench/scripts/run_infer.sh
@@ -23,19 +23,19 @@ if [ -z "$AGENT" ]; then
  AGENT="CodeActAgent"
 fi

-get_openhands_version
+get_agent_version

 echo "AGENT: $AGENT"
-echo "OPENHANDS_VERSION: $OPENHANDS_VERSION"
+echo "AGENT_VERSION: $AGENT_VERSION"
 echo "MODEL_CONFIG: $MODEL_CONFIG"

-COMMAND="poetry run python evaluation/benchmarks/discoverybench/run_infer.py \
+COMMAND="poetry run python evaluation/discoverybench/run_infer.py \
  --agent-cls $AGENT \
  --llm-config $MODEL_CONFIG \
  --max-iterations 10 \
  --max-chars 10000000 \
  --eval-num-workers $NUM_WORKERS \
-  --eval-note $OPENHANDS_VERSION"
+  --eval-note $AGENT_VERSION"

 if [ -n "$EVAL_LIMIT" ]; then
  echo "EVAL_LIMIT: $EVAL_LIMIT"
--- a/evaluation/benchmarks/gaia/README.md
+++ b/evaluation/benchmarks/gaia/README.md
@@ -4,18 +4,17 @@ This folder contains evaluation harness for evaluating agents on the [GAIA bench

 ## Setup Environment and LLM Configuration

-Please follow instruction [here](../../README.md#setup) to setup your local development environment and LLM.
+Please follow instruction [here](../README.md#setup) to setup your local development environment and LLM.

 ## Run the evaluation
-
 We are using the GAIA dataset hosted on [Hugging Face](https://huggingface.co/datasets/gaia-benchmark/GAIA).
 Please accept the terms and make sure to have logged in on your computer by `huggingface-cli login` before running the evaluation.

-Following is the basic command to start the evaluation. Here we are evaluating on the validation set for the `2023_all` split. You can adjust `./evaluation/benchmarks/gaia/scripts/run_infer.sh` to change the subset you want to evaluate on.
+Following is the basic command to start the evaluation. Here we are evaluating on the validation set for the `2023_all` split. You can adjust `./evaluation/gaia/scripts/run_infer.sh` to change the subset you want to evaluate on.

 ```bash
-./evaluation/benchmarks/gaia/scripts/run_infer.sh [model_config] [git-version] [agent] [eval_limit] [gaia_subset]
-# e.g., ./evaluation/benchmarks/gaia/scripts/run_infer.sh eval_gpt4_1106_preview 0.6.2 CodeActAgent 300
+./evaluation/gaia/scripts/run_infer.sh [model_config] [git-version] [agent] [eval_limit] [gaia_subset]
+# e.g., ./evaluation/gaia/scripts/run_infer.sh eval_gpt4_1106_preview 0.6.2 CodeActAgent 300
 ```

 where `model_config` is mandatory, while `git-version`, `agent`, `eval_limit` and `gaia_subset` are optional.
@@ -36,14 +35,13 @@ to `CodeActAgent`.
 For example,

 ```bash
-./evaluation/benchmarks/gaia/scripts/run_infer.sh eval_gpt4_1106_preview 0.6.2 CodeActAgent 10
+./evaluation/gaia/scripts/run_infer.sh eval_gpt4_1106_preview 0.6.2 CodeActAgent 10
 ```

 ## Get score

 Then you can get stats by running the following command:
-
 ```bash
-python ./evaluation/benchmarks/gaia/get_score.py \
+python ./evaluation/gaia/get_score.py \
 --file <path_to/output.json>
 ```
--- a/evaluation/benchmarks/gaia/get_score.py
+++ b/evaluation/benchmarks/gaia/get_score.py
--- a/evaluation/benchmarks/gaia/run_infer.py
+++ b/evaluation/benchmarks/gaia/run_infer.py
@@ -7,7 +7,7 @@ import huggingface_hub
 import pandas as pd
 from datasets import load_dataset

-from evaluation.benchmarks.gaia.scorer import question_scorer
+from evaluation.gaia.scorer import question_scorer
 from evaluation.utils.shared import (
    EvalMetadata,
    EvalOutput,
--- a/evaluation/benchmarks/gaia/scorer.py
+++ b/evaluation/benchmarks/gaia/scorer.py
--- a/evaluation/benchmarks/gaia/scripts/run_infer.sh
+++ b/evaluation/benchmarks/gaia/scripts/run_infer.sh
@@ -21,28 +21,28 @@ if [ -z "$AGENT" ]; then
  AGENT="CodeActAgent"
 fi

-get_openhands_version
+get_agent_version

 if [ -z "$LEVELS" ]; then
  LEVELS="2023_level1"
  echo "Levels not specified, use default $LEVELS"
 fi

-get_openhands_version
+get_agent_version

 echo "AGENT: $AGENT"
-echo "OPENHANDS_VERSION: $OPENHANDS_VERSION"
+echo "AGENT_VERSION: $AGENT_VERSION"
 echo "MODEL_CONFIG: $MODEL_CONFIG"
 echo "LEVELS: $LEVELS"

-COMMAND="poetry run python ./evaluation/benchmarks/gaia/run_infer.py \
+COMMAND="poetry run python ./evaluation/gaia/run_infer.py \
  --agent-cls $AGENT \
  --llm-config $MODEL_CONFIG \
  --max-iterations 30 \
  --level $LEVELS \
  --data-split validation \
  --eval-num-workers $NUM_WORKERS \
-  --eval-note ${OPENHANDS_VERSION}_${LEVELS}"
+  --eval-note ${AGENT_VERSION}_${LEVELS}"

 if [ -n "$EVAL_LIMIT" ]; then
  echo "EVAL_LIMIT: $EVAL_LIMIT"
--- a/evaluation/benchmarks/gorilla/README.md
+++ b/evaluation/benchmarks/gorilla/README.md
@@ -4,14 +4,14 @@ This folder contains evaluation harness we built on top of the original [Gorilla

 ## Setup Environment and LLM Configuration

-Please follow instruction [here](../../README.md#setup) to setup your local development environment and LLM.
+Please follow instruction [here](../README.md#setup) to setup your local development environment and LLM.

 ## Run Inference on APIBench Instances

 Make sure your Docker daemon is running, then run this bash script:

 ```bash
-./evaluation/benchmarks/gorilla/scripts/run_infer.sh [model_config] [git-version] [agent] [eval_limit] [hubs]
+./evaluation/gorilla/scripts/run_infer.sh [model_config] [git-version] [agent] [eval_limit] [hubs]
 ```

 where `model_config` is mandatory, while all other arguments are optional.
@@ -35,5 +35,5 @@ Note: in order to use `eval_limit`, you must also set `agent`; in order to use `
 For example,

 ```bash
-./evaluation/benchmarks/gorilla/scripts/run_infer.sh llm 0.6.2 CodeActAgent 10 th
+./evaluation/gorilla/scripts/run_infer.sh llm 0.6.2 CodeActAgent 10 th
 ```
--- a/evaluation/benchmarks/gorilla/ast_eval_hf.py
+++ b/evaluation/benchmarks/gorilla/ast_eval_hf.py
--- a/evaluation/benchmarks/gorilla/ast_eval_tf.py
+++ b/evaluation/benchmarks/gorilla/ast_eval_tf.py
--- a/evaluation/benchmarks/gorilla/ast_eval_th.py
+++ b/evaluation/benchmarks/gorilla/ast_eval_th.py
--- a/evaluation/benchmarks/gorilla/run_infer.py
+++ b/evaluation/benchmarks/gorilla/run_infer.py
@@ -5,7 +5,7 @@ import os
 import pandas as pd
 import requests

-from evaluation.benchmarks.gorilla.utils import encode_question, get_data_for_hub
+from evaluation.gorilla.utils import encode_question, get_data_for_hub
 from evaluation.utils.shared import (
    EvalMetadata,
    EvalOutput,
--- a/evaluation/benchmarks/gorilla/scripts/run_infer.sh
+++ b/evaluation/benchmarks/gorilla/scripts/run_infer.sh
@@ -21,7 +21,7 @@ if [ -z "$AGENT" ]; then
  AGENT="CodeActAgent"
 fi

-get_openhands_version
+get_agent_version

 if [ -z "$HUBS" ]; then
  HUBS="hf,torch,tf"
@@ -29,18 +29,18 @@ if [ -z "$HUBS" ]; then
 fi

 echo "AGENT: $AGENT"
-echo "OPENHANDS_VERSION: $OPENHANDS_VERSION"
+echo "AGENT_VERSION: $AGENT_VERSION"
 echo "MODEL_CONFIG: $MODEL_CONFIG"
 echo "HUBS: $HUBS"

-COMMAND="poetry run python evaluation/benchmarks/gorilla/run_infer.py \
+COMMAND="poetry run python evaluation/gorilla/run_infer.py \
  --agent-cls $AGENT \
  --llm-config $MODEL_CONFIG \
  --max-iterations 30 \
  --hubs $HUBS \
  --data-split validation \
  --eval-num-workers $NUM_WORKERS \
-  --eval-note ${OPENHANDS_VERSION}_${LEVELS}"
+  --eval-note ${AGENT_VERSION}_${LEVELS}"

 if [ -n "$EVAL_LIMIT" ]; then
  echo "EVAL_LIMIT: $EVAL_LIMIT"
--- a/evaluation/benchmarks/gorilla/utils.py
+++ b/evaluation/benchmarks/gorilla/utils.py
--- a/evaluation/benchmarks/gpqa/README.md
+++ b/evaluation/benchmarks/gpqa/README.md
@@ -3,7 +3,6 @@
 Implements the evaluation of agents on the GPQA benchmark introduced in [GPQA: A Graduate-Level Google-Proof Q&A Benchmark](https://arxiv.org/abs/2308.07124).

 This code implements the evaluation of agents on the GPQA Benchmark with Open Book setting.
-
 - The benchmark consists of 448 high-quality and extremely difficult multiple-choice questions in the domains of biology, physics, and chemistry. The questions are intentionally designed to be "Google-proof," meaning that even highly skilled non-expert validators achieve only 34% accuracy despite unrestricted access to the web.
 - Even experts in the corresponding domains achieve only 65% accuracy.
 - State-of-the-art AI systems achieve only 39% accuracy on this challenging dataset.
@@ -12,24 +11,20 @@ This code implements the evaluation of agents on the GPQA Benchmark with Open Bo
 Accurate solving of above graduate level questions would require both tool use (e.g., python for calculations) and web-search for finding related facts as information required for the questions might not be part of the LLM knowledge / training data.

 Further references:
-
- <https://arxiv.org/pdf/2311.12022>
- <https://paperswithcode.com/dataset/gpqa>
- <https://github.com/idavidrein/gpqa>
+- https://arxiv.org/pdf/2311.12022
+- https://paperswithcode.com/dataset/gpqa
+- https://github.com/idavidrein/gpqa

 ## Setup Environment and LLM Configuration

-Please follow instruction [here](../../README.md#setup) to setup your local development environment and LLM.
+Please follow instruction [here](../README.md#setup) to setup your local development environment and LLM.

 ## Run Inference on GPQA Benchmark
-
 'gpqa_main', 'gqpa_diamond', 'gpqa_experts', 'gpqa_extended' -- data split options
 From the root of the OpenHands repo, run the following command:
-
 ```bash
-./evaluation/benchmarks/gpqa/scripts/run_infer.sh [model_config_name] [git-version] [num_samples_eval] [data_split] [AgentClass]
+./evaluation/gpqa/scripts/run_infer.sh [model_config_name] [git-version] [num_samples_eval] [data_split] [AgentClass]
 ```
-
 You can replace `model_config_name` with any model you set up in `config.toml`.

 - `model_config_name`: The model configuration name from `config.toml` that you want to evaluate.
--- a/evaluation/benchmarks/gpqa/init.py
+++ b/evaluation/benchmarks/gpqa/init.py
--- a/evaluation/benchmarks/gpqa/run_infer.py
+++ b/evaluation/benchmarks/gpqa/run_infer.py
--- a/evaluation/benchmarks/gpqa/scripts/run_infer.sh
+++ b/evaluation/benchmarks/gpqa/scripts/run_infer.sh
@@ -27,19 +27,19 @@ if [ -z "$DATA_SPLIT" ]; then
  DATA_SPLIT="gpqa_diamond"
 fi

-get_openhands_version
+get_agent_version

 echo "AGENT: $AGENT"
-echo "OPENHANDS_VERSION: $OPENHANDS_VERSION"
+echo "AGENT_VERSION: $AGENT_VERSION"
 echo "MODEL_CONFIG: $MODEL_CONFIG"

-COMMAND="poetry run python evaluation/benchmarks/gpqa/run_infer.py \
+COMMAND="poetry run python evaluation/gpqa/run_infer.py \
  --agent-cls $AGENT \
  --llm-config $MODEL_CONFIG \
  --max-iterations 10 \
  --eval-num-workers $NUM_WORKERS \
  --data-split $DATA_SPLIT \
-  --eval-note $OPENHANDS_VERSION"
+  --eval-note $AGENT_VERSION"

 if [ -n "$EVAL_LIMIT" ]; then
  echo "EVAL_LIMIT: $EVAL_LIMIT"
--- a/evaluation/benchmarks/humanevalfix/README.md
+++ b/evaluation/benchmarks/humanevalfix/README.md
@@ -4,21 +4,23 @@ Implements evaluation of agents on HumanEvalFix from the HumanEvalPack benchmark

 ## Setup Environment and LLM Configuration

-Please follow instruction [here](../../README.md#setup) to setup your local development environment and LLM.
+Please follow instruction [here](../README.md#setup) to setup your local development environment and LLM.

 ## Run Inference on HumanEvalFix

 ```bash
-./evaluation/benchmarks/humanevalfix/scripts/run_infer.sh eval_gpt4_1106_preview
+./evaluation/humanevalfix/scripts/run_infer.sh eval_gpt4_1106_preview
 ```

 You can replace `eval_gpt4_1106_preview` with any model you set up in `config.toml`.

+
 ## Examples

 For each problem, OpenHands is given a set number of iterations to fix the failing code. The history field shows each iteration's response to correct its code that fails any test case.

-```json
+
+```
 {
    "task_id": "Python/2",
    "instruction": "Please fix the function in Python__2.py such that all test cases pass.\nEnvironment has been set up for you to start working. You may assume all necessary tools are installed.\n\n# Problem Statement\ndef truncate_number(number: float) -> float:\n    return number % 1.0 + 1.0\n\n\n\n\n\n\ndef check(truncate_number):\n    assert truncate_number(3.5) == 0.5\n    assert abs(truncate_number(1.33) - 0.33) < 1e-6\n    assert abs(truncate_number(123.456) - 0.456) < 1e-6\n\ncheck(truncate_number)\n\nIMPORTANT: You should ONLY interact with the environment provided to you AND NEVER ASK FOR HUMAN HELP.\nYou should NOT modify any existing test case files. If needed, you can add new test cases in a NEW file to reproduce the issue.\nYou SHOULD INCLUDE PROPER INDENTATION in your edit commands.\nWhen you think you have fixed the issue through code changes, please finish the interaction using the "finish" tool.\n",
--- a/evaluation/benchmarks/humanevalfix/init.py
+++ b/evaluation/benchmarks/humanevalfix/init.py
--- a/evaluation/benchmarks/humanevalfix/run_infer.py
+++ b/evaluation/benchmarks/humanevalfix/run_infer.py
--- a/Show More
+++ b/Show More