Merge branch 'main' into openhands-fix-issue-5112

Fix pr #5118 : Fix issue #5112 : [Bug]: "Push to GitHub" shows up even if there's no repo connected
Fix issue #5112 : [Bug]: "Push to GitHub" shows up even if there's no repo connected
2026-04-29 03:00:45 -04:00 · 2024-11-22 09:11:22 -05:00 · 2024-11-19 12:31:03 +00:00 · 2024-11-19 06:03:47 +00:00
412 changed files with 7651 additions and 11981 deletions
--- a/.github/dependabot.yml
+++ b/.github/dependabot.yml
@@ -16,9 +16,6 @@ updates:
      chromadb:
        patterns:
          - "chromadb"
-      browsergym:
-        patterns:
-          - "browsergym"
      security-all:
        applies-to: "security-updates"
        patterns:
--- a/.github/workflows/eval-runner.yml
+++ b/.github/workflows/eval-runner.yml
@@ -1,8 +1,10 @@
-name: Run SWE-Bench Evaluation
+name: Run Evaluation

 on:
  pull_request:
    types: [labeled]
+  schedule:
+    - cron: "0 1 * * *" # Run daily at 1 AM UTC
  workflow_dispatch:
    inputs:
      reason:
@@ -58,6 +60,24 @@ jobs:
          echo "api_key = \"$DEEPSEEK_API_KEY\"" >> config.toml
          echo "temperature = 0.0" >> config.toml

+      - name: Run integration test evaluation
+        env:
+          ALLHANDS_API_KEY: ${{ secrets.ALLHANDS_EVAL_RUNTIME_API_KEY }}
+          RUNTIME: remote
+          SANDBOX_REMOTE_RUNTIME_API_URL: https://runtime.eval.all-hands.dev
+          EVAL_DOCKER_IMAGE_PREFIX: us-central1-docker.pkg.dev/evaluation-092424/swe-bench-images
+
+        run: |
+          poetry run ./evaluation/integration_tests/scripts/run_infer.sh llm.eval HEAD CodeActAgent '' $N_PROCESSES
+
+          # get evaluation report
+          REPORT_FILE=$(find evaluation/evaluation_outputs/outputs/integration_tests/CodeActAgent/deepseek-chat_maxiter_10_N* -name "report.md" -type f | head -n 1)
+          echo "REPORT_FILE: $REPORT_FILE"
+          echo "INTEGRATION_TEST_REPORT<<EOF" >> $GITHUB_ENV
+          cat $REPORT_FILE >> $GITHUB_ENV
+          echo >> $GITHUB_ENV
+          echo "EOF" >> $GITHUB_ENV
+
      - name: Run SWE-Bench evaluation
        env:
          ALLHANDS_API_KEY: ${{ secrets.ALLHANDS_EVAL_RUNTIME_API_KEY }}
@@ -66,12 +86,12 @@ jobs:
          EVAL_DOCKER_IMAGE_PREFIX: us-central1-docker.pkg.dev/evaluation-092424/swe-bench-images

        run: |
-          poetry run ./evaluation/benchmarks/swe_bench/scripts/run_infer.sh llm.eval HEAD CodeActAgent 300 30 $N_PROCESSES "princeton-nlp/SWE-bench_Lite" test
+          poetry run ./evaluation/swe_bench/scripts/run_infer.sh llm.eval HEAD CodeActAgent 300 30 $N_PROCESSES "princeton-nlp/SWE-bench_Lite" test
          OUTPUT_FOLDER=$(find evaluation/evaluation_outputs/outputs/princeton-nlp__SWE-bench_Lite-test/CodeActAgent -name "deepseek-chat_maxiter_50_N_*-no-hint-run_1" -type d | head -n 1)
          echo "OUTPUT_FOLDER for SWE-bench evaluation: $OUTPUT_FOLDER"
-          poetry run ./evaluation/benchmarks/swe_bench/scripts/eval_infer_remote.sh $OUTPUT_FOLDER/output.jsonl $N_PROCESSES "princeton-nlp/SWE-bench_Lite" test
+          poetry run ./evaluation/swe_bench/scripts/eval_infer_remote.sh $OUTPUT_FOLDER/output.jsonl $N_PROCESSES "princeton-nlp/SWE-bench_Lite" test

-          poetry run ./evaluation/benchmarks/swe_bench/scripts/eval/summarize_outputs.py $OUTPUT_FOLDER/output.jsonl > summarize_outputs.log 2>&1
+          poetry run ./evaluation/swe_bench/scripts/eval/summarize_outputs.py $OUTPUT_FOLDER/output.jsonl > summarize_outputs.log 2>&1
          echo "SWEBENCH_REPORT<<EOF" >> $GITHUB_ENV
          cat summarize_outputs.log >> $GITHUB_ENV
          echo "EOF" >> $GITHUB_ENV
@@ -125,6 +145,9 @@ jobs:
              **SWE-Bench Evaluation Report**
              ${{ env.SWEBENCH_REPORT }}
              ---
+              **Integration Tests Evaluation Report**
+              ${{ env.INTEGRATION_TEST_REPORT }}
+              ---
              You can download the full evaluation outputs [here](${{ env.ARTIFACT_URL }}).

      - name: Post to a Slack channel
--- a/.github/workflows/integration-runner.yml
+++ b/.github/workflows/integration-runner.yml
@@ -1,158 +0,0 @@
-name: Run Integration Tests
-
-on:
-  pull_request:
-    types: [labeled]
-  workflow_dispatch:
-    inputs:
-      reason:
-        description: 'Reason for manual trigger'
-        required: true
-        default: ''
-  schedule:
-    - cron: '30 22 * * *'  # Runs at 10:30pm UTC every day
-
-env:
-  N_PROCESSES: 10 # Global configuration for number of parallel processes for evaluation
-
-jobs:
-  run-integration-tests:
-    if: github.event.label.name == 'integration-test' || github.event_name == 'workflow_dispatch' || github.event_name == 'schedule'
-    runs-on: ubuntu-latest
-    permissions:
-      contents: "read"
-      id-token: "write"
-      pull-requests: "write"
-      issues: "write"
-    strategy:
-      matrix:
-        python-version: ["3.12"]
-    steps:
-      - name: Checkout repository
-        uses: actions/checkout@v4
-
-      - name: Install poetry via pipx
-        run: pipx install poetry
-
-      - name: Set up Python
-        uses: actions/setup-python@v5
-        with:
-          python-version: ${{ matrix.python-version }}
-          cache: "poetry"
-
-      - name: Comment on PR if 'integration-test' label is present
-        if: github.event_name == 'pull_request' && github.event.label.name == 'integration-test'
-        uses: KeisukeYamashita/create-comment@v1
-        with:
-          unique: false
-          comment: |
-            Hi! I started running the integration tests on your PR. You will receive a comment with the results shortly.
-
-      - name: Install Python dependencies using Poetry
-        run: poetry install --without evaluation,llama-index
-
-      - name: Configure config.toml for testing with Haiku
-        env:
-          LLM_MODEL: "litellm_proxy/claude-3-5-haiku-20241022"
-          LLM_API_KEY: ${{ secrets.LLM_API_KEY }}
-          LLM_BASE_URL: ${{ secrets.LLM_BASE_URL }}
-        run: |
-          echo "[llm.eval]" > config.toml
-          echo "model = \"$LLM_MODEL\"" >> config.toml
-          echo "api_key = \"$LLM_API_KEY\"" >> config.toml
-          echo "base_url = \"$LLM_BASE_URL\"" >> config.toml
-          echo "temperature = 0.0" >> config.toml
-
-      - name: Build environment
-        run: make build
-
-      - name: Run integration test evaluation for Haiku
-        env:
-          SANDBOX_FORCE_REBUILD_RUNTIME: True
-        run: |
-          poetry run ./evaluation/integration_tests/scripts/run_infer.sh llm.eval HEAD CodeActAgent '' $N_PROCESSES '' 'haiku_run'
-
-          # get integration tests report
-          REPORT_FILE_HAIKU=$(find evaluation/evaluation_outputs/outputs/integration_tests/CodeActAgent/*haiku*_maxiter_10_N* -name "report.md" -type f | head -n 1)
-          echo "REPORT_FILE: $REPORT_FILE_HAIKU"
-          echo "INTEGRATION_TEST_REPORT_HAIKU<<EOF" >> $GITHUB_ENV
-          cat $REPORT_FILE_HAIKU >> $GITHUB_ENV
-          echo >> $GITHUB_ENV
-          echo "EOF" >> $GITHUB_ENV
-
-      - name: Wait a little bit
-        run: sleep 10
-
-      - name: Configure config.toml for testing with DeepSeek
-        env:
-          LLM_MODEL: "litellm_proxy/deepseek-chat"
-          LLM_API_KEY: ${{ secrets.LLM_API_KEY }}
-          LLM_BASE_URL: ${{ secrets.LLM_BASE_URL }}
-        run: |
-          echo "[llm.eval]" > config.toml
-          echo "model = \"$LLM_MODEL\"" >> config.toml
-          echo "api_key = \"$LLM_API_KEY\"" >> config.toml
-          echo "base_url = \"$LLM_BASE_URL\"" >> config.toml
-          echo "temperature = 0.0" >> config.toml
-
-      - name: Run integration test evaluation for DeepSeek
-        env:
-          SANDBOX_FORCE_REBUILD_RUNTIME: True
-        run: |
-          poetry run ./evaluation/integration_tests/scripts/run_infer.sh llm.eval HEAD CodeActAgent '' $N_PROCESSES '' 'deepseek_run'
-
-          # get integration tests report
-          REPORT_FILE_DEEPSEEK=$(find evaluation/evaluation_outputs/outputs/integration_tests/CodeActAgent/deepseek*_maxiter_10_N* -name "report.md" -type f | head -n 1)
-          echo "REPORT_FILE: $REPORT_FILE_DEEPSEEK"
-          echo "INTEGRATION_TEST_REPORT_DEEPSEEK<<EOF" >> $GITHUB_ENV
-          cat $REPORT_FILE_DEEPSEEK >> $GITHUB_ENV
-          echo >> $GITHUB_ENV
-          echo "EOF" >> $GITHUB_ENV
-
-      - name: Create archive of evaluation outputs
-        run: |
-          TIMESTAMP=$(date +'%y-%m-%d-%H-%M')
-          cd evaluation/evaluation_outputs/outputs  # Change to the outputs directory
-          tar -czvf ../../../integration_tests_${TIMESTAMP}.tar.gz integration_tests/CodeActAgent/*  # Only include the actual result directories
-
-      - name: Upload evaluation results as artifact
-        uses: actions/upload-artifact@v4
-        id: upload_results_artifact
-        with:
-          name: integration-test-outputs-${{ github.run_id }}-${{ github.run_attempt }}
-          path: integration_tests_*.tar.gz
-
-      - name: Get artifact URLs
-        run: |
-          echo "ARTIFACT_URL=${{ steps.upload_results_artifact.outputs.artifact-url }}" >> $GITHUB_ENV
-
-      - name: Set timestamp and trigger reason
-        run: |
-          echo "TIMESTAMP=$(date +'%Y-%m-%d-%H-%M')" >> $GITHUB_ENV
-          if [[ "${{ github.event_name }}" == "pull_request" ]]; then
-            echo "TRIGGER_REASON=pr-${{ github.event.pull_request.number }}" >> $GITHUB_ENV
-          elif [[ "${{ github.event_name }}" == "workflow_dispatch" ]]; then
-            echo "TRIGGER_REASON=manual-${{ github.event.inputs.reason }}" >> $GITHUB_ENV
-          else
-            echo "TRIGGER_REASON=nightly-scheduled" >> $GITHUB_ENV
-          fi
-
-      - name: Comment with results and artifact link
-        id: create_comment
-        uses: KeisukeYamashita/create-comment@v1
-        with:
-          # if triggered by PR, use PR number, otherwise use 5077 as fallback issue number for manual triggers
-          number: ${{ github.event_name == 'pull_request' && github.event.pull_request.number || 5077 }}
-          unique: false
-          comment: |
-              Trigger by: ${{ github.event_name == 'pull_request' && format('Pull Request (integration-test label on PR #{0})', github.event.pull_request.number) || (github.event_name == 'workflow_dispatch' && format('Manual Trigger: {0}', github.event.inputs.reason)) || 'Nightly Scheduled Run' }}
-              Commit: ${{ github.sha }}
-              **Integration Tests Report (Haiku)**
-              Haiku LLM Test Results:
-              ${{ env.INTEGRATION_TEST_REPORT_HAIKU }}
-              ---
-              **Integration Tests Report (DeepSeek)**
-              DeepSeek LLM Test Results:
-              ${{ env.INTEGRATION_TEST_REPORT_DEEPSEEK }}
-              ---
-              Download evaluation outputs (includes both Haiku and DeepSeek results): [Download](${{ steps.upload_results_artifact.outputs.artifact-url }})
--- a/.github/workflows/openhands-resolver.yml
+++ b/.github/workflows/openhands-resolver.yml
@@ -11,11 +11,6 @@ on:
        required: false
        type: string
        default: "@openhands-agent"
-      target_branch:
-        required: false
-        type: string
-        default: "main"
-        description: "Target branch to pull and create PR against"
    secrets:
      LLM_MODEL:
        required: true
@@ -53,12 +48,12 @@ jobs:

      (
        ((github.event_name == 'issue_comment' || github.event_name == 'pull_request_review_comment') &&
-        contains(github.event.comment.body, inputs.macro || '@openhands-agent') &&
+        startsWith(github.event.comment.body, inputs.macro || '@openhands-agent') &&
        (github.event.comment.author_association == 'OWNER' || github.event.comment.author_association == 'COLLABORATOR' || github.event.comment.author_association == 'MEMBER')
        ) ||

        (github.event_name == 'pull_request_review' &&
-        contains(github.event.review.body, inputs.macro || '@openhands-agent') &&
+        startsWith(github.event.review.body, inputs.macro || '@openhands-agent') &&
        (github.event.review.author_association == 'OWNER' || github.event.review.author_association == 'COLLABORATOR' || github.event.review.author_association == 'MEMBER')
        )
      )
@@ -85,11 +80,11 @@ jobs:
            github.event.label.name == 'fix-me-experimental' ||
            (
              (github.event_name == 'issue_comment' || github.event_name == 'pull_request_review_comment') &&
-              contains(github.event.comment.body, '@openhands-agent-exp')
+              startsWith(github.event.comment.body, '@openhands-agent-exp')
            ) ||
            (
              github.event_name == 'pull_request_review' &&
-              contains(github.event.review.body, '@openhands-agent-exp')
+              startsWith(github.event.review.body, '@openhands-agent-exp')
            )
          )
        uses: actions/cache@v3
@@ -140,9 +135,6 @@ jobs:
          echo "MAX_ITERATIONS=${{ inputs.max_iterations || 50 }}" >> $GITHUB_ENV
          echo "SANDBOX_ENV_GITHUB_TOKEN=${{ secrets.GITHUB_TOKEN }}" >> $GITHUB_ENV

-          # Set branch variables
-          echo "TARGET_BRANCH=${{ inputs.target_branch }}" >> $GITHUB_ENV
-
      - name: Comment on issue with start message
        uses: actions/github-script@v7
        with:
@@ -183,7 +175,7 @@ jobs:
            --issue-number ${{ env.ISSUE_NUMBER }} \
            --issue-type ${{ env.ISSUE_TYPE }} \
            --max-iterations ${{ env.MAX_ITERATIONS }} \
-            --comment-id ${{ env.COMMENT_ID }} \
+            --comment-id ${{ env.COMMENT_ID }}

      - name: Check resolution result
        id: check_result
--- a/.github/workflows/review-pr.yml
+++ b/.github/workflows/review-pr.yml
@@ -0,0 +1,81 @@
+# Workflow that uses OpenHands to review a pull request. PR must be labeled 'review-this'
+name: Use OpenHands to Review Pull Request
+
+on:
+  pull_request:
+    types: [synchronize, labeled]
+
+permissions:
+  contents: write
+  pull-requests: write
+
+jobs:
+  dogfood:
+    if: contains(github.event.pull_request.labels.*.name, 'review-this')
+    runs-on: ubuntu-latest
+    steps:
+    - uses: actions/checkout@v4
+    - name: Set up Docker Buildx
+      id: buildx
+      uses: docker/setup-buildx-action@v3
+    - name: Set up Python
+      uses: actions/setup-python@v5
+      with:
+        python-version: '3.12'
+    - name: install git, github cli
+      run: |
+        sudo apt-get install -y git gh
+        git config --global --add safe.directory $PWD
+    - name: Checkout Repository
+      uses: actions/checkout@v4
+      with:
+        ref: ${{ github.event.pull_request.base.ref }} # check out the target branch
+    - name: Download Diff
+      run: |
+        curl -O "${{ github.event.pull_request.diff_url }}" -L
+    - name: Write Task File
+      run: |
+        echo "Your coworker wants to apply a pull request to this project." > task.txt
+        echo "Read and review ${{ github.event.pull_request.number }}.diff file. Create a review-${{ github.event.pull_request.number }}.txt and write your concise comments and suggestions there." >> task.txt
+        echo "Do not ask me for confirmation at any point." >> task.txt
+        echo "" >> task.txt
+        echo "Title" >> task.txt
+        echo "${{ github.event.pull_request.title }}" >> task.txt
+        echo "" >> task.txt
+        echo "Description" >> task.txt
+        echo "${{ github.event.pull_request.body }}" >> task.txt
+        echo "" >> task.txt
+        echo "Diff file is: ${{ github.event.pull_request.number }}.diff" >> task.txt
+    - name: Set up environment
+      run: |
+        curl -sSL https://install.python-poetry.org | python3 -
+        export PATH="/github/home/.local/bin:$PATH"
+        poetry install --without evaluation,llama-index
+        poetry run playwright install --with-deps chromium
+    - name: Run OpenHands
+      env:
+        LLM_API_KEY: ${{ secrets.LLM_API_KEY }}
+        LLM_MODEL: ${{ vars.LLM_MODEL }}
+      run: |
+        # Append path to launch poetry
+        export PATH="/github/home/.local/bin:$PATH"
+        # Append path to correctly import package, note: must set pwd at first
+        export PYTHONPATH=$(pwd):$PYTHONPATH
+        export WORKSPACE_MOUNT_PATH=$GITHUB_WORKSPACE
+        export WORKSPACE_BASE=$GITHUB_WORKSPACE
+        echo -e "/exit\n" | poetry run python openhands/core/main.py -i 50 -f task.txt
+        rm task.txt
+    - name: Check if review file is non-empty
+      id: check_file
+      run: |
+        ls -la
+        if [[ -s review-${{ github.event.pull_request.number }}.txt ]]; then
+          echo "non_empty=true" >> $GITHUB_OUTPUT
+        fi
+      shell: bash
+    - name: Create PR review if file is non-empty
+      env:
+        GH_TOKEN: ${{ github.token }}
+      if: steps.check_file.outputs.non_empty == 'true'
+      run: |
+        gh pr review ${{ github.event.pull_request.number }} --comment --body-file "review-${{ github.event.pull_request.number }}.txt"
--- a/.github/workflows/run-eval.yml
+++ b/.github/workflows/run-eval.yml
@@ -1,53 +0,0 @@
-# Run evaluation on a PR
-name: Run Eval
-
-# Runs when a PR is labeled with one of the "run-eval-" labels
-on:
-  pull_request:
-    types: [labeled]
-
-jobs:
-  trigger-job:
-    name: Trigger remote eval job
-    if: ${{ github.event.label.name == 'run-eval-xs' || github.event.label.name == 'run-eval-s' || github.event.label.name == 'run-eval-m' }}
-    runs-on: ubuntu-latest
-
-    steps:
-      - name: Checkout PR branch
-        uses: actions/checkout@v3
-        with:
-          ref: ${{ github.head_ref }}
-
-      - name: Trigger remote job
-        run: |
-          REPO_URL="https://github.com/${{ github.repository }}"
-          PR_BRANCH="${{ github.head_ref }}"
-          echo "Repository URL: $REPO_URL"
-          echo "PR Branch: $PR_BRANCH"
-
-          if [[ "${{ github.event.label.name }}" == "run-eval-xs" ]]; then
-            EVAL_INSTANCES="1"
-          elif [[ "${{ github.event.label.name }}" == "run-eval-s" ]]; then
-            EVAL_INSTANCES="5"
-          elif [[ "${{ github.event.label.name }}" == "run-eval-m" ]]; then
-            EVAL_INSTANCES="30"
-          fi
-
-          curl -X POST \
-            -H "Authorization: Bearer ${{ secrets.PAT_TOKEN }}" \
-            -H "Accept: application/vnd.github+json" \
-            -d "{\"ref\": \"main\", \"inputs\": {\"github-repo\": \"${REPO_URL}\", \"github-branch\": \"${PR_BRANCH}\", \"pr-number\": \"${{ github.event.pull_request.number }}\", \"eval-instances\": \"${EVAL_INSTANCES}\"}}" \
-            https://api.github.com/repos/All-Hands-AI/evaluation/actions/workflows/create-branch.yml/dispatches
-
-          # Send Slack message
-          PR_URL="https://github.com/${{ github.repository }}/pull/${{ github.event.pull_request.number }}"
-          slack_text="PR $PR_URL has triggered evaluation on $EVAL_INSTANCES instances..."
-          curl -X POST -H 'Content-type: application/json' --data '{"text":"'"$slack_text"'"}' \
-            https://hooks.slack.com/services/${{ secrets.SLACK_TOKEN }}
-
-      - name: Comment on PR
-        uses: KeisukeYamashita/create-comment@v1
-        with:
-          unique: false
-          comment: |
-            Running evaluation on the PR. Once eval is done, the results will be posted.
--- a/.gitignore
+++ b/.gitignore
@@ -175,7 +175,6 @@ evaluation/gaia/data
 evaluation/gorilla/data
 evaluation/toolqa/data
 evaluation/scienceagentbench/benchmark
-evaluation/commit0_bench/repos

 # openhands resolver
 output/
--- a/README.md
+++ b/README.md
@@ -90,8 +90,6 @@ See more about the community in [COMMUNITY.md](./COMMUNITY.md) or find details o

 ## 📈 Progress

-See the monthly OpenHands roadmap [here](https://github.com/orgs/All-Hands-AI/projects/1) (updated at the maintainer's meeting at the end of each month).
-
 <p align="center">
  <a href="https://star-history.com/#All-Hands-AI/OpenHands&Date">
    <img src="https://api.star-history.com/svg?repos=All-Hands-AI/OpenHands&type=Date" width="500" alt="Star History Chart">
--- a/docs/i18n/fr/docusaurus-plugin-content-docs/current/usage/how-to/evaluation-harness.md
+++ b/docs/i18n/fr/docusaurus-plugin-content-docs/current/usage/how-to/evaluation-harness.md
@@ -76,7 +76,7 @@ La fonction `run_controller()` est le cœur de l'exécution d'OpenHands. Elle g

 ## Le moyen le plus simple de commencer : Explorer les benchmarks existants

-Nous vous encourageons à examiner les différents benchmarks d'évaluation disponibles dans le [répertoire `evaluation/benchmarks/`](https://github.com/All-Hands-AI/OpenHands/blob/main/evaluation/benchmarks) de notre dépôt.
+Nous vous encourageons à examiner les différents benchmarks d'évaluation disponibles dans le [répertoire `evaluation/`](https://github.com/All-Hands-AI/OpenHands/blob/main/evaluation) de notre dépôt.

 Pour intégrer votre propre benchmark, nous vous suggérons de commencer par celui qui ressemble le plus à vos besoins. Cette approche peut considérablement rationaliser votre processus d'intégration, vous permettant de vous appuyer sur les structures existantes et de les adapter à vos exigences spécifiques.

--- a/docs/i18n/zh-Hans/docusaurus-plugin-content-docs/current/usage/how-to/evaluation-harness.md
+++ b/docs/i18n/zh-Hans/docusaurus-plugin-content-docs/current/usage/how-to/evaluation-harness.md
@@ -73,7 +73,7 @@ OpenHands 的主要入口点在 `openhands/core/main.py` 中。以下是它工

 ## 入门最简单的方法：探索现有基准

-我们鼓励您查看我们仓库的 [`evaluation/benchmarks/` 目录](https://github.com/All-Hands-AI/OpenHands/blob/main/evaluation/benchmarks)中提供的各种评估基准。
+我们鼓励您查看我们仓库的 [`evaluation/` 目录](https://github.com/All-Hands-AI/OpenHands/blob/main/evaluation)中提供的各种评估基准。

 要集成您自己的基准，我们建议从最接近您需求的基准开始。这种方法可以显著简化您的集成过程，允许您在现有结构的基础上进行构建并使其适应您的特定要求。

--- a/docs/modules/usage/configuration-options.md
+++ b/docs/modules/usage/configuration-options.md
@@ -1,459 +0,0 @@
---
-id: configuration-options
-title: Configuration Options
-sidebar_label: Configuration Options
---
-
-# Configuration Options
-
-This guide details all configuration options available for OpenHands, helping you customize its behavior and integrate it with other services.
-
---
-
-# Table of Contents
-
-1. [Core Configuration](#core-configuration)
-   - [API Keys](#api-keys)
-   - [Workspace](#workspace)
-   - [Debugging and Logging](#debugging-and-logging)
-   - [Session Management](#session-management)
-   - [Trajectories](#trajectories)
-   - [File Store](#file-store)
-   - [Task Management](#task-management)
-   - [Sandbox Configuration](#sandbox-configuration)
-   - [Miscellaneous](#miscellaneous)
-2. [LLM Configuration](#llm-configuration)
-   - [AWS Credentials](#aws-credentials)
-   - [API Configuration](#api-configuration)
-   - [Custom LLM Provider](#custom-llm-provider)
-   - [Embeddings](#embeddings)
-   - [Message Handling](#message-handling)
-   - [Model Selection](#model-selection)
-   - [Retrying](#retrying)
-   - [Advanced Options](#advanced-options)
-3. [Agent Configuration](#agent-configuration)
-   - [Microagent Configuration](#microagent-configuration)
-   - [Memory Configuration](#memory-configuration)
-   - [LLM Configuration](#llm-configuration-2)
-   - [ActionSpace Configuration](#actionspace-configuration)
-   - [Microagent Usage](#microagent-usage)
-4. [Sandbox Configuration](#sandbox-configuration-2)
-   - [Execution](#execution)
-   - [Container Image](#container-image)
-   - [Networking](#networking)
-   - [Linting and Plugins](#linting-and-plugins)
-   - [Dependencies and Environment](#dependencies-and-environment)
-   - [Evaluation](#evaluation)
-5. [Security Configuration](#security-configuration)
-   - [Confirmation Mode](#confirmation-mode)
-   - [Security Analyzer](#security-analyzer)
-
-
---
-
-## Core Configuration
-The core configuration options are defined in the `[core]` section of the `config.toml` file.
-
-**API Keys**
- `e2b_api_key`
-  - Type: `str`
-  - Default: `""`
-  - Description: API key for E2B
-
- `modal_api_token_id`
-  - Type: `str` 
-  - Default: `""`
-  - Description: API token ID for Modal
-
- `modal_api_token_secret`
-  - Type: `str`
-  - Default: `""`
-  - Description: API token secret for Modal
-
-**Workspace**
- `workspace_base`
-  - Type: `str`
-  - Default: `"./workspace"`
-  - Description: Base path for the workspace
-
- `cache_dir`
-  - Type: `str`
-  - Default: `"/tmp/cache"`
-  - Description: Cache directory path
-
-**Debugging and Logging**
- `debug`
-  - Type: `bool`
-  - Default: `false`
-  - Description: Enable debugging
-
- `disable_color`
-  - Type: `bool`
-  - Default: `false`
-  - Description: Disable color in terminal output
-
-**Trajectories**
- `trajectories_path`
-  - Type: `str`
-  - Default: `"./trajectories"`
-  - Description: Path to store trajectories (can be a folder or a file). If it's a folder, the trajectories will be saved in a file named with the session id name and .json extension, in that folder.
-
-**File Store**
- `file_store_path`
-  - Type: `str`
-  - Default: `"/tmp/file_store"`
-  - Description: File store path
-
- `file_store`
-  - Type: `str`
-  - Default: `"memory"`
-  - Description: File store type
-
- `file_uploads_allowed_extensions`
-  - Type: `list of str`
-  - Default: `[".*"]`
-  - Description: List of allowed file extensions for uploads
-
- `file_uploads_max_file_size_mb`
-  - Type: `int`
-  - Default: `0`
-  - Description: Maximum file size for uploads, in megabytes
-
- `file_uploads_restrict_file_types`
-  - Type: `bool`
-  - Default: `false`
-  - Description: Restrict file types for file uploads
-
- `file_uploads_allowed_extensions`
-  - Type: `list of str`
-  - Default: `[".*"]`
-  - Description: List of allowed file extensions for uploads
-
-**Task Management**
- `max_budget_per_task`
-  - Type: `float`
-  - Default: `0.0`
-  - Description: Maximum budget per task (0.0 means no limit)
-
- `max_iterations`
-  - Type: `int`
-  - Default: `100`
-  - Description: Maximum number of iterations
-
-**Sandbox Configuration**
- `workspace_mount_path_in_sandbox`
-  - Type: `str`
-  - Default: `"/workspace"`
-  - Description: Path to mount the workspace in the sandbox
-
- `workspace_mount_path`
-  - Type: `str`
-  - Default: `""`
-  - Description: Path to mount the workspace
-
- `workspace_mount_rewrite`
-  - Type: `str`
-  - Default: `""`
-  - Description: Path to rewrite the workspace mount path to. You can usually ignore this, it refers to special cases of running inside another container.
-
-**Miscellaneous**
- `run_as_openhands`
-  - Type: `bool`
-  - Default: `true`
-  - Description: Run as OpenHands
-
- `runtime`
-  - Type: `str`
-  - Default: `"eventstream"`
-  - Description: Runtime environment
-
- `default_agent`
-  - Type: `str`
-  - Default: `"CodeActAgent"`
-  - Description: Name of the default agent
-
- `jwt_secret`
-  - Type: `str`
-  - Default: `uuid.uuid4().hex`
-  - Description: JWT secret for authentication. Please set it to your own value.
-
-## LLM Configuration
-The LLM (Large Language Model) configuration options are defined in the `[llm]` section of the `config.toml` file.
-
-**AWS Credentials**
- `aws_access_key_id`
-  - Type: `str`
-  - Default: `""`
-  - Description: AWS access key ID
-
- `aws_region_name`
-  - Type: `str`
-  - Default: `""`
-  - Description: AWS region name
-
- `aws_secret_access_key`
-  - Type: `str`
-  - Default: `""`
-  - Description: AWS secret access key
-
-**API Configuration**
- `api_key`
-  - Type: `str`
-  - Default: `None`
-  - Description: API key to use
-
- `base_url`
-  - Type: `str`
-  - Default: `""`
-  - Description: API base URL
-
- `api_version`
-  - Type: `str`
-  - Default: `""`
-  - Description: API version
-
- `input_cost_per_token`
-  - Type: `float`
-  - Default: `0.0`
-  - Description: Cost per input token
-
- `output_cost_per_token`
-  - Type: `float`
-  - Default: `0.0`
-  - Description: Cost per output token
-
-**Custom LLM Provider**
- `custom_llm_provider`
-  - Type: `str`
-  - Default: `""`
-  - Description: Custom LLM provider
-
-**Embeddings**
- `embedding_base_url`
-  - Type: `str`
-  - Default: `""`
-  - Description: Embedding API base URL
-
- `embedding_deployment_name`
-  - Type: `str`
-  - Default: `""`
-  - Description: Embedding deployment name
-
- `embedding_model`
-  - Type: `str`
-  - Default: `"local"`
-  - Description: Embedding model to use
-
-**Message Handling**
- `max_message_chars`
-  - Type: `int`
-  - Default: `30000`
-  - Description: The approximate maximum number of characters in the content of an event included in the prompt to the LLM. Larger observations are truncated.
-
- `max_input_tokens`
-  - Type: `int`
-  - Default: `0`
-  - Description: Maximum number of input tokens
-
- `max_output_tokens`
-  - Type: `int`
-  - Default: `0`
-  - Description: Maximum number of output tokens
-
-**Model Selection**
- `model`
-  - Type: `str`
-  - Default: `"claude-3-5-sonnet-20241022"`
-  - Description: Model to use
-
-**Retrying**
- `num_retries`
-  - Type: `int`
-  - Default: `8`
-  - Description: Number of retries to attempt
-
- `retry_max_wait`
-  - Type: `int`
-  - Default: `120`
-  - Description: Maximum wait time (in seconds) between retry attempts
-
- `retry_min_wait`
-  - Type: `int`
-  - Default: `15`
-  - Description: Minimum wait time (in seconds) between retry attempts
-
- `retry_multiplier`
-  - Type: `float`
-  - Default: `2.0`
-  - Description: Multiplier for exponential backoff calculation
-
-**Advanced Options**
- `drop_params`
-  - Type: `bool`
-  - Default: `false`
-  - Description: Drop any unmapped (unsupported) params without causing an exception
-
- `caching_prompt`
-  - Type: `bool`
-  - Default: `true`
-  - Description: Using the prompt caching feature if provided by the LLM and supported
-
- `ollama_base_url`
-  - Type: `str`
-  - Default: `""`
-  - Description: Base URL for the OLLAMA API
-
- `temperature`
-  - Type: `float`
-  - Default: `0.0`
-  - Description: Temperature for the API
-
- `timeout`
-  - Type: `int`
-  - Default: `0`
-  - Description: Timeout for the API
-
- `top_p`
-  - Type: `float`
-  - Default: `1.0`
-  - Description: Top p for the API
-
- `disable_vision`
-  - Type: `bool`
-  - Default: `None`
-  - Description: If model is vision capable, this option allows to disable image processing (useful for cost reduction)
-
-## Agent Configuration
-The agent configuration options are defined in the `[agent]` and `[agent.<agent_name>]` sections of the `config.toml` file.
-
-**Microagent Configuration**
- `micro_agent_name`
-  - Type: `str`
-  - Default: `""`
-  - Description: Name of the micro agent to use for this agent
-
-**Memory Configuration**
- `memory_enabled`
-  - Type: `bool`
-  - Default: `false`
-  - Description: Whether long-term memory (embeddings) is enabled
-
- `memory_max_threads`
-  - Type: `int`
-  - Default: `3`
-  - Description: The maximum number of threads indexing at the same time for embeddings
-
-**LLM Configuration**
- `llm_config`
-  - Type: `str`
-  - Default: `'your-llm-config-group'`
-  - Description: The name of the LLM config to use
-
-**ActionSpace Configuration**
- `function_calling`
-  - Type: `bool`
-  - Default: `true`
-  - Description: Whether function calling is enabled
-
- `codeact_enable_browsing`
-  - Type: `bool`
-  - Default: `false`
-  - Description: Whether browsing delegate is enabled in the action space (only works with function calling)
-
- `codeact_enable_llm_editor`
-  - Type: `bool`
-  - Default: `false`
-  - Description: Whether LLM editor is enabled in the action space (only works with function calling)
-
- `codeact_enable_jupyter`
-  - Type: `bool`
-  - Default: `false`
-  - Description: Whether Jupyter is enabled in the action space
-
-**Microagent Usage**
- `use_microagents`
-  - Type: `bool`
-  - Default: `true`
-  - Description: Whether to use microagents at all
-
- `disabled_microagents`
-  - Type: `list of str`
-  - Default: `None`
-  - Description: A list of microagents to disable
-
-## Sandbox Configuration
-The sandbox configuration options are defined in the `[sandbox]` section of the `config.toml` file.
-
-**Execution**
- `timeout`
-  - Type: `int`
-  - Default: `120`
-  - Description: Sandbox timeout in seconds
-
- `user_id`
-  - Type: `int`
-  - Default: `1000`
-  - Description: Sandbox user ID
-
-**Container Image**
- `base_container_image`
-  - Type: `str`
-  - Default: `"nikolaik/python-nodejs:python3.12-nodejs22"`
-  - Description: Container image to use for the sandbox
-
-**Networking**
- `use_host_network`
-  - Type: `bool`
-  - Default: `false`
-  - Description: Use host network
-
-**Linting and Plugins**
- `enable_auto_lint`
-  - Type: `bool`
-  - Default: `false`
-  - Description: Enable auto linting after editing
-
- `initialize_plugins`
-  - Type: `bool`
-  - Default: `true`
-  - Description: Whether to initialize plugins
-
-**Dependencies and Environment**
- `runtime_extra_deps`
-  - Type: `str`
-  - Default: `""`
-  - Description: Extra dependencies to install in the runtime image
-
- `runtime_startup_env_vars`
-  - Type: `dict`
-  - Default: `{}`
-  - Description: Environment variables to set at the launch of the runtime
-
-**Evaluation**
- `browsergym_eval_env`
-  - Type: `str`
-  - Default: `""`
-  - Description: BrowserGym environment to use for evaluation
-
-## Security Configuration
-The security configuration options are defined in the `[security]` section of the `config.toml` file.
-
-**Confirmation Mode**
- `confirmation_mode`
-  - Type: `bool`
-  - Default: `false`
-  - Description: Enable confirmation mode
-
-**Security Analyzer**
- `security_analyzer`
-  - Type: `str`
-  - Default: `""`
-  - Description: The security analyzer to use
-
-
---
-
-> **Note**: Adjust configurations carefully, especially for memory, security, and network-related settings to ensure optimal performance and security.
-Please note that the configuration options may be subject to change in future versions of OpenHands. It's recommended to refer to the official documentation for the most up-to-date information.
-
-
--- a/docs/modules/usage/how-to/evaluation-harness.md
+++ b/docs/modules/usage/how-to/evaluation-harness.md
@@ -73,7 +73,7 @@ The `run_controller()` function is the core of OpenHands's execution. It manages

 ## Easiest way to get started: Exploring Existing Benchmarks

-We encourage you to review the various evaluation benchmarks available in the [`evaluation/benchmarks/` directory](https://github.com/All-Hands-AI/OpenHands/blob/main/evaluation/benchmarks) of our repository.
+We encourage you to review the various evaluation benchmarks available in the [`evaluation/` directory](https://github.com/All-Hands-AI/OpenHands/blob/main/evaluation) of our repository.

 To integrate your own benchmark, we suggest starting with the one that most closely resembles your needs. This approach can significantly streamline your integration process, allowing you to build upon existing structures and adapt them to your specific requirements.

--- a/docs/package-lock.json
+++ b/docs/package-lock.json
@@ -8,10 +8,10 @@
      "name": "docs",
      "version": "0.0.0",
      "dependencies": {
-        "@docusaurus/core": "^3.6.3",
-        "@docusaurus/plugin-content-pages": "^3.6.3",
-        "@docusaurus/preset-classic": "^3.6.3",
-        "@docusaurus/theme-mermaid": "^3.6.3",
+        "@docusaurus/core": "^3.6.2",
+        "@docusaurus/plugin-content-pages": "^3.6.2",
+        "@docusaurus/preset-classic": "^3.6.2",
+        "@docusaurus/theme-mermaid": "^3.6.2",
        "@mdx-js/react": "^3.1.0",
        "clsx": "^2.0.0",
        "prism-react-renderer": "^2.4.0",
@@ -22,7 +22,7 @@
      },
      "devDependencies": {
        "@docusaurus/module-type-aliases": "^3.5.1",
-        "@docusaurus/tsconfig": "^3.6.3",
+        "@docusaurus/tsconfig": "^3.6.2",
        "@docusaurus/types": "^3.5.1",
        "typescript": "~5.6.3"
      },
@@ -3168,9 +3168,9 @@
      }
    },
    "node_modules/@docusaurus/babel": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/babel/-/babel-3.6.3.tgz",
-      "integrity": "sha512-7dW9Hat9EHYCVicFXYA4hjxBY38+hPuCURL8oRF9fySRm7vzNWuEOghA1TXcykuXZp0HLG2td4RhDxCvGG7tNw==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/babel/-/babel-3.6.2.tgz",
+      "integrity": "sha512-v8N8TWGXDsb5sxQC3Rcqb1CZr0LlU1OgqqVBUchN6cpIUr7EJuVJs5eHcIu5Ag8mwO/hWN3f7FE9uaHTMapAbg==",
      "dependencies": {
        "@babel/core": "^7.25.9",
        "@babel/generator": "^7.25.9",
@@ -3182,8 +3182,8 @@
        "@babel/runtime": "^7.25.9",
        "@babel/runtime-corejs3": "^7.25.9",
        "@babel/traverse": "^7.25.9",
-        "@docusaurus/logger": "3.6.3",
-        "@docusaurus/utils": "3.6.3",
+        "@docusaurus/logger": "3.6.2",
+        "@docusaurus/utils": "3.6.2",
        "babel-plugin-dynamic-import-node": "^2.3.3",
        "fs-extra": "^11.1.1",
        "tslib": "^2.6.0"
@@ -3193,16 +3193,16 @@
      }
    },
    "node_modules/@docusaurus/bundler": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/bundler/-/bundler-3.6.3.tgz",
-      "integrity": "sha512-47JLuc8D4wA+6VOvmMd5fUC9rFppBQpQOnxDYiVXffm/DeV/wmm3sbpNd5Y+O+G2+nevLTRnvCm/qyancv0Y3A==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/bundler/-/bundler-3.6.2.tgz",
+      "integrity": "sha512-YkEifEVs4lV931SrHBB4n6WqRowMw+aM/QPH3z8aU+5t1dWa+1p2OPqARS+tSbh3la9ns+L1zIfSbd8RHi2/PQ==",
      "dependencies": {
        "@babel/core": "^7.25.9",
-        "@docusaurus/babel": "3.6.3",
-        "@docusaurus/cssnano-preset": "3.6.3",
-        "@docusaurus/logger": "3.6.3",
-        "@docusaurus/types": "3.6.3",
-        "@docusaurus/utils": "3.6.3",
+        "@docusaurus/babel": "3.6.2",
+        "@docusaurus/cssnano-preset": "3.6.2",
+        "@docusaurus/logger": "3.6.2",
+        "@docusaurus/types": "3.6.2",
+        "@docusaurus/utils": "3.6.2",
        "babel-loader": "^9.2.1",
        "clean-css": "^5.3.2",
        "copy-webpack-plugin": "^11.0.0",
@@ -3236,17 +3236,17 @@
      }
    },
    "node_modules/@docusaurus/core": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/core/-/core-3.6.3.tgz",
-      "integrity": "sha512-xL7FRY9Jr5DWqB6pEnqgKqcMPJOX5V0pgWXi5lCiih11sUBmcFKM7c3+GyxcVeeWFxyYSDP3grLTWqJoP4P9Vw==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/core/-/core-3.6.2.tgz",
+      "integrity": "sha512-irMts/mGLZv8dWcy0WUtbY/U6b5qIfHgQd1/kXMyAxUJo99fL0wFSqhMI+tcxjk0HYy427MXerLMqFJj+Arg1w==",
      "dependencies": {
-        "@docusaurus/babel": "3.6.3",
-        "@docusaurus/bundler": "3.6.3",
-        "@docusaurus/logger": "3.6.3",
-        "@docusaurus/mdx-loader": "3.6.3",
-        "@docusaurus/utils": "3.6.3",
-        "@docusaurus/utils-common": "3.6.3",
-        "@docusaurus/utils-validation": "3.6.3",
+        "@docusaurus/babel": "3.6.2",
+        "@docusaurus/bundler": "3.6.2",
+        "@docusaurus/logger": "3.6.2",
+        "@docusaurus/mdx-loader": "3.6.2",
+        "@docusaurus/utils": "3.6.2",
+        "@docusaurus/utils-common": "3.6.2",
+        "@docusaurus/utils-validation": "3.6.2",
        "boxen": "^6.2.1",
        "chalk": "^4.1.2",
        "chokidar": "^3.5.3",
@@ -3310,9 +3310,9 @@
      }
    },
    "node_modules/@docusaurus/cssnano-preset": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/cssnano-preset/-/cssnano-preset-3.6.3.tgz",
-      "integrity": "sha512-qP7SXrwZ+23GFJdPN4aIHQrZW+oH/7tzwEuc/RNL0+BdZdmIjYQqUxdXsjE4lFxLNZjj0eUrSNYIS6xwfij+5Q==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/cssnano-preset/-/cssnano-preset-3.6.2.tgz",
+      "integrity": "sha512-mBkVa4QMHRwCFCVLYdBlOZuAT1iVVsS7GGSgliSVAeTOagP/AbtlBsCVrBs+keEuDuRF1w/6QEcqDoZe9fa5pw==",
      "dependencies": {
        "cssnano-preset-advanced": "^6.1.2",
        "postcss": "^8.4.38",
@@ -3324,9 +3324,9 @@
      }
    },
    "node_modules/@docusaurus/logger": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/logger/-/logger-3.6.3.tgz",
-      "integrity": "sha512-xSubJixcNyMV9wMV4q0s47CBz3Rlc5jbcCCuij8pfQP8qn/DIpt0ks8W6hQWzHAedg/J/EwxxUOUrnEoKzJo8g==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/logger/-/logger-3.6.2.tgz",
+      "integrity": "sha512-1p4IQhhgLyIfsey4UAdAIW69aUE1Ei6O91Nsw30ryZeDWSG5dh4o3zaRGOLxfAX69Ac/yDm6YCwJOafUxL6Vxg==",
      "dependencies": {
        "chalk": "^4.1.2",
        "tslib": "^2.6.0"
@@ -3336,13 +3336,13 @@
      }
    },
    "node_modules/@docusaurus/mdx-loader": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/mdx-loader/-/mdx-loader-3.6.3.tgz",
-      "integrity": "sha512-3iJdiDz9540ppBseeI93tWTDtUGVkxzh59nMq4ignylxMuXBLK8dFqVeaEor23v1vx6TrGKZ2FuLaTB+U7C0QQ==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/mdx-loader/-/mdx-loader-3.6.2.tgz",
+      "integrity": "sha512-7fbRmNgF3CR96Ja82Ya0/Cdu1OL9UJ/22llNMY8lr5gAbw718Y5ryXMVRIYn0JNLTiSxzgtvW4DIsUWEB8NMpw==",
      "dependencies": {
-        "@docusaurus/logger": "3.6.3",
-        "@docusaurus/utils": "3.6.3",
-        "@docusaurus/utils-validation": "3.6.3",
+        "@docusaurus/logger": "3.6.2",
+        "@docusaurus/utils": "3.6.2",
+        "@docusaurus/utils-validation": "3.6.2",
        "@mdx-js/mdx": "^3.0.0",
        "@slorber/remark-comment": "^1.0.0",
        "escape-html": "^1.0.3",
@@ -3374,11 +3374,11 @@
      }
    },
    "node_modules/@docusaurus/module-type-aliases": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/module-type-aliases/-/module-type-aliases-3.6.3.tgz",
-      "integrity": "sha512-MjaXX9PN/k5ugNvfRZdWyKWq4FsrhN4LEXaj0pEmMebJuBNlFeGyKQUa9DRhJHpadNaiMLrbo9m3U7Ig5YlsZg==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/module-type-aliases/-/module-type-aliases-3.6.2.tgz",
+      "integrity": "sha512-NrJkL2rLTCjHtWOqUvWzwqvJrsKLj0gVJeV6q5yeKdKKgItietcTf2fTRkM9LHKSUN8CBDXxwHABeQvTahvmXQ==",
      "dependencies": {
-        "@docusaurus/types": "3.6.3",
+        "@docusaurus/types": "3.6.2",
        "@types/history": "^4.7.11",
        "@types/react": "*",
        "@types/react-router-config": "*",
@@ -3392,18 +3392,18 @@
      }
    },
    "node_modules/@docusaurus/plugin-content-blog": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/plugin-content-blog/-/plugin-content-blog-3.6.3.tgz",
-      "integrity": "sha512-k0ogWwwJU3pFRFfvW1kRVHxzf2DutLGaaLjAnHVEU6ju+aRP0Z5ap/13DHyPOfHeE4WKpn/M0TqjdwZAcY3kAw==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/plugin-content-blog/-/plugin-content-blog-3.6.2.tgz",
+      "integrity": "sha512-6bJxr6Or4NslEVH3BJuPH30kUWiqUjDRdGPhvxpHmt9W/RY2/6u72WICG3bW3dLFxJ/2uDLBU92lHnatpvo7Ew==",
      "dependencies": {
-        "@docusaurus/core": "3.6.3",
-        "@docusaurus/logger": "3.6.3",
-        "@docusaurus/mdx-loader": "3.6.3",
-        "@docusaurus/theme-common": "3.6.3",
-        "@docusaurus/types": "3.6.3",
-        "@docusaurus/utils": "3.6.3",
-        "@docusaurus/utils-common": "3.6.3",
-        "@docusaurus/utils-validation": "3.6.3",
+        "@docusaurus/core": "3.6.2",
+        "@docusaurus/logger": "3.6.2",
+        "@docusaurus/mdx-loader": "3.6.2",
+        "@docusaurus/theme-common": "3.6.2",
+        "@docusaurus/types": "3.6.2",
+        "@docusaurus/utils": "3.6.2",
+        "@docusaurus/utils-common": "3.6.2",
+        "@docusaurus/utils-validation": "3.6.2",
        "cheerio": "1.0.0-rc.12",
        "feed": "^4.2.2",
        "fs-extra": "^11.1.1",
@@ -3425,19 +3425,19 @@
      }
    },
    "node_modules/@docusaurus/plugin-content-docs": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/plugin-content-docs/-/plugin-content-docs-3.6.3.tgz",
-      "integrity": "sha512-r2wS8y/fsaDcxkm20W5bbYJFPzdWdEaTWVYjNxlHlcmX086eqQR1Fomlg9BHTJ0dLXPzAlbC8EN4XqMr3QzNCQ==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/plugin-content-docs/-/plugin-content-docs-3.6.2.tgz",
+      "integrity": "sha512-e6WW1g10RIXXLN/rrtqTi/FyJ1Hj3X9Mmgz4V11/0pDCxIGGI8m4ocbAglUlLtgvbLD5viNLefl/NwbOW3JXiQ==",
      "dependencies": {
-        "@docusaurus/core": "3.6.3",
-        "@docusaurus/logger": "3.6.3",
-        "@docusaurus/mdx-loader": "3.6.3",
-        "@docusaurus/module-type-aliases": "3.6.3",
-        "@docusaurus/theme-common": "3.6.3",
-        "@docusaurus/types": "3.6.3",
-        "@docusaurus/utils": "3.6.3",
-        "@docusaurus/utils-common": "3.6.3",
-        "@docusaurus/utils-validation": "3.6.3",
+        "@docusaurus/core": "3.6.2",
+        "@docusaurus/logger": "3.6.2",
+        "@docusaurus/mdx-loader": "3.6.2",
+        "@docusaurus/module-type-aliases": "3.6.2",
+        "@docusaurus/theme-common": "3.6.2",
+        "@docusaurus/types": "3.6.2",
+        "@docusaurus/utils": "3.6.2",
+        "@docusaurus/utils-common": "3.6.2",
+        "@docusaurus/utils-validation": "3.6.2",
        "@types/react-router-config": "^5.0.7",
        "combine-promises": "^1.1.0",
        "fs-extra": "^11.1.1",
@@ -3456,15 +3456,15 @@
      }
    },
    "node_modules/@docusaurus/plugin-content-pages": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/plugin-content-pages/-/plugin-content-pages-3.6.3.tgz",
-      "integrity": "sha512-eHrmTgjgLZsuqfsYr5X2xEwyIcck0wseSofWrjTwT9FLOWp+KDmMAuVK+wRo7sFImWXZk3oV/xX/g9aZrhD7OA==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/plugin-content-pages/-/plugin-content-pages-3.6.2.tgz",
+      "integrity": "sha512-fo4NyGkw10lYHyHaTxE6TZLYnxNtCfRHeZkNK1N9pBYqe7TT2dBUNAEeVW2U3ed9m6YuB7JKSQsa++GGmcP+6g==",
      "dependencies": {
-        "@docusaurus/core": "3.6.3",
-        "@docusaurus/mdx-loader": "3.6.3",
-        "@docusaurus/types": "3.6.3",
-        "@docusaurus/utils": "3.6.3",
-        "@docusaurus/utils-validation": "3.6.3",
+        "@docusaurus/core": "3.6.2",
+        "@docusaurus/mdx-loader": "3.6.2",
+        "@docusaurus/types": "3.6.2",
+        "@docusaurus/utils": "3.6.2",
+        "@docusaurus/utils-validation": "3.6.2",
        "fs-extra": "^11.1.1",
        "tslib": "^2.6.0",
        "webpack": "^5.88.1"
@@ -3478,13 +3478,13 @@
      }
    },
    "node_modules/@docusaurus/plugin-debug": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/plugin-debug/-/plugin-debug-3.6.3.tgz",
-      "integrity": "sha512-zB9GXfIZNPRfzKnNjU6xGVrqn9bPXuGhpjgsuc/YtcTDjnjhasg38NdYd5LEqXex5G/zIorQgWB3n6x/Ut62vQ==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/plugin-debug/-/plugin-debug-3.6.2.tgz",
+      "integrity": "sha512-T/eS3VvHElpeV5S8uwp7Si4ujEynmgFtJLvA2CSa5pzQuOF1EEghF9nekAIj0cWtDHsqNUDZNr8hK1brivFXSg==",
      "dependencies": {
-        "@docusaurus/core": "3.6.3",
-        "@docusaurus/types": "3.6.3",
-        "@docusaurus/utils": "3.6.3",
+        "@docusaurus/core": "3.6.2",
+        "@docusaurus/types": "3.6.2",
+        "@docusaurus/utils": "3.6.2",
        "fs-extra": "^11.1.1",
        "react-json-view-lite": "^1.2.0",
        "tslib": "^2.6.0"
@@ -3498,13 +3498,13 @@
      }
    },
    "node_modules/@docusaurus/plugin-google-analytics": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/plugin-google-analytics/-/plugin-google-analytics-3.6.3.tgz",
-      "integrity": "sha512-rCDNy1QW8Dag7nZq67pcum0bpFLrwvxJhYuVprhFh8BMBDxV0bY+bAkGHbSf68P3Bk9C3hNOAXX1srGLIDvcTA==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/plugin-google-analytics/-/plugin-google-analytics-3.6.2.tgz",
+      "integrity": "sha512-B7ihrr3wz8e4XqW+dIAtq844u3Z83u5CeiL1xrCqzFH+vDCjUZHTamS3zKXNcgi6YVVe6hUQXPG15ltaqQaVPQ==",
      "dependencies": {
-        "@docusaurus/core": "3.6.3",
-        "@docusaurus/types": "3.6.3",
-        "@docusaurus/utils-validation": "3.6.3",
+        "@docusaurus/core": "3.6.2",
+        "@docusaurus/types": "3.6.2",
+        "@docusaurus/utils-validation": "3.6.2",
        "tslib": "^2.6.0"
      },
      "engines": {
@@ -3516,13 +3516,13 @@
      }
    },
    "node_modules/@docusaurus/plugin-google-gtag": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/plugin-google-gtag/-/plugin-google-gtag-3.6.3.tgz",
-      "integrity": "sha512-+OyDvhM6rqVkQOmLVkQWVJAizEEfkPzVWtIHXlWPOCFGK9X4/AWeBSrU0WG4iMg9Z4zD4YDRrU+lvI4s6DSC+w==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/plugin-google-gtag/-/plugin-google-gtag-3.6.2.tgz",
+      "integrity": "sha512-V8ijI6qddAAkJ0vd8sjZ7S/apRTLJn9dAwvj/rSMd93witGdKINemL+9TyfLkhcXKTxyqRT8zKdu8ewjPXqKHg==",
      "dependencies": {
-        "@docusaurus/core": "3.6.3",
-        "@docusaurus/types": "3.6.3",
-        "@docusaurus/utils-validation": "3.6.3",
+        "@docusaurus/core": "3.6.2",
+        "@docusaurus/types": "3.6.2",
+        "@docusaurus/utils-validation": "3.6.2",
        "@types/gtag.js": "^0.0.12",
        "tslib": "^2.6.0"
      },
@@ -3535,13 +3535,13 @@
      }
    },
    "node_modules/@docusaurus/plugin-google-tag-manager": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/plugin-google-tag-manager/-/plugin-google-tag-manager-3.6.3.tgz",
-      "integrity": "sha512-1M6UPB13gWUtN2UHX083/beTn85PlRI9ABItTl/JL1FJ5dJTWWFXXsHf9WW/6hrVwthwTeV/AGbGKvLKV+IlCA==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/plugin-google-tag-manager/-/plugin-google-tag-manager-3.6.2.tgz",
+      "integrity": "sha512-fnWQ5FdN9f8c8VTgjaQ98208Y+d/JjHhD506rWIIL9rt1cJOf29XElxvOeKpMJadfkgY5KLZSAiHkGt+4qgN4g==",
      "dependencies": {
-        "@docusaurus/core": "3.6.3",
-        "@docusaurus/types": "3.6.3",
-        "@docusaurus/utils-validation": "3.6.3",
+        "@docusaurus/core": "3.6.2",
+        "@docusaurus/types": "3.6.2",
+        "@docusaurus/utils-validation": "3.6.2",
        "tslib": "^2.6.0"
      },
      "engines": {
@@ -3553,16 +3553,16 @@
      }
    },
    "node_modules/@docusaurus/plugin-sitemap": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/plugin-sitemap/-/plugin-sitemap-3.6.3.tgz",
-      "integrity": "sha512-94qOO4M9Fwv9KfVQJsgbe91k+fPJ4byf1L3Ez8TUa6TAFPo/BrLwQ80zclHkENlL1824TuxkcMKv33u6eydQCg==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/plugin-sitemap/-/plugin-sitemap-3.6.2.tgz",
+      "integrity": "sha512-qcAQAP1Ot0dZpeRoJ0L/Zck5FVDkll2IleVZQLzxeRVDZIw1P9/TK7/Aw1w2pmH7dmw/Cwk/cLSVRvLAmp9k7A==",
      "dependencies": {
-        "@docusaurus/core": "3.6.3",
-        "@docusaurus/logger": "3.6.3",
-        "@docusaurus/types": "3.6.3",
-        "@docusaurus/utils": "3.6.3",
-        "@docusaurus/utils-common": "3.6.3",
-        "@docusaurus/utils-validation": "3.6.3",
+        "@docusaurus/core": "3.6.2",
+        "@docusaurus/logger": "3.6.2",
+        "@docusaurus/types": "3.6.2",
+        "@docusaurus/utils": "3.6.2",
+        "@docusaurus/utils-common": "3.6.2",
+        "@docusaurus/utils-validation": "3.6.2",
        "fs-extra": "^11.1.1",
        "sitemap": "^7.1.1",
        "tslib": "^2.6.0"
@@ -3576,23 +3576,23 @@
      }
    },
    "node_modules/@docusaurus/preset-classic": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/preset-classic/-/preset-classic-3.6.3.tgz",
-      "integrity": "sha512-VHSYWROT3flvNNI1SrnMOtW1EsjeHNK9dhU6s9eY5hryZe79lUqnZJyze/ymDe2LXAqzyj6y5oYvyBoZZk6ErA==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/preset-classic/-/preset-classic-3.6.2.tgz",
+      "integrity": "sha512-r2n5eHdhiNSrJGsrrYcw+WsyStmXxe0ZG3RdA9LVyK5+jBHM8blrUWJEDugnzCNbyhUzhdtcmgCC9fhdAvKuQw==",
      "dependencies": {
-        "@docusaurus/core": "3.6.3",
-        "@docusaurus/plugin-content-blog": "3.6.3",
-        "@docusaurus/plugin-content-docs": "3.6.3",
-        "@docusaurus/plugin-content-pages": "3.6.3",
-        "@docusaurus/plugin-debug": "3.6.3",
-        "@docusaurus/plugin-google-analytics": "3.6.3",
-        "@docusaurus/plugin-google-gtag": "3.6.3",
-        "@docusaurus/plugin-google-tag-manager": "3.6.3",
-        "@docusaurus/plugin-sitemap": "3.6.3",
-        "@docusaurus/theme-classic": "3.6.3",
-        "@docusaurus/theme-common": "3.6.3",
-        "@docusaurus/theme-search-algolia": "3.6.3",
-        "@docusaurus/types": "3.6.3"
+        "@docusaurus/core": "3.6.2",
+        "@docusaurus/plugin-content-blog": "3.6.2",
+        "@docusaurus/plugin-content-docs": "3.6.2",
+        "@docusaurus/plugin-content-pages": "3.6.2",
+        "@docusaurus/plugin-debug": "3.6.2",
+        "@docusaurus/plugin-google-analytics": "3.6.2",
+        "@docusaurus/plugin-google-gtag": "3.6.2",
+        "@docusaurus/plugin-google-tag-manager": "3.6.2",
+        "@docusaurus/plugin-sitemap": "3.6.2",
+        "@docusaurus/theme-classic": "3.6.2",
+        "@docusaurus/theme-common": "3.6.2",
+        "@docusaurus/theme-search-algolia": "3.6.2",
+        "@docusaurus/types": "3.6.2"
      },
      "engines": {
        "node": ">=18.0"
@@ -3603,23 +3603,23 @@
      }
    },
    "node_modules/@docusaurus/theme-classic": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/theme-classic/-/theme-classic-3.6.3.tgz",
-      "integrity": "sha512-1RRLK1tSArI2c00qugWYO3jRocjOZwGF1mBzPPylDVRwWCS/rnWWR91ChdbbaxIupRJ+hX8ZBYrwr5bbU0oztQ==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/theme-classic/-/theme-classic-3.6.2.tgz",
+      "integrity": "sha512-bCdOPqPNezhLx+hgNVO2Cf+8/1AHa9uHDOqTx/CKAx2I0J/jV9G+6JiMtpSRKGNfBoLT1O+56/7+WtkOf54xTw==",
      "dependencies": {
-        "@docusaurus/core": "3.6.3",
-        "@docusaurus/logger": "3.6.3",
-        "@docusaurus/mdx-loader": "3.6.3",
-        "@docusaurus/module-type-aliases": "3.6.3",
-        "@docusaurus/plugin-content-blog": "3.6.3",
-        "@docusaurus/plugin-content-docs": "3.6.3",
-        "@docusaurus/plugin-content-pages": "3.6.3",
-        "@docusaurus/theme-common": "3.6.3",
-        "@docusaurus/theme-translations": "3.6.3",
-        "@docusaurus/types": "3.6.3",
-        "@docusaurus/utils": "3.6.3",
-        "@docusaurus/utils-common": "3.6.3",
-        "@docusaurus/utils-validation": "3.6.3",
+        "@docusaurus/core": "3.6.2",
+        "@docusaurus/logger": "3.6.2",
+        "@docusaurus/mdx-loader": "3.6.2",
+        "@docusaurus/module-type-aliases": "3.6.2",
+        "@docusaurus/plugin-content-blog": "3.6.2",
+        "@docusaurus/plugin-content-docs": "3.6.2",
+        "@docusaurus/plugin-content-pages": "3.6.2",
+        "@docusaurus/theme-common": "3.6.2",
+        "@docusaurus/theme-translations": "3.6.2",
+        "@docusaurus/types": "3.6.2",
+        "@docusaurus/utils": "3.6.2",
+        "@docusaurus/utils-common": "3.6.2",
+        "@docusaurus/utils-validation": "3.6.2",
        "@mdx-js/react": "^3.0.0",
        "clsx": "^2.0.0",
        "copy-text-to-clipboard": "^3.2.0",
@@ -3643,14 +3643,14 @@
      }
    },
    "node_modules/@docusaurus/theme-common": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/theme-common/-/theme-common-3.6.3.tgz",
-      "integrity": "sha512-b8ZkhczXHDxWWyvz+YJy4t/PlPbEogTTbgnHoflYnH7rmRtyoodTsu8WVM12la5LmlMJBclBXFl29OH8kPE7gg==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/theme-common/-/theme-common-3.6.2.tgz",
+      "integrity": "sha512-lfgsL064KEHpCkgGUc0OYoUPCpYfzggp6Hof8sz59UuKiLvb/Z7raewE9/NfocrJ2HZI17rLgMX3SQlRDh/5gg==",
      "dependencies": {
-        "@docusaurus/mdx-loader": "3.6.3",
-        "@docusaurus/module-type-aliases": "3.6.3",
-        "@docusaurus/utils": "3.6.3",
-        "@docusaurus/utils-common": "3.6.3",
+        "@docusaurus/mdx-loader": "3.6.2",
+        "@docusaurus/module-type-aliases": "3.6.2",
+        "@docusaurus/utils": "3.6.2",
+        "@docusaurus/utils-common": "3.6.2",
        "@types/history": "^4.7.11",
        "@types/react": "*",
        "@types/react-router-config": "*",
@@ -3670,15 +3670,15 @@
      }
    },
    "node_modules/@docusaurus/theme-mermaid": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/theme-mermaid/-/theme-mermaid-3.6.3.tgz",
-      "integrity": "sha512-kIqpjNCP/9R2GGf8UmiDxD3CkOAEJuJIEFlaKMgQtjVxa/vH+9PLI1+DFbArGoG4+0ENTYUq8phHPW7SeL36uQ==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/theme-mermaid/-/theme-mermaid-3.6.2.tgz",
+      "integrity": "sha512-Ui+rBtqMPKj3RCOxNlY04i1tEjNg+fZg4URTvkHmYR07hcKaJw+vkw+wlaYjd0HFZk+3Er9vUAcwsCWuea4cVQ==",
      "dependencies": {
-        "@docusaurus/core": "3.6.3",
-        "@docusaurus/module-type-aliases": "3.6.3",
-        "@docusaurus/theme-common": "3.6.3",
-        "@docusaurus/types": "3.6.3",
-        "@docusaurus/utils-validation": "3.6.3",
+        "@docusaurus/core": "3.6.2",
+        "@docusaurus/module-type-aliases": "3.6.2",
+        "@docusaurus/theme-common": "3.6.2",
+        "@docusaurus/types": "3.6.2",
+        "@docusaurus/utils-validation": "3.6.2",
        "mermaid": ">=10.4",
        "tslib": "^2.6.0"
      },
@@ -3691,18 +3691,18 @@
      }
    },
    "node_modules/@docusaurus/theme-search-algolia": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/theme-search-algolia/-/theme-search-algolia-3.6.3.tgz",
-      "integrity": "sha512-rt+MGCCpYgPyWCGXtbxlwFbTSobu15jWBTPI2LHsHNa5B0zSmOISX6FWYAPt5X1rNDOqMGM0FATnh7TBHRohVA==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/theme-search-algolia/-/theme-search-algolia-3.6.2.tgz",
+      "integrity": "sha512-SFLS+Rq8Cg2yepnHucA9sRpIR97yHvZWlCgMzBLunV3KHbB6hD2h5HPhFV39wYHYCjJUAOH1lX9poJ1qKYuSvg==",
      "dependencies": {
        "@docsearch/react": "^3.5.2",
-        "@docusaurus/core": "3.6.3",
-        "@docusaurus/logger": "3.6.3",
-        "@docusaurus/plugin-content-docs": "3.6.3",
-        "@docusaurus/theme-common": "3.6.3",
-        "@docusaurus/theme-translations": "3.6.3",
-        "@docusaurus/utils": "3.6.3",
-        "@docusaurus/utils-validation": "3.6.3",
+        "@docusaurus/core": "3.6.2",
+        "@docusaurus/logger": "3.6.2",
+        "@docusaurus/plugin-content-docs": "3.6.2",
+        "@docusaurus/theme-common": "3.6.2",
+        "@docusaurus/theme-translations": "3.6.2",
+        "@docusaurus/utils": "3.6.2",
+        "@docusaurus/utils-validation": "3.6.2",
        "algoliasearch": "^4.18.0",
        "algoliasearch-helper": "^3.13.3",
        "clsx": "^2.0.0",
@@ -3721,9 +3721,9 @@
      }
    },
    "node_modules/@docusaurus/theme-translations": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/theme-translations/-/theme-translations-3.6.3.tgz",
-      "integrity": "sha512-Gb0regclToVlngSIIwUCtBMQBq48qVUaN1XQNKW4XwlsgUyk0vP01LULdqbem7czSwIeBAFXFoORJ0RPX7ht/w==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/theme-translations/-/theme-translations-3.6.2.tgz",
+      "integrity": "sha512-LIWrYoDUsOTKmb0c7IQzawiPUTAaczBs5IOx6srxOWoTHVUMLzJCkl5Y6whfuRrnul8G05qv2vk238bN5Ko62g==",
      "dependencies": {
        "fs-extra": "^11.1.1",
        "tslib": "^2.6.0"
@@ -3733,15 +3733,15 @@
      }
    },
    "node_modules/@docusaurus/tsconfig": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/tsconfig/-/tsconfig-3.6.3.tgz",
-      "integrity": "sha512-1pT/rTrRpMV15E4tJH95W5PrjboMn5JkKF+Ys8cTjMegetiXjs0gPFOSDA5hdTlberKQLDO50xPjMJHondLuzA==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/tsconfig/-/tsconfig-3.6.2.tgz",
+      "integrity": "sha512-TWLkyYHBYhIJNcXCEc3D1M9R8UFV4IZ82rGef5U9mE1ZrcgDUlZxYaYdoSuHrPrzPRIl3orjmpscO2FAk2gdZw==",
      "dev": true
    },
    "node_modules/@docusaurus/types": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/types/-/types-3.6.3.tgz",
-      "integrity": "sha512-xD9oTGDrouWzefkhe9ogB2fDV96/82cRpNGx2HIvI5L87JHNhQVIWimQ/3JIiiX/TEd5S9s+VO6FFguwKNRVow==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/types/-/types-3.6.2.tgz",
+      "integrity": "sha512-117Wsk6xXrWEAsCYCXS3TGJv5tkdIZDcd7T/V0UJvKYmY0gyVPPcEQChy8yTdjbIkbB2q4fa7Jpox72Qv86mqQ==",
      "dependencies": {
        "@mdx-js/mdx": "^3.0.0",
        "@types/history": "^4.7.11",
@@ -3759,13 +3759,13 @@
      }
    },
    "node_modules/@docusaurus/utils": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/utils/-/utils-3.6.3.tgz",
-      "integrity": "sha512-0R/FR3bKVl4yl8QwbL4TYFfR+OXBRpVUaTJdENapBGR3YMwfM6/JnhGilWQO8AOwPJGtGoDK7ib8+8UF9f3OZQ==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/utils/-/utils-3.6.2.tgz",
+      "integrity": "sha512-oxnpUcFZGE3uPCDoXr8GJriB3VWM9sFjPedFidX3Fsz87l1NZNc1wtbKPfQ7GYFDMYo2IGlAv5+47Me9RkM6lg==",
      "dependencies": {
-        "@docusaurus/logger": "3.6.3",
-        "@docusaurus/types": "3.6.3",
-        "@docusaurus/utils-common": "3.6.3",
+        "@docusaurus/logger": "3.6.2",
+        "@docusaurus/types": "3.6.2",
+        "@docusaurus/utils-common": "3.6.2",
        "@svgr/webpack": "^8.1.0",
        "escape-string-regexp": "^4.0.0",
        "file-loader": "^6.2.0",
@@ -3790,11 +3790,11 @@
      }
    },
    "node_modules/@docusaurus/utils-common": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/utils-common/-/utils-common-3.6.3.tgz",
-      "integrity": "sha512-v4nKDaANLgT3pMBewHYEMAl/ufY0LkXao1QkFWzI5huWFOmNQ2UFzv2BiKeHX5Ownis0/w6cAyoxPhVdDonlSQ==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/utils-common/-/utils-common-3.6.2.tgz",
+      "integrity": "sha512-dr5wK+OyU2QAWxG7S5siD2bPgS7+ZeqWHfgLNHZ5yalaZf8TbeNNLqydfngukAY56BGZN0NbMkX6jGIr7ZF0sA==",
      "dependencies": {
-        "@docusaurus/types": "3.6.3",
+        "@docusaurus/types": "3.6.2",
        "tslib": "^2.6.0"
      },
      "engines": {
@@ -3802,13 +3802,13 @@
      }
    },
    "node_modules/@docusaurus/utils-validation": {
-      "version": "3.6.3",
-      "resolved": "https://registry.npmjs.org/@docusaurus/utils-validation/-/utils-validation-3.6.3.tgz",
-      "integrity": "sha512-bhEGGiN5BE38h21vjqD70Gxg++j+PfYVddDUE5UFvLDup68QOcpD33CLr+2knPorlxRbEaNfz6HQDUMQ3HuqKw==",
+      "version": "3.6.2",
+      "resolved": "https://registry.npmjs.org/@docusaurus/utils-validation/-/utils-validation-3.6.2.tgz",
+      "integrity": "sha512-Y3EwblDz72KOcobb5t2zlhHSmrfE8EaHusPJ96Kx2JYtNXL2omqCoOb6FpaXWhES75wvjUpkFLYfiNqAqEov8g==",
      "dependencies": {
-        "@docusaurus/logger": "3.6.3",
-        "@docusaurus/utils": "3.6.3",
-        "@docusaurus/utils-common": "3.6.3",
+        "@docusaurus/logger": "3.6.2",
+        "@docusaurus/utils": "3.6.2",
+        "@docusaurus/utils-common": "3.6.2",
        "fs-extra": "^11.2.0",
        "joi": "^17.9.2",
        "js-yaml": "^4.1.0",
--- a/docs/package.json
+++ b/docs/package.json
@@ -15,10 +15,10 @@
    "typecheck": "tsc"
  },
  "dependencies": {
-    "@docusaurus/core": "^3.6.3",
-    "@docusaurus/plugin-content-pages": "^3.6.3",
-    "@docusaurus/preset-classic": "^3.6.3",
-    "@docusaurus/theme-mermaid": "^3.6.3",
+    "@docusaurus/core": "^3.6.2",
+    "@docusaurus/plugin-content-pages": "^3.6.2",
+    "@docusaurus/preset-classic": "^3.6.2",
+    "@docusaurus/theme-mermaid": "^3.6.2",
    "@mdx-js/react": "^3.1.0",
    "clsx": "^2.0.0",
    "prism-react-renderer": "^2.4.0",
@@ -29,7 +29,7 @@
  },
  "devDependencies": {
    "@docusaurus/module-type-aliases": "^3.5.1",
-    "@docusaurus/tsconfig": "^3.6.3",
+    "@docusaurus/tsconfig": "^3.6.2",
    "@docusaurus/types": "^3.5.1",
    "typescript": "~5.6.3"
  },
--- a/docs/sidebars.ts
+++ b/docs/sidebars.ts
@@ -44,17 +44,6 @@ const sidebars: SidebarsConfig = {
        },
      ],
    },
-    {
-      type: 'category',
-      label: 'Configuration Options',
-      items: [
-        {
-          type: 'doc',
-          label: 'Overview',
-          id: 'usage/configuration-options',
-        },
-      ],
-    },
    {
      type: 'category',
      label: 'Advanced Configuration',
--- a/docs/static/img/teaser.mp4
+++ b/docs/static/img/teaser.mp4
--- a/docs/yarn.lock
+++ b/docs/yarn.lock
@@ -1550,10 +1550,10 @@
    "@docsearch/css" "3.6.1"
    algoliasearch "^4.19.1"

-"@docusaurus/babel@3.6.3":
-  version "3.6.3"
-  resolved "https://registry.yarnpkg.com/@docusaurus/babel/-/babel-3.6.3.tgz#016714fe7a8807d0fc2f7180eace5e82bebbb8a6"
-  integrity sha512-7dW9Hat9EHYCVicFXYA4hjxBY38+hPuCURL8oRF9fySRm7vzNWuEOghA1TXcykuXZp0HLG2td4RhDxCvGG7tNw==
+"@docusaurus/babel@3.6.2":
+  version "3.6.2"
+  resolved "https://registry.yarnpkg.com/@docusaurus/babel/-/babel-3.6.2.tgz#c63dd2d9d7861189fe51950b3b6550477057bcee"
+  integrity sha512-v8N8TWGXDsb5sxQC3Rcqb1CZr0LlU1OgqqVBUchN6cpIUr7EJuVJs5eHcIu5Ag8mwO/hWN3f7FE9uaHTMapAbg==
  dependencies:
    "@babel/core" "^7.25.9"
    "@babel/generator" "^7.25.9"
@@ -1565,23 +1565,23 @@
    "@babel/runtime" "^7.25.9"
    "@babel/runtime-corejs3" "^7.25.9"
    "@babel/traverse" "^7.25.9"
-    "@docusaurus/logger" "3.6.3"
-    "@docusaurus/utils" "3.6.3"
+    "@docusaurus/logger" "3.6.2"
+    "@docusaurus/utils" "3.6.2"
    babel-plugin-dynamic-import-node "^2.3.3"
    fs-extra "^11.1.1"
    tslib "^2.6.0"

-"@docusaurus/bundler@3.6.3":
-  version "3.6.3"
-  resolved "https://registry.yarnpkg.com/@docusaurus/bundler/-/bundler-3.6.3.tgz#f09c2e29613f988b874a4be2247708e121b7fc5c"
-  integrity sha512-47JLuc8D4wA+6VOvmMd5fUC9rFppBQpQOnxDYiVXffm/DeV/wmm3sbpNd5Y+O+G2+nevLTRnvCm/qyancv0Y3A==
+"@docusaurus/bundler@3.6.2":
+  version "3.6.2"
+  resolved "https://registry.yarnpkg.com/@docusaurus/bundler/-/bundler-3.6.2.tgz#5bdd46862b40f1eea93f14714b858d07c2dd8c2f"
+  integrity sha512-YkEifEVs4lV931SrHBB4n6WqRowMw+aM/QPH3z8aU+5t1dWa+1p2OPqARS+tSbh3la9ns+L1zIfSbd8RHi2/PQ==
  dependencies:
    "@babel/core" "^7.25.9"
-    "@docusaurus/babel" "3.6.3"
-    "@docusaurus/cssnano-preset" "3.6.3"
-    "@docusaurus/logger" "3.6.3"
-    "@docusaurus/types" "3.6.3"
-    "@docusaurus/utils" "3.6.3"
+    "@docusaurus/babel" "3.6.2"
+    "@docusaurus/cssnano-preset" "3.6.2"
+    "@docusaurus/logger" "3.6.2"
+    "@docusaurus/types" "3.6.2"
+    "@docusaurus/utils" "3.6.2"
    babel-loader "^9.2.1"
    clean-css "^5.3.2"
    copy-webpack-plugin "^11.0.0"
@@ -1602,18 +1602,18 @@
    webpack "^5.95.0"
    webpackbar "^6.0.1"

-"@docusaurus/core@3.6.3", "@docusaurus/core@^3.6.3":
-  version "3.6.3"
-  resolved "https://registry.yarnpkg.com/@docusaurus/core/-/core-3.6.3.tgz#6bf968ee26a36d71387bab293f27ccffc0e428b6"
-  integrity sha512-xL7FRY9Jr5DWqB6pEnqgKqcMPJOX5V0pgWXi5lCiih11sUBmcFKM7c3+GyxcVeeWFxyYSDP3grLTWqJoP4P9Vw==
+"@docusaurus/core@3.6.2", "@docusaurus/core@^3.6.2":
+  version "3.6.2"
+  resolved "https://registry.yarnpkg.com/@docusaurus/core/-/core-3.6.2.tgz#78628790f255555bb4c81e5952d16ea1412c4548"
+  integrity sha512-irMts/mGLZv8dWcy0WUtbY/U6b5qIfHgQd1/kXMyAxUJo99fL0wFSqhMI+tcxjk0HYy427MXerLMqFJj+Arg1w==
  dependencies:
-    "@docusaurus/babel" "3.6.3"
-    "@docusaurus/bundler" "3.6.3"
-    "@docusaurus/logger" "3.6.3"
-    "@docusaurus/mdx-loader" "3.6.3"
-    "@docusaurus/utils" "3.6.3"
-    "@docusaurus/utils-common" "3.6.3"
-    "@docusaurus/utils-validation" "3.6.3"
+    "@docusaurus/babel" "3.6.2"
+    "@docusaurus/bundler" "3.6.2"
+    "@docusaurus/logger" "3.6.2"
+    "@docusaurus/mdx-loader" "3.6.2"
+    "@docusaurus/utils" "3.6.2"
+    "@docusaurus/utils-common" "3.6.2"
+    "@docusaurus/utils-validation" "3.6.2"
    boxen "^6.2.1"
    chalk "^4.1.2"
    chokidar "^3.5.3"
@@ -1651,32 +1651,32 @@
    webpack-dev-server "^4.15.2"
    webpack-merge "^6.0.1"

-"@docusaurus/cssnano-preset@3.6.3":
-  version "3.6.3"
-  resolved "https://registry.yarnpkg.com/@docusaurus/cssnano-preset/-/cssnano-preset-3.6.3.tgz#ea19b307183ec20dea4927efc4ddf249150b8c6a"
-  integrity sha512-qP7SXrwZ+23GFJdPN4aIHQrZW+oH/7tzwEuc/RNL0+BdZdmIjYQqUxdXsjE4lFxLNZjj0eUrSNYIS6xwfij+5Q==
+"@docusaurus/cssnano-preset@3.6.2":
+  version "3.6.2"
+  resolved "https://registry.yarnpkg.com/@docusaurus/cssnano-preset/-/cssnano-preset-3.6.2.tgz#007e7150b099ea2e9e874dd48a809614c628a335"
+  integrity sha512-mBkVa4QMHRwCFCVLYdBlOZuAT1iVVsS7GGSgliSVAeTOagP/AbtlBsCVrBs+keEuDuRF1w/6QEcqDoZe9fa5pw==
  dependencies:
    cssnano-preset-advanced "^6.1.2"
    postcss "^8.4.38"
    postcss-sort-media-queries "^5.2.0"
    tslib "^2.6.0"

-"@docusaurus/logger@3.6.3":
-  version "3.6.3"
-  resolved "https://registry.yarnpkg.com/@docusaurus/logger/-/logger-3.6.3.tgz#c6e514c9429487ef38be2f2129b2b842740d92fd"
-  integrity sha512-xSubJixcNyMV9wMV4q0s47CBz3Rlc5jbcCCuij8pfQP8qn/DIpt0ks8W6hQWzHAedg/J/EwxxUOUrnEoKzJo8g==
+"@docusaurus/logger@3.6.2":
+  version "3.6.2"
+  resolved "https://registry.yarnpkg.com/@docusaurus/logger/-/logger-3.6.2.tgz#4f73f82b33e1d432f3940fc208b3c0646ca5549c"
+  integrity sha512-1p4IQhhgLyIfsey4UAdAIW69aUE1Ei6O91Nsw30ryZeDWSG5dh4o3zaRGOLxfAX69Ac/yDm6YCwJOafUxL6Vxg==
  dependencies:
    chalk "^4.1.2"
    tslib "^2.6.0"

-"@docusaurus/mdx-loader@3.6.3":
-  version "3.6.3"
-  resolved "https://registry.yarnpkg.com/@docusaurus/mdx-loader/-/mdx-loader-3.6.3.tgz#127babc7cdb26d37c723bc3ae518bda17ce40160"
-  integrity sha512-3iJdiDz9540ppBseeI93tWTDtUGVkxzh59nMq4ignylxMuXBLK8dFqVeaEor23v1vx6TrGKZ2FuLaTB+U7C0QQ==
+"@docusaurus/mdx-loader@3.6.2":
+  version "3.6.2"
+  resolved "https://registry.yarnpkg.com/@docusaurus/mdx-loader/-/mdx-loader-3.6.2.tgz#d240974b0e754d5a5d8eb3f9d0a00a2055fabc68"
+  integrity sha512-7fbRmNgF3CR96Ja82Ya0/Cdu1OL9UJ/22llNMY8lr5gAbw718Y5ryXMVRIYn0JNLTiSxzgtvW4DIsUWEB8NMpw==
  dependencies:
-    "@docusaurus/logger" "3.6.3"
-    "@docusaurus/utils" "3.6.3"
-    "@docusaurus/utils-validation" "3.6.3"
+    "@docusaurus/logger" "3.6.2"
+    "@docusaurus/utils" "3.6.2"
+    "@docusaurus/utils-validation" "3.6.2"
    "@mdx-js/mdx" "^3.0.0"
    "@slorber/remark-comment" "^1.0.0"
    escape-html "^1.0.3"
@@ -1699,12 +1699,12 @@
    vfile "^6.0.1"
    webpack "^5.88.1"

-"@docusaurus/module-type-aliases@3.6.3", "@docusaurus/module-type-aliases@^3.5.1":
-  version "3.6.3"
-  resolved "https://registry.yarnpkg.com/@docusaurus/module-type-aliases/-/module-type-aliases-3.6.3.tgz#1f7030b1cf1f658cf664d41b6eadba93bbe51d87"
-  integrity sha512-MjaXX9PN/k5ugNvfRZdWyKWq4FsrhN4LEXaj0pEmMebJuBNlFeGyKQUa9DRhJHpadNaiMLrbo9m3U7Ig5YlsZg==
+"@docusaurus/module-type-aliases@3.6.2", "@docusaurus/module-type-aliases@^3.5.1":
+  version "3.6.2"
+  resolved "https://registry.yarnpkg.com/@docusaurus/module-type-aliases/-/module-type-aliases-3.6.2.tgz#7432618696668acc9a7cfb47de66c6987cd57680"
+  integrity sha512-NrJkL2rLTCjHtWOqUvWzwqvJrsKLj0gVJeV6q5yeKdKKgItietcTf2fTRkM9LHKSUN8CBDXxwHABeQvTahvmXQ==
  dependencies:
-    "@docusaurus/types" "3.6.3"
+    "@docusaurus/types" "3.6.2"
    "@types/history" "^4.7.11"
    "@types/react" "*"
    "@types/react-router-config" "*"
@@ -1712,19 +1712,19 @@
    react-helmet-async "*"
    react-loadable "npm:@docusaurus/react-loadable@6.0.0"

-"@docusaurus/plugin-content-blog@3.6.3":
-  version "3.6.3"
-  resolved "https://registry.yarnpkg.com/@docusaurus/plugin-content-blog/-/plugin-content-blog-3.6.3.tgz#d6a597e4bfdeb3f1f6ce06d2ac86207296988cc9"
-  integrity sha512-k0ogWwwJU3pFRFfvW1kRVHxzf2DutLGaaLjAnHVEU6ju+aRP0Z5ap/13DHyPOfHeE4WKpn/M0TqjdwZAcY3kAw==
+"@docusaurus/plugin-content-blog@3.6.2":
+  version "3.6.2"
+  resolved "https://registry.yarnpkg.com/@docusaurus/plugin-content-blog/-/plugin-content-blog-3.6.2.tgz#b197dd920e380bf1394995215ba7fee8019baa82"
+  integrity sha512-6bJxr6Or4NslEVH3BJuPH30kUWiqUjDRdGPhvxpHmt9W/RY2/6u72WICG3bW3dLFxJ/2uDLBU92lHnatpvo7Ew==
  dependencies:
-    "@docusaurus/core" "3.6.3"
-    "@docusaurus/logger" "3.6.3"
-    "@docusaurus/mdx-loader" "3.6.3"
-    "@docusaurus/theme-common" "3.6.3"
-    "@docusaurus/types" "3.6.3"
-    "@docusaurus/utils" "3.6.3"
-    "@docusaurus/utils-common" "3.6.3"
-    "@docusaurus/utils-validation" "3.6.3"
+    "@docusaurus/core" "3.6.2"
+    "@docusaurus/logger" "3.6.2"
+    "@docusaurus/mdx-loader" "3.6.2"
+    "@docusaurus/theme-common" "3.6.2"
+    "@docusaurus/types" "3.6.2"
+    "@docusaurus/utils" "3.6.2"
+    "@docusaurus/utils-common" "3.6.2"
+    "@docusaurus/utils-validation" "3.6.2"
    cheerio "1.0.0-rc.12"
    feed "^4.2.2"
    fs-extra "^11.1.1"
@@ -1736,20 +1736,20 @@
    utility-types "^3.10.0"
    webpack "^5.88.1"

-"@docusaurus/plugin-content-docs@3.6.3":
-  version "3.6.3"
-  resolved "https://registry.yarnpkg.com/@docusaurus/plugin-content-docs/-/plugin-content-docs-3.6.3.tgz#aae044d2af6996d1a6de8d815aca8a83b485e0a5"
-  integrity sha512-r2wS8y/fsaDcxkm20W5bbYJFPzdWdEaTWVYjNxlHlcmX086eqQR1Fomlg9BHTJ0dLXPzAlbC8EN4XqMr3QzNCQ==
+"@docusaurus/plugin-content-docs@3.6.2":
+  version "3.6.2"
+  resolved "https://registry.yarnpkg.com/@docusaurus/plugin-content-docs/-/plugin-content-docs-3.6.2.tgz#3a8b4b162a2688e5855c04ed6c4ec0b6951619a0"
+  integrity sha512-e6WW1g10RIXXLN/rrtqTi/FyJ1Hj3X9Mmgz4V11/0pDCxIGGI8m4ocbAglUlLtgvbLD5viNLefl/NwbOW3JXiQ==
  dependencies:
-    "@docusaurus/core" "3.6.3"
-    "@docusaurus/logger" "3.6.3"
-    "@docusaurus/mdx-loader" "3.6.3"
-    "@docusaurus/module-type-aliases" "3.6.3"
-    "@docusaurus/theme-common" "3.6.3"
-    "@docusaurus/types" "3.6.3"
-    "@docusaurus/utils" "3.6.3"
-    "@docusaurus/utils-common" "3.6.3"
-    "@docusaurus/utils-validation" "3.6.3"
+    "@docusaurus/core" "3.6.2"
+    "@docusaurus/logger" "3.6.2"
+    "@docusaurus/mdx-loader" "3.6.2"
+    "@docusaurus/module-type-aliases" "3.6.2"
+    "@docusaurus/theme-common" "3.6.2"
+    "@docusaurus/types" "3.6.2"
+    "@docusaurus/utils" "3.6.2"
+    "@docusaurus/utils-common" "3.6.2"
+    "@docusaurus/utils-validation" "3.6.2"
    "@types/react-router-config" "^5.0.7"
    combine-promises "^1.1.0"
    fs-extra "^11.1.1"
@@ -1759,115 +1759,115 @@
    utility-types "^3.10.0"
    webpack "^5.88.1"

-"@docusaurus/plugin-content-pages@3.6.3", "@docusaurus/plugin-content-pages@^3.6.3":
-  version "3.6.3"
-  resolved "https://registry.yarnpkg.com/@docusaurus/plugin-content-pages/-/plugin-content-pages-3.6.3.tgz#0a5a43d1677ee519f63a54634653c54ddf41f475"
-  integrity sha512-eHrmTgjgLZsuqfsYr5X2xEwyIcck0wseSofWrjTwT9FLOWp+KDmMAuVK+wRo7sFImWXZk3oV/xX/g9aZrhD7OA==
+"@docusaurus/plugin-content-pages@3.6.2", "@docusaurus/plugin-content-pages@^3.6.2":
+  version "3.6.2"
+  resolved "https://registry.yarnpkg.com/@docusaurus/plugin-content-pages/-/plugin-content-pages-3.6.2.tgz#49b1a033d41841f7a8bcbbe67511609b402cc80f"
+  integrity sha512-fo4NyGkw10lYHyHaTxE6TZLYnxNtCfRHeZkNK1N9pBYqe7TT2dBUNAEeVW2U3ed9m6YuB7JKSQsa++GGmcP+6g==
  dependencies:
-    "@docusaurus/core" "3.6.3"
-    "@docusaurus/mdx-loader" "3.6.3"
-    "@docusaurus/types" "3.6.3"
-    "@docusaurus/utils" "3.6.3"
-    "@docusaurus/utils-validation" "3.6.3"
+    "@docusaurus/core" "3.6.2"
+    "@docusaurus/mdx-loader" "3.6.2"
+    "@docusaurus/types" "3.6.2"
+    "@docusaurus/utils" "3.6.2"
+    "@docusaurus/utils-validation" "3.6.2"
    fs-extra "^11.1.1"
    tslib "^2.6.0"
    webpack "^5.88.1"

-"@docusaurus/plugin-debug@3.6.3":
-  version "3.6.3"
-  resolved "https://registry.yarnpkg.com/@docusaurus/plugin-debug/-/plugin-debug-3.6.3.tgz#4e62ddfbae4d597b073f8e3c632cc12d012339e3"
-  integrity sha512-zB9GXfIZNPRfzKnNjU6xGVrqn9bPXuGhpjgsuc/YtcTDjnjhasg38NdYd5LEqXex5G/zIorQgWB3n6x/Ut62vQ==
+"@docusaurus/plugin-debug@3.6.2":
+  version "3.6.2"
+  resolved "https://registry.yarnpkg.com/@docusaurus/plugin-debug/-/plugin-debug-3.6.2.tgz#6983f64954fe69a51b65b2d9431bdf0b5ccf1884"
+  integrity sha512-T/eS3VvHElpeV5S8uwp7Si4ujEynmgFtJLvA2CSa5pzQuOF1EEghF9nekAIj0cWtDHsqNUDZNr8hK1brivFXSg==
  dependencies:
-    "@docusaurus/core" "3.6.3"
-    "@docusaurus/types" "3.6.3"
-    "@docusaurus/utils" "3.6.3"
+    "@docusaurus/core" "3.6.2"
+    "@docusaurus/types" "3.6.2"
+    "@docusaurus/utils" "3.6.2"
    fs-extra "^11.1.1"
    react-json-view-lite "^1.2.0"
    tslib "^2.6.0"

-"@docusaurus/plugin-google-analytics@3.6.3":
-  version "3.6.3"
-  resolved "https://registry.yarnpkg.com/@docusaurus/plugin-google-analytics/-/plugin-google-analytics-3.6.3.tgz#63648d469b1e3c50fad8878e7a7db9856e503d5f"
-  integrity sha512-rCDNy1QW8Dag7nZq67pcum0bpFLrwvxJhYuVprhFh8BMBDxV0bY+bAkGHbSf68P3Bk9C3hNOAXX1srGLIDvcTA==
+"@docusaurus/plugin-google-analytics@3.6.2":
+  version "3.6.2"
+  resolved "https://registry.yarnpkg.com/@docusaurus/plugin-google-analytics/-/plugin-google-analytics-3.6.2.tgz#4266b4b2273998e87fa733d932d5b464c2a10b21"
+  integrity sha512-B7ihrr3wz8e4XqW+dIAtq844u3Z83u5CeiL1xrCqzFH+vDCjUZHTamS3zKXNcgi6YVVe6hUQXPG15ltaqQaVPQ==
  dependencies:
-    "@docusaurus/core" "3.6.3"
-    "@docusaurus/types" "3.6.3"
-    "@docusaurus/utils-validation" "3.6.3"
+    "@docusaurus/core" "3.6.2"
+    "@docusaurus/types" "3.6.2"
+    "@docusaurus/utils-validation" "3.6.2"
    tslib "^2.6.0"

-"@docusaurus/plugin-google-gtag@3.6.3":
-  version "3.6.3"
-  resolved "https://registry.yarnpkg.com/@docusaurus/plugin-google-gtag/-/plugin-google-gtag-3.6.3.tgz#8a1388b4123904be17e661ea7aa71d798d0c046e"
-  integrity sha512-+OyDvhM6rqVkQOmLVkQWVJAizEEfkPzVWtIHXlWPOCFGK9X4/AWeBSrU0WG4iMg9Z4zD4YDRrU+lvI4s6DSC+w==
+"@docusaurus/plugin-google-gtag@3.6.2":
+  version "3.6.2"
+  resolved "https://registry.yarnpkg.com/@docusaurus/plugin-google-gtag/-/plugin-google-gtag-3.6.2.tgz#23f70f7a05e61cfb9d9d7ee18dbff3ef2b129f2c"
+  integrity sha512-V8ijI6qddAAkJ0vd8sjZ7S/apRTLJn9dAwvj/rSMd93witGdKINemL+9TyfLkhcXKTxyqRT8zKdu8ewjPXqKHg==
  dependencies:
-    "@docusaurus/core" "3.6.3"
-    "@docusaurus/types" "3.6.3"
-    "@docusaurus/utils-validation" "3.6.3"
+    "@docusaurus/core" "3.6.2"
+    "@docusaurus/types" "3.6.2"
+    "@docusaurus/utils-validation" "3.6.2"
    "@types/gtag.js" "^0.0.12"
    tslib "^2.6.0"

-"@docusaurus/plugin-google-tag-manager@3.6.3":
-  version "3.6.3"
-  resolved "https://registry.yarnpkg.com/@docusaurus/plugin-google-tag-manager/-/plugin-google-tag-manager-3.6.3.tgz#38cbe416803f29782807cebf3ebf240cb47c3c74"
-  integrity sha512-1M6UPB13gWUtN2UHX083/beTn85PlRI9ABItTl/JL1FJ5dJTWWFXXsHf9WW/6hrVwthwTeV/AGbGKvLKV+IlCA==
+"@docusaurus/plugin-google-tag-manager@3.6.2":
+  version "3.6.2"
+  resolved "https://registry.yarnpkg.com/@docusaurus/plugin-google-tag-manager/-/plugin-google-tag-manager-3.6.2.tgz#36ab95fcd5c1bf96fd18c0cf9b208bb428b81242"
+  integrity sha512-fnWQ5FdN9f8c8VTgjaQ98208Y+d/JjHhD506rWIIL9rt1cJOf29XElxvOeKpMJadfkgY5KLZSAiHkGt+4qgN4g==
  dependencies:
-    "@docusaurus/core" "3.6.3"
-    "@docusaurus/types" "3.6.3"
-    "@docusaurus/utils-validation" "3.6.3"
+    "@docusaurus/core" "3.6.2"
+    "@docusaurus/types" "3.6.2"
+    "@docusaurus/utils-validation" "3.6.2"
    tslib "^2.6.0"

-"@docusaurus/plugin-sitemap@3.6.3":
-  version "3.6.3"
-  resolved "https://registry.yarnpkg.com/@docusaurus/plugin-sitemap/-/plugin-sitemap-3.6.3.tgz#0458e6f7476ab6fd1466e01b153a3211d3223c53"
-  integrity sha512-94qOO4M9Fwv9KfVQJsgbe91k+fPJ4byf1L3Ez8TUa6TAFPo/BrLwQ80zclHkENlL1824TuxkcMKv33u6eydQCg==
+"@docusaurus/plugin-sitemap@3.6.2":
+  version "3.6.2"
+  resolved "https://registry.yarnpkg.com/@docusaurus/plugin-sitemap/-/plugin-sitemap-3.6.2.tgz#c8ff7cf82bd5d943a13bb8d0ae690080a025029e"
+  integrity sha512-qcAQAP1Ot0dZpeRoJ0L/Zck5FVDkll2IleVZQLzxeRVDZIw1P9/TK7/Aw1w2pmH7dmw/Cwk/cLSVRvLAmp9k7A==
  dependencies:
-    "@docusaurus/core" "3.6.3"
-    "@docusaurus/logger" "3.6.3"
-    "@docusaurus/types" "3.6.3"
-    "@docusaurus/utils" "3.6.3"
-    "@docusaurus/utils-common" "3.6.3"
-    "@docusaurus/utils-validation" "3.6.3"
+    "@docusaurus/core" "3.6.2"
+    "@docusaurus/logger" "3.6.2"
+    "@docusaurus/types" "3.6.2"
+    "@docusaurus/utils" "3.6.2"
+    "@docusaurus/utils-common" "3.6.2"
+    "@docusaurus/utils-validation" "3.6.2"
    fs-extra "^11.1.1"
    sitemap "^7.1.1"
    tslib "^2.6.0"

-"@docusaurus/preset-classic@^3.6.3":
-  version "3.6.3"
-  resolved "https://registry.yarnpkg.com/@docusaurus/preset-classic/-/preset-classic-3.6.3.tgz#072298b5b6d0de7d0346b1e9b550a30ef2add56d"
-  integrity sha512-VHSYWROT3flvNNI1SrnMOtW1EsjeHNK9dhU6s9eY5hryZe79lUqnZJyze/ymDe2LXAqzyj6y5oYvyBoZZk6ErA==
+"@docusaurus/preset-classic@^3.6.2":
+  version "3.6.2"
+  resolved "https://registry.yarnpkg.com/@docusaurus/preset-classic/-/preset-classic-3.6.2.tgz#5ec801fa317123ba8458af3105eca8eac78a49bc"
+  integrity sha512-r2n5eHdhiNSrJGsrrYcw+WsyStmXxe0ZG3RdA9LVyK5+jBHM8blrUWJEDugnzCNbyhUzhdtcmgCC9fhdAvKuQw==
  dependencies:
-    "@docusaurus/core" "3.6.3"
-    "@docusaurus/plugin-content-blog" "3.6.3"
-    "@docusaurus/plugin-content-docs" "3.6.3"
-    "@docusaurus/plugin-content-pages" "3.6.3"
-    "@docusaurus/plugin-debug" "3.6.3"
-    "@docusaurus/plugin-google-analytics" "3.6.3"
-    "@docusaurus/plugin-google-gtag" "3.6.3"
-    "@docusaurus/plugin-google-tag-manager" "3.6.3"
-    "@docusaurus/plugin-sitemap" "3.6.3"
-    "@docusaurus/theme-classic" "3.6.3"
-    "@docusaurus/theme-common" "3.6.3"
-    "@docusaurus/theme-search-algolia" "3.6.3"
-    "@docusaurus/types" "3.6.3"
+    "@docusaurus/core" "3.6.2"
+    "@docusaurus/plugin-content-blog" "3.6.2"
+    "@docusaurus/plugin-content-docs" "3.6.2"
+    "@docusaurus/plugin-content-pages" "3.6.2"
+    "@docusaurus/plugin-debug" "3.6.2"
+    "@docusaurus/plugin-google-analytics" "3.6.2"
+    "@docusaurus/plugin-google-gtag" "3.6.2"
+    "@docusaurus/plugin-google-tag-manager" "3.6.2"
+    "@docusaurus/plugin-sitemap" "3.6.2"
+    "@docusaurus/theme-classic" "3.6.2"
+    "@docusaurus/theme-common" "3.6.2"
+    "@docusaurus/theme-search-algolia" "3.6.2"
+    "@docusaurus/types" "3.6.2"

-"@docusaurus/theme-classic@3.6.3":
-  version "3.6.3"
-  resolved "https://registry.yarnpkg.com/@docusaurus/theme-classic/-/theme-classic-3.6.3.tgz#00599a9de5fd5c122fd1b8c59d3b755878f2a72c"
-  integrity sha512-1RRLK1tSArI2c00qugWYO3jRocjOZwGF1mBzPPylDVRwWCS/rnWWR91ChdbbaxIupRJ+hX8ZBYrwr5bbU0oztQ==
+"@docusaurus/theme-classic@3.6.2":
+  version "3.6.2"
+  resolved "https://registry.yarnpkg.com/@docusaurus/theme-classic/-/theme-classic-3.6.2.tgz#4c2770d3609176dd2462dfb0cb4d0b3d3010404b"
+  integrity sha512-bCdOPqPNezhLx+hgNVO2Cf+8/1AHa9uHDOqTx/CKAx2I0J/jV9G+6JiMtpSRKGNfBoLT1O+56/7+WtkOf54xTw==
  dependencies:
-    "@docusaurus/core" "3.6.3"
-    "@docusaurus/logger" "3.6.3"
-    "@docusaurus/mdx-loader" "3.6.3"
-    "@docusaurus/module-type-aliases" "3.6.3"
-    "@docusaurus/plugin-content-blog" "3.6.3"
-    "@docusaurus/plugin-content-docs" "3.6.3"
-    "@docusaurus/plugin-content-pages" "3.6.3"
-    "@docusaurus/theme-common" "3.6.3"
-    "@docusaurus/theme-translations" "3.6.3"
-    "@docusaurus/types" "3.6.3"
-    "@docusaurus/utils" "3.6.3"
-    "@docusaurus/utils-common" "3.6.3"
-    "@docusaurus/utils-validation" "3.6.3"
+    "@docusaurus/core" "3.6.2"
+    "@docusaurus/logger" "3.6.2"
+    "@docusaurus/mdx-loader" "3.6.2"
+    "@docusaurus/module-type-aliases" "3.6.2"
+    "@docusaurus/plugin-content-blog" "3.6.2"
+    "@docusaurus/plugin-content-docs" "3.6.2"
+    "@docusaurus/plugin-content-pages" "3.6.2"
+    "@docusaurus/theme-common" "3.6.2"
+    "@docusaurus/theme-translations" "3.6.2"
+    "@docusaurus/types" "3.6.2"
+    "@docusaurus/utils" "3.6.2"
+    "@docusaurus/utils-common" "3.6.2"
+    "@docusaurus/utils-validation" "3.6.2"
    "@mdx-js/react" "^3.0.0"
    clsx "^2.0.0"
    copy-text-to-clipboard "^3.2.0"
@@ -1882,15 +1882,15 @@
    tslib "^2.6.0"
    utility-types "^3.10.0"

-"@docusaurus/theme-common@3.6.3":
-  version "3.6.3"
-  resolved "https://registry.yarnpkg.com/@docusaurus/theme-common/-/theme-common-3.6.3.tgz#a8a6ebd2b0fd7a5cca4d0c6a2f9ccff905fa7438"
-  integrity sha512-b8ZkhczXHDxWWyvz+YJy4t/PlPbEogTTbgnHoflYnH7rmRtyoodTsu8WVM12la5LmlMJBclBXFl29OH8kPE7gg==
+"@docusaurus/theme-common@3.6.2":
+  version "3.6.2"
+  resolved "https://registry.yarnpkg.com/@docusaurus/theme-common/-/theme-common-3.6.2.tgz#a520d9053b6ea0fa913d42898d35f73ed5ca3b9b"
+  integrity sha512-lfgsL064KEHpCkgGUc0OYoUPCpYfzggp6Hof8sz59UuKiLvb/Z7raewE9/NfocrJ2HZI17rLgMX3SQlRDh/5gg==
  dependencies:
-    "@docusaurus/mdx-loader" "3.6.3"
-    "@docusaurus/module-type-aliases" "3.6.3"
-    "@docusaurus/utils" "3.6.3"
-    "@docusaurus/utils-common" "3.6.3"
+    "@docusaurus/mdx-loader" "3.6.2"
+    "@docusaurus/module-type-aliases" "3.6.2"
+    "@docusaurus/utils" "3.6.2"
+    "@docusaurus/utils-common" "3.6.2"
    "@types/history" "^4.7.11"
    "@types/react" "*"
    "@types/react-router-config" "*"
@@ -1900,32 +1900,32 @@
    tslib "^2.6.0"
    utility-types "^3.10.0"

-"@docusaurus/theme-mermaid@^3.6.3":
-  version "3.6.3"
-  resolved "https://registry.yarnpkg.com/@docusaurus/theme-mermaid/-/theme-mermaid-3.6.3.tgz#4bb0b4940b39cc855626a23190aa45928aa168b4"
-  integrity sha512-kIqpjNCP/9R2GGf8UmiDxD3CkOAEJuJIEFlaKMgQtjVxa/vH+9PLI1+DFbArGoG4+0ENTYUq8phHPW7SeL36uQ==
+"@docusaurus/theme-mermaid@^3.6.2":
+  version "3.6.2"
+  resolved "https://registry.yarnpkg.com/@docusaurus/theme-mermaid/-/theme-mermaid-3.6.2.tgz#f7ad64ae5b9510ddbf77bc4264ada059c8e815b1"
+  integrity sha512-Ui+rBtqMPKj3RCOxNlY04i1tEjNg+fZg4URTvkHmYR07hcKaJw+vkw+wlaYjd0HFZk+3Er9vUAcwsCWuea4cVQ==
  dependencies:
-    "@docusaurus/core" "3.6.3"
-    "@docusaurus/module-type-aliases" "3.6.3"
-    "@docusaurus/theme-common" "3.6.3"
-    "@docusaurus/types" "3.6.3"
-    "@docusaurus/utils-validation" "3.6.3"
+    "@docusaurus/core" "3.6.2"
+    "@docusaurus/module-type-aliases" "3.6.2"
+    "@docusaurus/theme-common" "3.6.2"
+    "@docusaurus/types" "3.6.2"
+    "@docusaurus/utils-validation" "3.6.2"
    mermaid ">=10.4"
    tslib "^2.6.0"

-"@docusaurus/theme-search-algolia@3.6.3":
-  version "3.6.3"
-  resolved "https://registry.yarnpkg.com/@docusaurus/theme-search-algolia/-/theme-search-algolia-3.6.3.tgz#1a3331a489f392f5b032c4efc5f431e57eddf7ce"
-  integrity sha512-rt+MGCCpYgPyWCGXtbxlwFbTSobu15jWBTPI2LHsHNa5B0zSmOISX6FWYAPt5X1rNDOqMGM0FATnh7TBHRohVA==
+"@docusaurus/theme-search-algolia@3.6.2":
+  version "3.6.2"
+  resolved "https://registry.yarnpkg.com/@docusaurus/theme-search-algolia/-/theme-search-algolia-3.6.2.tgz#b03b7d35a385004d089d000be764abdfb3fa5721"
+  integrity sha512-SFLS+Rq8Cg2yepnHucA9sRpIR97yHvZWlCgMzBLunV3KHbB6hD2h5HPhFV39wYHYCjJUAOH1lX9poJ1qKYuSvg==
  dependencies:
    "@docsearch/react" "^3.5.2"
-    "@docusaurus/core" "3.6.3"
-    "@docusaurus/logger" "3.6.3"
-    "@docusaurus/plugin-content-docs" "3.6.3"
-    "@docusaurus/theme-common" "3.6.3"
-    "@docusaurus/theme-translations" "3.6.3"
-    "@docusaurus/utils" "3.6.3"
-    "@docusaurus/utils-validation" "3.6.3"
+    "@docusaurus/core" "3.6.2"
+    "@docusaurus/logger" "3.6.2"
+    "@docusaurus/plugin-content-docs" "3.6.2"
+    "@docusaurus/theme-common" "3.6.2"
+    "@docusaurus/theme-translations" "3.6.2"
+    "@docusaurus/utils" "3.6.2"
+    "@docusaurus/utils-validation" "3.6.2"
    algoliasearch "^4.18.0"
    algoliasearch-helper "^3.13.3"
    clsx "^2.0.0"
@@ -1935,23 +1935,23 @@
    tslib "^2.6.0"
    utility-types "^3.10.0"

-"@docusaurus/theme-translations@3.6.3":
-  version "3.6.3"
-  resolved "https://registry.yarnpkg.com/@docusaurus/theme-translations/-/theme-translations-3.6.3.tgz#6e473835ea016ce4acd7d2997f411811db8c4f6b"
-  integrity sha512-Gb0regclToVlngSIIwUCtBMQBq48qVUaN1XQNKW4XwlsgUyk0vP01LULdqbem7czSwIeBAFXFoORJ0RPX7ht/w==
+"@docusaurus/theme-translations@3.6.2":
+  version "3.6.2"
+  resolved "https://registry.yarnpkg.com/@docusaurus/theme-translations/-/theme-translations-3.6.2.tgz#ff6d2588aa9bf9fb1e07465def067529d5668665"
+  integrity sha512-LIWrYoDUsOTKmb0c7IQzawiPUTAaczBs5IOx6srxOWoTHVUMLzJCkl5Y6whfuRrnul8G05qv2vk238bN5Ko62g==
  dependencies:
    fs-extra "^11.1.1"
    tslib "^2.6.0"

-"@docusaurus/tsconfig@^3.6.3":
-  version "3.6.3"
-  resolved "https://registry.yarnpkg.com/@docusaurus/tsconfig/-/tsconfig-3.6.3.tgz#8af20c45f0a67e193debedcb341c0a1e78b1dd63"
-  integrity sha512-1pT/rTrRpMV15E4tJH95W5PrjboMn5JkKF+Ys8cTjMegetiXjs0gPFOSDA5hdTlberKQLDO50xPjMJHondLuzA==
+"@docusaurus/tsconfig@^3.6.2":
+  version "3.6.2"
+  resolved "https://registry.yarnpkg.com/@docusaurus/tsconfig/-/tsconfig-3.6.2.tgz#a6d3fec19ae45a67da678b54c019b0e3839b6711"
+  integrity sha512-TWLkyYHBYhIJNcXCEc3D1M9R8UFV4IZ82rGef5U9mE1ZrcgDUlZxYaYdoSuHrPrzPRIl3orjmpscO2FAk2gdZw==

-"@docusaurus/types@3.6.3", "@docusaurus/types@^3.5.1":
-  version "3.6.3"
-  resolved "https://registry.yarnpkg.com/@docusaurus/types/-/types-3.6.3.tgz#e87592e31616da1b8dc473e4c8205c61885a1518"
-  integrity sha512-xD9oTGDrouWzefkhe9ogB2fDV96/82cRpNGx2HIvI5L87JHNhQVIWimQ/3JIiiX/TEd5S9s+VO6FFguwKNRVow==
+"@docusaurus/types@3.6.2", "@docusaurus/types@^3.5.1":
+  version "3.6.2"
+  resolved "https://registry.yarnpkg.com/@docusaurus/types/-/types-3.6.2.tgz#bd69c4c99b535b67f01276dc186622e0b1fc1305"
+  integrity sha512-117Wsk6xXrWEAsCYCXS3TGJv5tkdIZDcd7T/V0UJvKYmY0gyVPPcEQChy8yTdjbIkbB2q4fa7Jpox72Qv86mqQ==
  dependencies:
    "@mdx-js/mdx" "^3.0.0"
    "@types/history" "^4.7.11"
@@ -1963,36 +1963,36 @@
    webpack "^5.95.0"
    webpack-merge "^5.9.0"

-"@docusaurus/utils-common@3.6.3":
-  version "3.6.3"
-  resolved "https://registry.yarnpkg.com/@docusaurus/utils-common/-/utils-common-3.6.3.tgz#57f840bd6f0928cf10060198cb421f1b9212c8f5"
-  integrity sha512-v4nKDaANLgT3pMBewHYEMAl/ufY0LkXao1QkFWzI5huWFOmNQ2UFzv2BiKeHX5Ownis0/w6cAyoxPhVdDonlSQ==
+"@docusaurus/utils-common@3.6.2":
+  version "3.6.2"
+  resolved "https://registry.yarnpkg.com/@docusaurus/utils-common/-/utils-common-3.6.2.tgz#3367572d72090b7f17e721af7f020f8e39931662"
+  integrity sha512-dr5wK+OyU2QAWxG7S5siD2bPgS7+ZeqWHfgLNHZ5yalaZf8TbeNNLqydfngukAY56BGZN0NbMkX6jGIr7ZF0sA==
  dependencies:
-    "@docusaurus/types" "3.6.3"
+    "@docusaurus/types" "3.6.2"
    tslib "^2.6.0"

-"@docusaurus/utils-validation@3.6.3":
-  version "3.6.3"
-  resolved "https://registry.yarnpkg.com/@docusaurus/utils-validation/-/utils-validation-3.6.3.tgz#3eca7125235eb90983ff660b97a71f331e331f57"
-  integrity sha512-bhEGGiN5BE38h21vjqD70Gxg++j+PfYVddDUE5UFvLDup68QOcpD33CLr+2knPorlxRbEaNfz6HQDUMQ3HuqKw==
+"@docusaurus/utils-validation@3.6.2":
+  version "3.6.2"
+  resolved "https://registry.yarnpkg.com/@docusaurus/utils-validation/-/utils-validation-3.6.2.tgz#62b97a0d72694c85fa63928c494dd238a84c991f"
+  integrity sha512-Y3EwblDz72KOcobb5t2zlhHSmrfE8EaHusPJ96Kx2JYtNXL2omqCoOb6FpaXWhES75wvjUpkFLYfiNqAqEov8g==
  dependencies:
-    "@docusaurus/logger" "3.6.3"
-    "@docusaurus/utils" "3.6.3"
-    "@docusaurus/utils-common" "3.6.3"
+    "@docusaurus/logger" "3.6.2"
+    "@docusaurus/utils" "3.6.2"
+    "@docusaurus/utils-common" "3.6.2"
    fs-extra "^11.2.0"
    joi "^17.9.2"
    js-yaml "^4.1.0"
    lodash "^4.17.21"
    tslib "^2.6.0"

-"@docusaurus/utils@3.6.3":
-  version "3.6.3"
-  resolved "https://registry.yarnpkg.com/@docusaurus/utils/-/utils-3.6.3.tgz#8dcb1969e4011a84dfb0a031da806dadddebf0ea"
-  integrity sha512-0R/FR3bKVl4yl8QwbL4TYFfR+OXBRpVUaTJdENapBGR3YMwfM6/JnhGilWQO8AOwPJGtGoDK7ib8+8UF9f3OZQ==
+"@docusaurus/utils@3.6.2":
+  version "3.6.2"
+  resolved "https://registry.yarnpkg.com/@docusaurus/utils/-/utils-3.6.2.tgz#727299c2051eee04c1b431bc6ccd55fd4e5a0d52"
+  integrity sha512-oxnpUcFZGE3uPCDoXr8GJriB3VWM9sFjPedFidX3Fsz87l1NZNc1wtbKPfQ7GYFDMYo2IGlAv5+47Me9RkM6lg==
  dependencies:
-    "@docusaurus/logger" "3.6.3"
-    "@docusaurus/types" "3.6.3"
-    "@docusaurus/utils-common" "3.6.3"
+    "@docusaurus/logger" "3.6.2"
+    "@docusaurus/types" "3.6.2"
+    "@docusaurus/utils-common" "3.6.2"
    "@svgr/webpack" "^8.1.0"
    escape-string-regexp "^4.0.0"
    file-loader "^6.2.0"
--- a/evaluation/benchmarks/EDA/README.md
+++ b/evaluation/benchmarks/EDA/README.md
@@ -12,7 +12,7 @@ Please follow instruction [here](../README.md#setup) to setup your local develop

 ```bash
 export OPENAI_API_KEY="sk-XXX"; # This is required for evaluation (to simulate another party of conversation)
-./evaluation/benchmarks/EDA/scripts/run_infer.sh [model_config] [git-version] [agent] [dataset] [eval_limit]
+./evaluation/EDA/scripts/run_infer.sh [model_config] [git-version] [agent] [dataset] [eval_limit]
 ```

 where `model_config` is mandatory, while `git-version`, `agent`, `dataset` and `eval_limit` are optional.
@@ -33,7 +33,7 @@ to `CodeActAgent`.
 For example,

 ```bash
-./evaluation/benchmarks/EDA/scripts/run_infer.sh eval_gpt4o_2024_05_13 0.6.2 CodeActAgent things
+./evaluation/EDA/scripts/run_infer.sh eval_gpt4o_2024_05_13 0.6.2 CodeActAgent things
 ```

 ## Reference
--- a/evaluation/benchmarks/EDA/game.py
+++ b/evaluation/benchmarks/EDA/game.py
--- a/evaluation/benchmarks/EDA/run_infer.py
+++ b/evaluation/benchmarks/EDA/run_infer.py
@@ -4,7 +4,7 @@ import os
 import pandas as pd
 from datasets import load_dataset

-from evaluation.benchmarks.EDA.game import Q20Game, Q20GameCelebrity
+from evaluation.EDA.game import Q20Game, Q20GameCelebrity
 from evaluation.utils.shared import (
    EvalMetadata,
    EvalOutput,
--- a/evaluation/benchmarks/EDA/scripts/run_infer.sh
+++ b/evaluation/benchmarks/EDA/scripts/run_infer.sh
@@ -43,7 +43,7 @@ echo "AGENT_VERSION: $AGENT_VERSION"
 echo "MODEL_CONFIG: $MODEL_CONFIG"
 echo "DATASET: $DATASET"

-COMMAND="poetry run python evaluation/benchmarks/EDA/run_infer.py \
+COMMAND="poetry run python evaluation/EDA/run_infer.py \
  --agent-cls $AGENT \
  --llm-config $MODEL_CONFIG \
  --dataset $DATASET \
--- a/evaluation/README.md
+++ b/evaluation/README.md
@@ -6,9 +6,9 @@ This folder contains code and resources to run experiments and evaluations.

 ### Setup

-Before starting evaluation, follow the instructions [here](https://github.com/All-Hands-AI/OpenHands/blob/main/Development.md) to setup your local development environment and LLM.
+Before starting evaluation, follow the instructions here [here](https://github.com/All-Hands-AI/OpenHands/blob/main/Development.md) to setup your local development environment and LLM.

-Once you are done with setup, you can follow the benchmark-specific instructions in each subdirectory of the [evaluation directory](#supported-benchmarks).
+Once you are done with setup, you can follow the benchmark-specific instructions in each subdirectory of the evaluation directory.
 Generally these will involve running `run_infer.py` to perform inference with the agents.

 ### Implementing and Evaluating an Agent
@@ -42,36 +42,32 @@ temperature = 0.0

 ## Supported Benchmarks

-The OpenHands evaluation harness supports a wide variety of benchmarks across [software engineering](#software-engineering), [web browsing](#web-browsing), and [miscellaneous assistance](#misc-assistance) tasks.
+The OpenHands evaluation harness supports a wide variety of benchmarks across software engineering, web browsing, and miscellaneous assistance tasks.

 ### Software Engineering

- SWE-Bench: [`evaluation/benchmarks/swe_bench`](./benchmarks/swe_bench)
- HumanEvalFix: [`evaluation/benchmarks/humanevalfix`](./benchmarks/humanevalfix)
- BIRD: [`evaluation/benchmarks/bird`](./benchmarks/bird)
- BioCoder: [`evaluation/benchmarks/ml_bench`](./benchmarks/ml_bench)
- ML-Bench: [`evaluation/benchmarks/ml_bench`](./benchmarks/ml_bench)
- APIBench: [`evaluation/benchmarks/gorilla`](./benchmarks/gorilla/)
- ToolQA: [`evaluation/benchmarks/toolqa`](./benchmarks/toolqa/)
- AiderBench: [`evaluation/benchmarks/aider_bench`](./benchmarks/aider_bench/)
- Commit0: [`evaluation/benchmarks/commit0_bench`](./benchmarks/commit0_bench/)
- DiscoveryBench: [`evaluation/benchmarks/discoverybench`](./benchmarks/discoverybench/)
+- SWE-Bench: [`evaluation/swe_bench`](./swe_bench)
+- HumanEvalFix: [`evaluation/humanevalfix`](./humanevalfix)
+- BIRD: [`evaluation/bird`](./bird)
+- BioCoder: [`evaluation/ml_bench`](./ml_bench)
+- ML-Bench: [`evaluation/ml_bench`](./ml_bench)
+- APIBench: [`evaluation/gorilla`](./gorilla/)
+- ToolQA: [`evaluation/toolqa`](./toolqa/)
+- AiderBench: [`evaluation/aider_bench`](./aider_bench/)

 ### Web Browsing

- WebArena: [`evaluation/benchmarks/webarena`](./benchmarks/webarena/)
- MiniWob++: [`evaluation/benchmarks/miniwob`](./benchmarks/miniwob/)
- Browsing Delegation: [`evaluation/benchmarks/browsing_delegation`](./benchmarks/browsing_delegation/)
+- WebArena: [`evaluation/webarena`](./webarena/)
+- MiniWob++: [`evaluation/miniwob`](./miniwob/)

 ### Misc. Assistance

- GAIA: [`evaluation/benchmarks/gaia`](./benchmarks/gaia)
- GPQA: [`evaluation/benchmarks/gpqa`](./benchmarks/gpqa)
- AgentBench: [`evaluation/benchmarks/agent_bench`](./benchmarks/agent_bench)
- MINT: [`evaluation/benchmarks/mint`](./benchmarks/mint)
- Entity deduction Arena (EDA): [`evaluation/benchmarks/EDA`](./benchmarks/EDA)
- ProofWriter: [`evaluation/benchmarks/logic_reasoning`](./benchmarks/logic_reasoning)
- ScienceAgentBench: [`evaluation/benchmarks/scienceagentbench`](./benchmarks/scienceagentbench)
+- GAIA: [`evaluation/gaia`](./gaia)
+- GPQA: [`evaluation/gpqa`](./gpqa)
+- AgentBench: [`evaluation/agent_bench`](./agent_bench)
+- MINT: [`evaluation/mint`](./mint)
+- Entity deduction Arena (EDA): [`evaluation/EDA`](./EDA)
+- ProofWriter: [`evaluation/logic_reasoning`](./logic_reasoning)

 ## Result Visualization

@@ -83,7 +79,7 @@ You can start your own fork of [our huggingface evaluation outputs](https://hugg

 To learn more about how to integrate your benchmark into OpenHands, check out [tutorial here](https://docs.all-hands.dev/modules/usage/how-to/evaluation-harness). Briefly,

- Each subfolder contains a specific benchmark or experiment. For example, [`evaluation/benchmarks/swe_bench`](./benchmarks/swe_bench) should contain
+- Each subfolder contains a specific benchmark or experiment. For example, `evaluation/swe_bench` should contain
 all the preprocessing/evaluation/analysis scripts.
 - Raw data and experimental records should not be stored within this repo.
 - For model outputs, they should be stored at [this huggingface space](https://huggingface.co/spaces/OpenHands/evaluation) for visualization.
--- a/evaluation/benchmarks/agent_bench/README.md
+++ b/evaluation/benchmarks/agent_bench/README.md
@@ -9,7 +9,7 @@ Please follow instruction [here](../README.md#setup) to setup your local develop
 ## Start the evaluation

 ```bash
-./evaluation/benchmarks/agent_bench/scripts/run_infer.sh [model_config] [git-version] [agent] [eval_limit]
+./evaluation/agent_bench/scripts/run_infer.sh [model_config] [git-version] [agent] [eval_limit]
 ```

 - `model_config`, e.g. `eval_gpt4_1106_preview`, is the config group name for your
@@ -25,7 +25,7 @@ in order to use `eval_limit`, you must also set `agent`.

 Following is the basic command to start the evaluation.

-You can update the arguments in the script `evaluation/benchmarks/agent_bench/scripts/run_infer.sh`, such as `--max-iterations`, `--eval-num-workers` and so on.
+You can update the arguments in the script `evaluation/agent_bench/scripts/run_infer.sh`, such as `--max-iterations`, `--eval-num-workers` and so on.

 - `--agent-cls`, the agent to use. For example, `CodeActAgent`.
 - `--llm-config`: the LLM configuration to use. For example, `eval_gpt4_1106_preview`.
@@ -34,23 +34,5 @@ You can update the arguments in the script `evaluation/benchmarks/agent_bench/sc
 - `--eval-n-limit`: the number of examples to evaluate. For example, `100`.

 ```bash
-./evaluation/benchmarks/agent_bench/scripts/run_infer.sh eval_gpt35_turbo HEAD CodeActAgent 1
+./evaluation/agent_bench/scripts/run_infer.sh eval_gpt35_turbo HEAD CodeActAgent 1
 ```
-
-## Run with Remote Runtime (experimental)
-
-You can run the evaluation using a remote runtime instead of a local Docker container. This is useful when you want to run the evaluation in a cloud environment or when you don't have Docker installed locally.
-
-To use the remote runtime, set the following environment variables:
-
-```bash
-# Required environment variables
-export ALLHANDS_API_KEY="your-api-key"  # Contact the team to get an API key
-export RUNTIME=remote
-export SANDBOX_REMOTE_RUNTIME_API_URL="https://runtime.eval.all-hands.dev"
-
-# Run the evaluation
-./evaluation/benchmarks/agent_bench/scripts/run_infer.sh llm.eval_gpt4_1106_preview HEAD CodeActAgent 1
-```
-
-The remote runtime will build a container image and run the evaluation in a cloud environment. The results will be saved locally in the same way as when running with a local runtime.
--- a/evaluation/benchmarks/agent_bench/init.py
+++ b/evaluation/benchmarks/agent_bench/init.py
--- a/evaluation/benchmarks/agent_bench/helper.py
+++ b/evaluation/benchmarks/agent_bench/helper.py
--- a/evaluation/benchmarks/agent_bench/run_infer.py
+++ b/evaluation/benchmarks/agent_bench/run_infer.py
@@ -7,7 +7,7 @@ from typing import Any
 import pandas as pd
 from datasets import load_dataset

-from evaluation.benchmarks.agent_bench.helper import (
+from evaluation.agent_bench.helper import (
    FAKE_RESPONSES,
    INST_SUFFIXES,
    compare_results,
@@ -43,16 +43,12 @@ def get_config(
    config = AppConfig(
        default_agent=metadata.agent_class,
        run_as_openhands=False,
-        runtime=os.environ.get('RUNTIME', 'eventstream'),
+        runtime='eventstream',
        max_iterations=metadata.max_iterations,
        sandbox=SandboxConfig(
-            base_container_image='python:3.12-slim',
+            base_container_image='python:3.12-bookworm',
            enable_auto_lint=True,
            use_host_network=False,
-            api_key=os.environ.get('ALLHANDS_API_KEY', None),
-            remote_runtime_api_url=os.environ.get('SANDBOX_REMOTE_RUNTIME_API_URL'),
-            keep_runtime_alive=False,
-            remote_runtime_init_timeout=3600,
        ),
        # do not mount workspace
        workspace_base=None,
--- a/evaluation/benchmarks/agent_bench/scripts/run_infer.sh
+++ b/evaluation/benchmarks/agent_bench/scripts/run_infer.sh
@@ -26,7 +26,7 @@ echo "AGENT: $AGENT"
 echo "AGENT_VERSION: $AGENT_VERSION"
 echo "MODEL_CONFIG: $MODEL_CONFIG"

-COMMAND="export PYTHONPATH=evaluation/benchmarks/agent_bench:\$PYTHONPATH && poetry run python evaluation/benchmarks/agent_bench/run_infer.py \
+COMMAND="export PYTHONPATH=evaluation/agent_bench:\$PYTHONPATH && poetry run python evaluation/agent_bench/run_infer.py \
  --agent-cls $AGENT \
  --llm-config $MODEL_CONFIG \
  --max-iterations 30 \
--- a/evaluation/benchmarks/agent_bench/scripts/summarise_results.py
+++ b/evaluation/benchmarks/agent_bench/scripts/summarise_results.py
--- a/evaluation/benchmarks/aider_bench/README.md
+++ b/evaluation/benchmarks/aider_bench/README.md
@@ -16,7 +16,7 @@ development environment and LLM.
 ## Start the evaluation

 ```bash
-./evaluation/benchmarks/aider_bench/scripts/run_infer.sh [model_config] [git-version] [agent] [eval_limit] [eval-num-workers] [eval_ids]
+./evaluation/aider_bench/scripts/run_infer.sh [model_config] [git-version] [agent] [eval_limit] [eval-num-workers] [eval_ids]
 ```

 - `model_config`, e.g. `eval_gpt4_1106_preview`, is the config group name for
@@ -42,7 +42,7 @@ export SKIP_NUM=12 # skip the first 12 instances from the dataset
 Following is the basic command to start the evaluation.

 You can update the arguments in the script
-`evaluation/benchmarks/aider_bench/scripts/run_infer.sh`, such as `--max-iterations`,
+`evaluation/aider_bench/scripts/run_infer.sh`, such as `--max-iterations`,
 `--eval-num-workers` and so on:

 - `--agent-cls`, the agent to use. For example, `CodeActAgent`.
@@ -53,7 +53,7 @@ You can update the arguments in the script
 - `--eval-ids`: the IDs of the examples to evaluate (comma separated). For example, `"1,3,10"`.

 ```bash
-./evaluation/benchmarks/aider_bench/scripts/run_infer.sh eval_gpt35_turbo HEAD CodeActAgent 100 1 "1,3,10"
+./evaluation/aider_bench/scripts/run_infer.sh eval_gpt35_turbo HEAD CodeActAgent 100 1 "1,3,10"
 ```

 ### Run Inference on `RemoteRuntime` (experimental)
@@ -61,25 +61,25 @@ You can update the arguments in the script
 This is in limited beta. Contact Xingyao over slack if you want to try this out!

 ```bash
-./evaluation/benchmarks/aider_bench/scripts/run_infer.sh [model_config] [git-version] [agent] [eval_limit] [eval-num-workers] [eval_ids]
+./evaluation/aider_bench/scripts/run_infer.sh [model_config] [git-version] [agent] [eval_limit] [eval-num-workers] [eval_ids]

 # Example - This runs evaluation on CodeActAgent for 133 instances on aider_bench test set, with 2 workers running in parallel
 export ALLHANDS_API_KEY="YOUR-API-KEY"
 export RUNTIME=remote
 export SANDBOX_REMOTE_RUNTIME_API_URL="https://runtime.eval.all-hands.dev"
-./evaluation/benchmarks/aider_bench/scripts/run_infer.sh llm.eval HEAD CodeActAgent 133 2
+./evaluation/aider_bench/scripts/run_infer.sh llm.eval HEAD CodeActAgent 133 2
 ```

 ## Summarize Results

 ```bash
-poetry run python ./evaluation/benchmarks/aider_bench/scripts/summarize_results.py [path_to_output_jsonl_file]
+poetry run python ./evaluation/aider_bench/scripts/summarize_results.py [path_to_output_jsonl_file]
 ```

 Full example:

 ```bash
-poetry run python ./evaluation/benchmarks/aider_bench/scripts/summarize_results.py evaluation/evaluation_outputs/outputs/AiderBench/CodeActAgent/claude-3-5-sonnet@20240620_maxiter_30_N_v1.9/output.jsonl
+poetry run python ./evaluation/aider_bench/scripts/summarize_results.py evaluation/evaluation_outputs/outputs/AiderBench/CodeActAgent/claude-3-5-sonnet@20240620_maxiter_30_N_v1.9/output.jsonl
 ```

 This will list the instances that passed and the instances that failed. For each
--- a/evaluation/benchmarks/aider_bench/create_dataset.py
+++ b/evaluation/benchmarks/aider_bench/create_dataset.py
--- a/evaluation/benchmarks/aider_bench/helper.py
+++ b/evaluation/benchmarks/aider_bench/helper.py
--- a/evaluation/benchmarks/aider_bench/run_infer.py
+++ b/evaluation/benchmarks/aider_bench/run_infer.py
@@ -7,7 +7,7 @@ from typing import Any
 import pandas as pd
 from datasets import load_dataset

-from evaluation.benchmarks.aider_bench.helper import (
+from evaluation.aider_bench.helper import (
    FAKE_RESPONSES,
    INST_SUFFIXES,
    INSTRUCTIONS_ADDENDUM,
--- a/evaluation/benchmarks/aider_bench/scripts/run_infer.sh
+++ b/evaluation/benchmarks/aider_bench/scripts/run_infer.sh
@@ -39,7 +39,7 @@ if [ "$USE_UNIT_TESTS" = true ]; then
  EVAL_NOTE=$EVAL_NOTE-w-test
 fi

-COMMAND="export PYTHONPATH=evaluation/benchmarks/aider_bench:\$PYTHONPATH && poetry run python evaluation/benchmarks/aider_bench/run_infer.py \
+COMMAND="export PYTHONPATH=evaluation/aider_bench:\$PYTHONPATH && poetry run python evaluation/aider_bench/run_infer.py \
  --agent-cls $AGENT \
  --llm-config $MODEL_CONFIG \
  --max-iterations 30 \
--- a/evaluation/benchmarks/aider_bench/scripts/summarize_results.py
+++ b/evaluation/benchmarks/aider_bench/scripts/summarize_results.py
--- a/evaluation/benchmarks/commit0_bench/README.md
+++ b/evaluation/benchmarks/commit0_bench/README.md
@@ -1,82 +0,0 @@
-# Commit0 Evaluation with OpenHands
-
-This folder contains the evaluation harness that we built on top of the original [Commit0](https://commit-0.github.io/) ([paper](TBD)).
-
-The evaluation consists of three steps:
-
-1. Environment setup: [install python environment](../README.md#development-environment), [configure LLM config](../README.md#configure-openhands-and-your-llm).
-2. [Run Evaluation](#run-inference-on-commit0-instances): Generate a edit patch for each Commit0 Repo, and get the evaluation results
-
-## Setup Environment and LLM Configuration
-
-Please follow instruction [here](../README.md#setup) to setup your local development environment and LLM.
-
-## OpenHands Commit0 Instance-level Docker Support
-
-OpenHands supports using the Commit0 Docker for **[inference](#run-inference-on-commit0-instances).
-This is now the default behavior.
-
-
-## Run Inference on Commit0 Instances
-
-Make sure your Docker daemon is running, and you have ample disk space (at least 200-500GB, depends on the Commit0 set you are running on) for the [instance-level docker image](#openhands-commit0-instance-level-docker-support).
-
-When the `run_infer.sh` script is started, it will automatically pull the `lite` split in Commit0. For example, for instance ID `commit-0/minitorch`, it will try to pull our pre-build docker image `wentingzhao/minitorch` from DockerHub. This image will be used create an OpenHands runtime image where the agent will operate on.
-
-```bash
-./evaluation/benchmarks/commit0_bench/scripts/run_infer.sh [repo_split] [model_config] [git-version] [agent] [eval_limit] [max_iter] [num_workers] [dataset] [dataset_split]
-
-# Example
-./evaluation/benchmarks/commit0_bench/scripts/run_infer.sh lite llm.eval_sonnet HEAD CodeActAgent 16 100 8 wentingzhao/commit0_combined test
-```
-
-where `model_config` is mandatory, and the rest are optional.
-
- `repo_split`, e.g. `lite`, is the split of the Commit0 dataset you would like to evaluate on. Available options are `lite`, `all` and each individual repo.
- `model_config`, e.g. `eval_gpt4_1106_preview`, is the config group name for your
-LLM settings, as defined in your `config.toml`.
- `git-version`, e.g. `HEAD`, is the git commit hash of the OpenHands version you would
-like to evaluate. It could also be a release tag like `0.6.2`.
- `agent`, e.g. `CodeActAgent`, is the name of the agent for benchmarks, defaulting
-to `CodeActAgent`.
- `eval_limit`, e.g. `10`, limits the evaluation to the first `eval_limit` instances. By
-default, the script evaluates the `lite` split of the Commit0 dataset (16 repos). Note:
-in order to use `eval_limit`, you must also set `agent`.
- `max_iter`, e.g. `20`, is the maximum number of iterations for the agent to run. By
-default, it is set to 30.
- `num_workers`, e.g. `3`, is the number of parallel workers to run the evaluation. By
-default, it is set to 1.
- `dataset`, a huggingface dataset name. e.g. `wentingzhao/commit0_combined`, specifies which dataset to evaluate on.
- `dataset_split`, split for the huggingface dataset. Notice only `test` is supported for Commit0.
-
-Note that the `USE_INSTANCE_IMAGE` environment variable is always set to `true` for Commit0.
-
-Let's say you'd like to run 10 instances using `llm.eval_sonnet` and CodeActAgent,
-
-then your command would be:
-
-```bash
-./evaluation/benchmarks/commit0_bench/scripts/run_infer.sh lite llm.eval_sonnet HEAD CodeActAgent 10 30 1 wentingzhao/commit0_combined test
-```
-
-### Run Inference on `RemoteRuntime` (experimental)
-
-This is in limited beta. Contact Xingyao over slack if you want to try this out!
-
-```bash
-./evaluation/benchmarks/commit0_bench/scripts/run_infer.sh [repo_split] [model_config] [git-version] [agent] [eval_limit] [max_iter] [num_workers] [dataset] [dataset_split]
-
-# Example - This runs evaluation on CodeActAgent for 10 instances on "wentingzhao/commit0_combined"'s test set, with max 30 iteration per instances, with 1 number of workers running in parallel
-ALLHANDS_API_KEY="YOUR-API-KEY" RUNTIME=remote SANDBOX_REMOTE_RUNTIME_API_URL="https://runtime.eval.all-hands.dev" EVAL_DOCKER_IMAGE_PREFIX="docker.io/wentingzhao" \
-./evaluation/benchmarks/commit0_bench/scripts/run_infer.sh lite llm.eval_sonnet HEAD CodeActAgent 10 30 1 wentingzhao/commit0_combined test
-```
-
-To clean-up all existing runtime you've already started, run:
-
-```bash
-ALLHANDS_API_KEY="YOUR-API-KEY" ./evaluation/benchmarks/commit0_bench/scripts/cleanup_remote_runtime.sh
-```
-
-### Specify a subset of tasks to run infer
-
-If you would like to specify a list of tasks you'd like to benchmark on, you just need to pass selected repo through `repo_split` option.
--- a/evaluation/benchmarks/commit0_bench/run_infer.py
+++ b/evaluation/benchmarks/commit0_bench/run_infer.py
@@ -1,606 +0,0 @@
-import asyncio
-import json
-import os
-from collections import Counter
-from typing import Any
-
-import pandas as pd
-from commit0.harness.constants import SPLIT
-from datasets import load_dataset
-
-import openhands.agenthub
-from evaluation.utils.shared import (
-    EvalException,
-    EvalMetadata,
-    EvalOutput,
-    assert_and_raise,
-    codeact_user_response,
-    make_metadata,
-    prepare_dataset,
-    reset_logger_for_multiprocessing,
-    run_evaluation,
-    update_llm_config_for_completions_logging,
-)
-from openhands.controller.state.state import State
-from openhands.core.config import (
-    AgentConfig,
-    AppConfig,
-    SandboxConfig,
-    get_llm_config_arg,
-    get_parser,
-)
-from openhands.core.logger import openhands_logger as logger
-from openhands.core.main import create_runtime, run_controller
-from openhands.events.action import CmdRunAction, MessageAction
-from openhands.events.observation import CmdOutputObservation, ErrorObservation
-from openhands.events.serialization.event import event_to_dict
-from openhands.runtime.base import Runtime
-from openhands.utils.async_utils import call_async_from_sync
-from openhands.utils.shutdown_listener import sleep_if_should_continue
-
-USE_HINT_TEXT = os.environ.get('USE_HINT_TEXT', 'false').lower() == 'true'
-USE_INSTANCE_IMAGE = os.environ.get('USE_INSTANCE_IMAGE', 'false').lower() == 'true'
-RUN_WITH_BROWSING = os.environ.get('RUN_WITH_BROWSING', 'false').lower() == 'true'
-
-AGENT_CLS_TO_FAKE_USER_RESPONSE_FN = {
-    'CodeActAgent': codeact_user_response,
-    'CodeActCommit0Agent': codeact_user_response,
-}
-
-
-def _get_commit0_workspace_dir_name(instance: pd.Series) -> str:
-    return instance['repo'].split('/')[1]
-
-
-def get_instruction(instance: pd.Series, metadata: EvalMetadata):
-    workspace_dir_name = _get_commit0_workspace_dir_name(instance)
-    # Prepare instruction
-    test_cmd = instance['test']['test_cmd']
-    test_dir = instance['test']['test_dir']
-    # Instruction based on Anthropic's official trajectory
-    # https://github.com/eschluntz/swe-bench-experiments/tree/main/evaluation/verified/20241022_tools_claude-3-5-sonnet-updated/trajs
-    instruction = (
-        '<uploaded_files>\n'
-        f'/workspace/{workspace_dir_name}\n'
-        '</uploaded_files>\n'
-        f"I've uploaded a python code repository in the directory {workspace_dir_name}. Here is your task:\n\n"
-        'Here is your task:\n\n'
-        '  You need to complete the implementations for all functions (i.e., those with pass\n'
-        '  statements) and pass the unit tests.\n\n'
-        '  Do not change the names of existing functions or classes, as they may be referenced\n'
-        '  from other code like unit tests, etc.\n\n'
-        '  When you generate code, you must maintain the original formatting of the function\n'
-        '  stubs (such as whitespaces), otherwise we will not able to search/replace blocks\n'
-        '  for code modifications, and therefore you will receive a score of 0 for your generated\n'
-        '  code.'
-        '\n\n'
-        'Here is the command to run the unit tests:\n'
-        '<test_command>\n'
-        f'{test_cmd} {test_dir}\n'
-        '</test_command>\n\n'
-        'Make a local git commit for each agent step for all code changes. If there is not change in current step, do not make a commit.'
-    )
-
-    if RUN_WITH_BROWSING:
-        instruction += (
-            '<IMPORTANT!>\n'
-            'You SHOULD NEVER attempt to browse the web. '
-            '</IMPORTANT!>\n'
-        )
-    return instruction
-
-
-# TODO: migrate all swe-bench docker to ghcr.io/openhands
-DOCKER_IMAGE_PREFIX = os.environ.get(
-    'EVAL_DOCKER_IMAGE_PREFIX', 'docker.io/wentingzhao/'
-)
-logger.info(f'Using docker image prefix: {DOCKER_IMAGE_PREFIX}')
-
-
-def get_instance_docker_image(repo_name: str) -> str:
-    return (DOCKER_IMAGE_PREFIX.rstrip('/') + '/' + repo_name).lower() + ':v0'
-
-
-def get_config(
-    instance: pd.Series,
-    metadata: EvalMetadata,
-) -> AppConfig:
-    # COMMIT0_CONTAINER_IMAGE = 'wentingzhao/'
-    assert USE_INSTANCE_IMAGE
-    # We use a different instance image for the each instance of commit0 eval
-    repo_name = instance['repo'].split('/')[1]
-    base_container_image = get_instance_docker_image(repo_name)
-    logger.info(
-        f'Using instance container image: {base_container_image}. '
-        f'Please make sure this image exists. '
-        f'Submit an issue on https://github.com/All-Hands-AI/OpenHands if you run into any issues.'
-    )
-    # else:
-    #     raise
-    # base_container_image = SWE_BENCH_CONTAINER_IMAGE
-    # logger.info(f'Using swe-bench container image: {base_container_image}')
-
-    config = AppConfig(
-        default_agent=metadata.agent_class,
-        run_as_openhands=False,
-        max_iterations=metadata.max_iterations,
-        runtime=os.environ.get('RUNTIME', 'eventstream'),
-        sandbox=SandboxConfig(
-            base_container_image=base_container_image,
-            enable_auto_lint=True,
-            use_host_network=False,
-            # large enough timeout, since some testcases take very long to run
-            timeout=300,
-            api_key=os.environ.get('ALLHANDS_API_KEY', None),
-            remote_runtime_api_url=os.environ.get('SANDBOX_REMOTE_RUNTIME_API_URL'),
-            keep_runtime_alive=False,
-            remote_runtime_init_timeout=3600,
-        ),
-        # do not mount workspace
-        workspace_base=None,
-        workspace_mount_path=None,
-    )
-    config.set_llm_config(
-        update_llm_config_for_completions_logging(
-            metadata.llm_config, metadata.eval_output_dir, instance['instance_id']
-        )
-    )
-    agent_config = AgentConfig(
-        codeact_enable_jupyter=False,
-        codeact_enable_browsing=RUN_WITH_BROWSING,
-        codeact_enable_llm_editor=False,
-    )
-    config.set_agent_config(agent_config)
-    return config
-
-
-def initialize_runtime(
-    runtime: Runtime,
-    instance: pd.Series,  # this argument is not required
-):
-    """Initialize the runtime for the agent.
-
-    This function is called before the runtime is used to run the agent.
-    """
-    logger.info('-' * 30)
-    logger.info('BEGIN Runtime Initialization Fn')
-    logger.info('-' * 30)
-    workspace_dir_name = _get_commit0_workspace_dir_name(instance)
-    obs: CmdOutputObservation
-
-    action = CmdRunAction(
-        command=f'git clone -b commit0_combined https://github.com/{instance["repo"]}.git'
-    )
-    action.timeout = 600
-    logger.info(action, extra={'msg_type': 'ACTION'})
-    obs = runtime.run_action(action)
-    logger.info(obs, extra={'msg_type': 'OBSERVATION'})
-    assert_and_raise(
-        obs.exit_code == 0,
-        f'Failed to git clone -b commit0_combined https://github.com/{instance["repo"]}.git: {str(obs)}',
-    )
-
-    action = CmdRunAction(command=f'cd /workspace/{workspace_dir_name}')
-    action.timeout = 600
-    logger.info(action, extra={'msg_type': 'ACTION'})
-    obs = runtime.run_action(action)
-    logger.info(obs, extra={'msg_type': 'OBSERVATION'})
-    assert_and_raise(
-        obs.exit_code == 0,
-        f'Failed to cd to /workspace/{workspace_dir_name}: {str(obs)}',
-    )
-
-    action = CmdRunAction(command='git checkout -b openhands')
-    action.timeout = 600
-    logger.info(action, extra={'msg_type': 'ACTION'})
-    obs = runtime.run_action(action)
-    logger.info(obs, extra={'msg_type': 'OBSERVATION'})
-    assert_and_raise(
-        obs.exit_code == 0, f'Failed to git checkout new branch openhands: {str(obs)}'
-    )
-
-    # Install commit0
-    action = CmdRunAction(command='/root/.cargo/bin/uv pip install commit0')
-    action.timeout = 600
-    logger.info(action, extra={'msg_type': 'ACTION'})
-    obs = runtime.run_action(action)
-    # logger.info(obs, extra={'msg_type': 'OBSERVATION'})
-    assert_and_raise(
-        obs.exit_code == 0,
-        f'Failed to install commit0: {str(obs)}',
-    )
-    logger.info('-' * 30)
-    logger.info('END Runtime Initialization Fn')
-    logger.info('-' * 30)
-
-
-def complete_runtime(
-    runtime: Runtime,
-    instance: pd.Series,  # this argument is not required, but it is used to get the workspace_dir_name
-) -> dict[str, Any]:
-    """Complete the runtime for the agent.
-
-    This function is called before the runtime is used to run the agent.
-    If you need to do something in the sandbox to get the correctness metric after
-    the agent has run, modify this function.
-    """
-    logger.info('-' * 30)
-    logger.info('BEGIN Runtime Completion Fn')
-    logger.info('-' * 30)
-    obs: CmdOutputObservation
-    workspace_dir_name = _get_commit0_workspace_dir_name(instance)
-
-    action = CmdRunAction(command='git add .')
-    action.timeout = 600
-    logger.info(action, extra={'msg_type': 'ACTION'})
-    obs = runtime.run_action(action)
-    logger.info(obs, extra={'msg_type': 'OBSERVATION'})
-    assert_and_raise(
-        isinstance(obs, CmdOutputObservation) and obs.exit_code == 0,
-        f'Failed to git add -A: {str(obs)}',
-    )
-
-    action = CmdRunAction(command='git commit -m "openhands edits"')
-    action.timeout = 600
-    logger.info(action, extra={'msg_type': 'ACTION'})
-    obs = runtime.run_action(action)
-    logger.info(obs, extra={'msg_type': 'OBSERVATION'})
-    assert_and_raise(
-        isinstance(obs, CmdOutputObservation)
-        and (obs.exit_code == 0 or obs.exit_code == 1),
-        f'Failed to git commit -m "openhands": {str(obs)}',
-    )
-
-    # Generate diff patch compared to base commit, excluding spec.pdf.bz2 files
-    n_retries = 0
-    git_patch = None
-    while n_retries < 5:
-        action = CmdRunAction(
-            command=f"git diff {instance['base_commit']} HEAD -- . ':(exclude)spec.pdf.bz2'"
-        )
-        action.timeout = 600 + 100 * n_retries
-        logger.info(action, extra={'msg_type': 'ACTION'})
-        obs = runtime.run_action(action)
-        # logger.info(obs, extra={'msg_type': 'OBSERVATION'})
-        n_retries += 1
-        if isinstance(obs, CmdOutputObservation):
-            if obs.exit_code == 0:
-                git_patch = obs.content.strip()
-                break
-            else:
-                logger.info('Failed to get git diff, retrying...')
-                sleep_if_should_continue(10)
-        elif isinstance(obs, ErrorObservation):
-            logger.error(f'Error occurred: {obs.content}. Retrying...')
-            sleep_if_should_continue(10)
-        else:
-            assert_and_raise(False, f'Unexpected observation type: {str(obs)}')
-
-    assert_and_raise(git_patch is not None, 'Failed to get git diff (None)')
-
-    test_dir = instance['test']['test_dir']
-    action = CmdRunAction(
-        command=f"{instance['test']['test_cmd']} --json-report --json-report-file=report.json --continue-on-collection-errors {test_dir} > test_output.txt 2>&1"
-    )
-    action.timeout = 600
-    logger.info(action, extra={'msg_type': 'ACTION'})
-    obs = runtime.run_action(action)
-    logger.info(obs, extra={'msg_type': 'OBSERVATION'})
-    assert_and_raise(
-        isinstance(obs, CmdOutputObservation),
-        f'Failed to run test command: {str(obs)}',
-    )
-    # Read test output
-    action = CmdRunAction(command='cat test_output.txt')
-    action.timeout = 600
-    logger.info(action, extra={'msg_type': 'ACTION'})
-    obs = runtime.run_action(action)
-    # logger.info(obs, extra={'msg_type': 'OBSERVATION'})
-    assert_and_raise(
-        isinstance(obs, CmdOutputObservation),
-        f'Failed to read test output: {str(obs)}',
-    )
-    test_output = obs.content.strip()
-    # logger.info(f'Test output: {test_output}')
-
-    # Save pytest exit code
-    action = CmdRunAction(command='echo $?')
-    action.timeout = 600
-    logger.info(action, extra={'msg_type': 'ACTION'})
-    obs = runtime.run_action(action)
-    # logger.info(obs, extra={'msg_type': 'OBSERVATION'})
-    assert_and_raise(
-        isinstance(obs, CmdOutputObservation) and obs.exit_code == 0,
-        f'Failed to save pytest exit code: {str(obs)}',
-    )
-    pytest_exit_code = obs.content.strip()
-    # logger.info(f'Pytest exit code: {pytest_exit_code}')
-
-    # Read the test report
-    action = CmdRunAction(command='cat report.json')
-    action.timeout = 600
-    logger.info(action, extra={'msg_type': 'ACTION'})
-    obs = runtime.run_action(action)
-    # logger.info(obs, extra={'msg_type': 'OBSERVATION'})
-    assert_and_raise(
-        isinstance(obs, CmdOutputObservation),
-        f'Failed to read test report: {str(obs)}',
-    )
-    # Get test IDs from instance
-    repo_name = instance['repo'].split('/')[1]
-    repo_name = repo_name.replace('.', '-')
-    action = CmdRunAction(command=f'commit0 get-tests {repo_name}')
-    action.timeout = 600
-    logger.info(action, extra={'msg_type': 'ACTION'})
-    obs = runtime.run_action(action)
-    # logger.info(obs, extra={'msg_type': 'OBSERVATION'})
-    test_ids = obs.content.strip().split('\n')
-
-    try:
-        report = json.loads(obs.content)
-        tests = {x['nodeid']: x['call'] for x in report['tests'] if 'call' in x}
-
-        # Calculate test statistics
-        status = []
-        runtimes = []
-        no_runs = 0
-
-        for test_id in test_ids:
-            if test_id in tests and tests[test_id] is not None:
-                status.append(tests[test_id]['outcome'])
-                runtimes.append(tests[test_id]['duration'])
-                no_runs += 1
-            else:
-                status.append('failed')
-                runtimes.append(0)
-
-        status_counts = Counter(status)
-        total_runtime = sum(runtimes) if no_runs > 0 else 0
-        num_passed = status_counts.get('passed', 0) + status_counts.get('xfail', 0)
-        passed_ratio = num_passed / len(status) if status else 0
-
-        eval_result = {
-            'name': workspace_dir_name,
-            'sum': total_runtime,
-            'passed': passed_ratio,
-            'num_passed': num_passed,
-            'num_tests': len(test_ids),
-        }
-
-    except json.JSONDecodeError:
-        logger.error('Failed to parse test report JSON')
-        eval_result = {
-            'name': workspace_dir_name,
-            'sum': 0,
-            'passed': 0,
-            'num_passed': 0,
-            'num_tests': len(test_ids),
-        }
-
-    # Create tarball of workspace
-    temp_zip = runtime.copy_from(f'/workspace/{workspace_dir_name}')
-
-    commit0_dir = os.path.dirname(__file__)
-    persistent_zip = os.path.join(commit0_dir, f'{workspace_dir_name}.zip')
-    with open(temp_zip, 'rb') as src, open(persistent_zip, 'wb') as dst:
-        dst.write(src.read())
-    zip_file = persistent_zip
-    return {
-        'eval_result': eval_result,
-        'git_patch': git_patch,
-        'test_output': test_output,
-        'pytest_exit_code': pytest_exit_code,
-        'zip_file': zip_file,
-    }
-
-
-def process_instance(
-    instance: pd.Series,
-    metadata: EvalMetadata,
-    reset_logger: bool = True,
-) -> EvalOutput:
-    config = get_config(instance, metadata)
-    # Setup the logger properly, so you can run multi-processing to parallelize the evaluation
-    if reset_logger:
-        log_dir = os.path.join(metadata.eval_output_dir, 'infer_logs')
-        reset_logger_for_multiprocessing(logger, instance.instance_id, log_dir)
-    else:
-        logger.info(f'Starting evaluation for instance {instance.instance_id}.')
-
-    runtime = create_runtime(config)
-    call_async_from_sync(runtime.connect)
-    try:
-        initialize_runtime(runtime, instance)
-
-        instruction = get_instruction(instance, metadata)
-
-        # Here's how you can run the agent (similar to the `main` function) and get the final task state
-        state: State | None = asyncio.run(
-            run_controller(
-                config=config,
-                initial_user_action=MessageAction(content=instruction),
-                runtime=runtime,
-                fake_user_response_fn=AGENT_CLS_TO_FAKE_USER_RESPONSE_FN[
-                    metadata.agent_class
-                ],
-            )
-        )
-
-        # if fatal error, throw EvalError to trigger re-run
-        if (
-            state.last_error
-            and 'fatal error during agent execution' in state.last_error
-            and 'stuck in a loop' not in state.last_error
-        ):
-            raise EvalException('Fatal error detected: ' + state.last_error)
-
-        # ======= THIS IS Commit0 specific =======
-        # Get git patch
-        return_val = complete_runtime(runtime, instance)
-        eval_result = return_val['eval_result']
-        git_patch = return_val['git_patch']
-        test_output = return_val['test_output']
-        pytest_exit_code = return_val['pytest_exit_code']
-        zip_file = return_val['zip_file']
-
-        repo_name = instance['repo'].split('/')[1]
-        zip_dest = os.path.join(
-            metadata.eval_output_dir, 'repos', repo_name, f'{repo_name}.zip'
-        )
-        patch_file = os.path.join(
-            metadata.eval_output_dir, 'repos', repo_name, f'{repo_name}_patch.diff'
-        )
-        test_output_file = os.path.join(
-            metadata.eval_output_dir, 'repos', repo_name, f'{repo_name}_test_output.txt'
-        )
-        pytest_exit_code_file = os.path.join(
-            metadata.eval_output_dir,
-            'repos',
-            repo_name,
-            f'{repo_name}_pytest_exit_code.txt',
-        )
-
-        os.makedirs(os.path.dirname(zip_dest), exist_ok=True)
-        os.rename(zip_file, zip_dest)
-
-        write_targets = [
-            (patch_file, git_patch),
-            (test_output_file, test_output),
-            (pytest_exit_code_file, pytest_exit_code),
-        ]
-
-        for write_target in write_targets:
-            with open(write_target[0], 'w') as f:
-                f.write(write_target[1])
-
-        logger.info(
-            f'Got evaluation result for repo {instance.instance_id}:\n--------\n{eval_result}\n--------'
-        )
-    finally:
-        runtime.close()
-    # ==========================================
-
-    # ======= Attempt to evaluate the agent's edits =======
-    # we use eval_infer.sh to evaluate the agent's edits, not here
-    # because the agent may alter the environment / testcases
-    test_result = {
-        'eval_result': eval_result,
-    }
-
-    # If you are working on some simpler benchmark that only evaluates the final model output (e.g., in a MessageAction)
-    # You can simply get the LAST `MessageAction` from the returned `state.history` and parse it for evaluation.
-    if state is None:
-        raise ValueError('State should not be None.')
-
-    # NOTE: this is NO LONGER the event stream, but an agent history that includes delegate agent's events
-    histories = [event_to_dict(event) for event in state.history]
-    metrics = state.metrics.get() if state.metrics else None
-
-    # Save the output
-    output = EvalOutput(
-        instance_id=instance.instance_id,
-        instruction=instruction,
-        instance=instance.to_dict(),
-        test_result=test_result,
-        metadata=metadata,
-        history=histories,
-        metrics=metrics,
-        error=state.last_error if state and state.last_error else None,
-    )
-    return output
-
-
-def commit0_setup(dataset: pd.DataFrame, repo_split: str) -> pd.DataFrame:
-    """Setup Commit0 dataset based on split type.
-
-    Args:
-        dataset: Full Commit0 dataset
-        repo_split: Split type ('all', 'lite' or specific repo name)
-
-    Returns:
-        Filtered dataset based on split type
-    """
-
-    filtered_dataset = pd.concat(
-        [
-            dataset[dataset['repo'].str.split('/').str[1] == repo]
-            for repo in SPLIT.get(repo_split, [])
-        ]
-    )
-
-    # Drop setup column if it exists
-    if 'setup' in filtered_dataset.columns:
-        filtered_dataset = filtered_dataset.drop('setup', axis=1)
-
-    # Replace all forward slashes in instance_id with hyphens
-    filtered_dataset['instance_id'] = filtered_dataset['repo'].str.split('/').str[1]
-
-    return filtered_dataset
-
-
-if __name__ == '__main__':
-    parser = get_parser()
-    parser.add_argument(
-        '--dataset',
-        type=str,
-        default='wentingzhao/commit0_combined',
-        help='dataset to evaluate on, only test split exists for this HF dataset',
-    )
-    parser.add_argument(
-        '--split',
-        type=str,
-        default='test',
-        help='this is the HF dataset split',
-    )
-    parser.add_argument(
-        '--repo-split',
-        type=str,
-        default='lite',
-        help='all, lite, or each repo name',
-    )
-    args, _ = parser.parse_known_args()
-
-    # NOTE: It is preferable to load datasets from huggingface datasets and perform post-processing
-    # so we don't need to manage file uploading to OpenHands's repo
-    dataset = load_dataset(args.dataset, split=args.split)
-
-    commit0_datasets = commit0_setup(dataset.to_pandas(), args.repo_split)
-
-    logger.info(f'Loaded dataset {args.dataset} with reposplit {args.repo_split}')
-
-    llm_config = None
-    if args.llm_config:
-        llm_config = get_llm_config_arg(args.llm_config)
-        llm_config.log_completions = True
-
-    if llm_config is None:
-        raise ValueError(f'Could not find LLM config: --llm_config {args.llm_config}')
-
-    details = {}
-    _agent_cls = openhands.agenthub.Agent.get_cls(args.agent_cls)
-
-    dataset_descrption = (
-        args.dataset.replace('/', '__') + '-' + args.repo_split.replace('/', '__')
-    )
-    metadata = make_metadata(
-        llm_config,
-        dataset_descrption,
-        args.agent_cls,
-        args.max_iterations,
-        args.eval_note,
-        args.eval_output_dir,
-        details=details,
-    )
-
-    output_file = os.path.join(metadata.eval_output_dir, 'output.jsonl')
-
-    instances = prepare_dataset(commit0_datasets, output_file, args.eval_n_limit)
-
-    run_evaluation(
-        instances,
-        metadata,
-        output_file,
-        args.eval_num_workers,
-        process_instance,
-        timeout_seconds=120 * 60,  # 2 hour PER instance should be more than enough
-    )
--- a/evaluation/benchmarks/commit0_bench/scripts/run_infer.sh
+++ b/evaluation/benchmarks/commit0_bench/scripts/run_infer.sh
@@ -1,125 +0,0 @@
-#!/bin/bash
-set -eo pipefail
-
-source "evaluation/utils/version_control.sh"
-
-REPO_SPLIT=$1
-MODEL_CONFIG=$2
-COMMIT_HASH=$3
-AGENT=$4
-EVAL_LIMIT=$5
-MAX_ITER=$6
-NUM_WORKERS=$7
-DATASET=$8
-SPLIT=$9
-N_RUNS=${10}
-
-if [ -z "$NUM_WORKERS" ]; then
-  NUM_WORKERS=1
-  echo "Number of workers not specified, use default $NUM_WORKERS"
-fi
-checkout_eval_branch
-
-if [ -z "$AGENT" ]; then
-  echo "Agent not specified, use default CodeActAgent"
-  AGENT="CodeActAgent"
-fi
-
-if [ -z "$MAX_ITER" ]; then
-  echo "MAX_ITER not specified, use default 100"
-  MAX_ITER=100
-fi
-
-if [ -z "$USE_INSTANCE_IMAGE" ]; then
-  echo "USE_INSTANCE_IMAGE not specified, use default true"
-  USE_INSTANCE_IMAGE=true
-fi
-
-if [ -z "$RUN_WITH_BROWSING" ]; then
-  echo "RUN_WITH_BROWSING not specified, use default false"
-  RUN_WITH_BROWSING=false
-fi
-
-
-if [ -z "$DATASET" ]; then
-  echo "DATASET not specified, use default wentingzhao/commit0_combined"
-  DATASET="wentingzhao/commit0_combined"
-fi
-
-if [ -z "$REPO_SPLIT" ]; then
-  echo "REPO_SPLIT not specified, use default lite"
-  REPO_SPLIT=0
-fi
-
-if [ -z "$SPLIT" ]; then
-  echo "HF SPLIT not specified, use default test"
-  SPLIT="test"
-fi
-
-export USE_INSTANCE_IMAGE=$USE_INSTANCE_IMAGE
-echo "USE_INSTANCE_IMAGE: $USE_INSTANCE_IMAGE"
-export RUN_WITH_BROWSING=$RUN_WITH_BROWSING
-echo "RUN_WITH_BROWSING: $RUN_WITH_BROWSING"
-
-get_agent_version
-
-echo "AGENT: $AGENT"
-echo "AGENT_VERSION: $AGENT_VERSION"
-echo "MODEL_CONFIG: $MODEL_CONFIG"
-echo "DATASET: $DATASET"
-echo "HF SPLIT: $SPLIT"
-echo "REPO SPLIT: $REPO_SPLIT"
-
-# Default to NOT use Hint
-if [ -z "$USE_HINT_TEXT" ]; then
-  export USE_HINT_TEXT=false
-fi
-echo "USE_HINT_TEXT: $USE_HINT_TEXT"
-EVAL_NOTE="$AGENT_VERSION"
-# if not using Hint, add -no-hint to the eval note
-if [ "$USE_HINT_TEXT" = false ]; then
-  EVAL_NOTE="$EVAL_NOTE-no-hint"
-fi
-
-if [ "$RUN_WITH_BROWSING" = true ]; then
-  EVAL_NOTE="$EVAL_NOTE-with-browsing"
-fi
-
-if [ -n "$EXP_NAME" ]; then
-  EVAL_NOTE="$EVAL_NOTE-$EXP_NAME"
-fi
-
-function run_eval() {
-  local eval_note=$1
-  COMMAND="poetry run python evaluation/benchmarks/commit0_bench/run_infer.py \
-    --agent-cls $AGENT \
-    --llm-config $MODEL_CONFIG \
-    --max-iterations $MAX_ITER \
-    --eval-num-workers $NUM_WORKERS \
-    --eval-note $eval_note \
-    --dataset $DATASET \
-    --split $SPLIT \
-    --repo-split $REPO_SPLIT"
-
-  if [ -n "$EVAL_LIMIT" ]; then
-    echo "EVAL_LIMIT: $EVAL_LIMIT"
-    COMMAND="$COMMAND --eval-n-limit $EVAL_LIMIT"
-  fi
-
-  # Run the command
-  eval $COMMAND
-}
-
-unset SANDBOX_ENV_GITHUB_TOKEN # prevent the agent from using the github token to push
-if [ -z "$N_RUNS" ]; then
-  N_RUNS=1
-  echo "N_RUNS not specified, use default $N_RUNS"
-fi
-
-for i in $(seq 1 $N_RUNS); do
-  current_eval_note="$EVAL_NOTE-run_$i"
-  echo "EVAL_NOTE: $current_eval_note"
-  run_eval $current_eval_note
-done
-
-checkout_original_branch
--- a/evaluation/benchmarks/swe_bench/scripts/cleanup_remote_runtime.sh
+++ b/evaluation/benchmarks/swe_bench/scripts/cleanup_remote_runtime.sh
@@ -1,33 +0,0 @@
-#!/bin/bash
-
-
-# API base URL
-BASE_URL="https://runtime.eval.all-hands.dev"
-
-# Get the list of runtimes
-response=$(curl --silent --location --request GET "${BASE_URL}/list" \
-  --header "X-API-Key: ${ALLHANDS_API_KEY}")
-
-n_runtimes=$(echo $response | jq -r '.total')
-echo "Found ${n_runtimes} runtimes. Stopping them..."
-
-runtime_ids=$(echo $response | jq -r '.runtimes | .[].runtime_id')
-
-# Function to stop a single runtime
-stop_runtime() {
-  local runtime_id=$1
-  local counter=$2
-  echo "Stopping runtime ${counter}/${n_runtimes}: ${runtime_id}"
-  curl --silent --location --request POST "${BASE_URL}/stop" \
-    --header "X-API-Key: ${ALLHANDS_API_KEY}" \
-    --header "Content-Type: application/json" \
-    --data-raw "{\"runtime_id\": \"${runtime_id}\"}"
-  echo
-}
-export -f stop_runtime
-export BASE_URL ALLHANDS_API_KEY n_runtimes
-
-# Use GNU Parallel to stop runtimes in parallel
-echo "$runtime_ids" | parallel -j 16 --progress stop_runtime {} {#}
-
-echo "All runtimes have been stopped."
--- a/evaluation/benchmarks/biocoder/README.md
+++ b/evaluation/benchmarks/biocoder/README.md
@@ -21,7 +21,7 @@ To reproduce this image, please see the Dockerfile_Openopenhands in the `biocode


 ```bash
-./evaluation/benchmarks/biocoder/scripts/run_infer.sh [model_config] [git-version] [agent] [eval_limit]
+./evaluation/biocoder/scripts/run_infer.sh [model_config] [git-version] [agent] [eval_limit]
 ```

 where `model_config` is mandatory, while `git-version`, `agent`, `dataset` and `eval_limit` are optional.
@@ -43,7 +43,7 @@ with current OpenHands version, then your command would be:
 ## Examples

 ```bash
-./evaluation/benchmarks/biocoder/scripts/run_infer.sh eval_gpt4o_2024_05_13 HEAD CodeActAgent 1
+./evaluation/biocoder/scripts/run_infer.sh eval_gpt4o_2024_05_13 HEAD CodeActAgent 1
 ```

 ## Reference
--- a/evaluation/benchmarks/biocoder/run_infer.py
+++ b/evaluation/benchmarks/biocoder/run_infer.py
@@ -8,7 +8,7 @@ from typing import Any
 import pandas as pd
 from datasets import load_dataset

-from evaluation.benchmarks.biocoder.utils import BiocoderData
+from evaluation.biocoder.utils import BiocoderData
 from evaluation.utils.shared import (
    EvalMetadata,
    EvalOutput,
--- a/evaluation/benchmarks/biocoder/scripts/run_infer.sh
+++ b/evaluation/benchmarks/biocoder/scripts/run_infer.sh
@@ -28,7 +28,7 @@ echo "AGENT_VERSION: $AGENT_VERSION"
 echo "MODEL_CONFIG: $MODEL_CONFIG"
 echo "DATASET: $DATASET"

-COMMAND="poetry run python evaluation/benchmarks/biocoder/run_infer.py \
+COMMAND="poetry run python evaluation/biocoder/run_infer.py \
  --agent-cls $AGENT \
  --llm-config $MODEL_CONFIG \
  --max-iterations 10 \
--- a/evaluation/benchmarks/biocoder/scripts/setup/copy_changed_code.py
+++ b/evaluation/benchmarks/biocoder/scripts/setup/copy_changed_code.py
--- a/evaluation/benchmarks/biocoder/scripts/setup/remove_code.py
+++ b/evaluation/benchmarks/biocoder/scripts/setup/remove_code.py
--- a/evaluation/benchmarks/biocoder/utils.py
+++ b/evaluation/benchmarks/biocoder/utils.py
--- a/evaluation/benchmarks/bird/README.md
+++ b/evaluation/benchmarks/bird/README.md
@@ -9,7 +9,7 @@ Please follow instruction [here](../README.md#setup) to setup your local develop
 ## Run Inference on Bird

 ```bash
-./evaluation/benchmarks/bird/scripts/run_infer.sh [model_config] [git-version]
+./evaluation/bird/scripts/run_infer.sh [model_config] [git-version]
 ```

 - `model_config`, e.g. `eval_gpt4_1106_preview`, is the config group name for your
--- a/evaluation/benchmarks/bird/init.py
+++ b/evaluation/benchmarks/bird/init.py
--- a/evaluation/benchmarks/bird/run_infer.py
+++ b/evaluation/benchmarks/bird/run_infer.py
--- a/evaluation/benchmarks/bird/scripts/run_infer.sh
+++ b/evaluation/benchmarks/bird/scripts/run_infer.sh
@@ -26,7 +26,7 @@ echo "AGENT: $AGENT"
 echo "AGENT_VERSION: $AGENT_VERSION"
 echo "MODEL_CONFIG: $MODEL_CONFIG"

-COMMAND="poetry run python evaluation/benchmarks/bird/run_infer.py \
+COMMAND="poetry run python evaluation/bird/run_infer.py \
  --agent-cls $AGENT \
  --llm-config $MODEL_CONFIG \
  --max-iterations 5 \
--- a/evaluation/benchmarks/browsing_delegation/README.md
+++ b/evaluation/benchmarks/browsing_delegation/README.md
@@ -12,7 +12,7 @@ Please follow instruction [here](../README.md#setup) to setup your local develop
 ## Run Inference

 ```bash
-./evaluation/benchmarks/browsing_delegation/scripts/run_infer.sh [model_config] [git-version] [agent] [eval_limit]
+./evaluation/browsing_delegation/scripts/run_infer.sh [model_config] [git-version] [agent] [eval_limit]
 # e.g., ./evaluation/swe_bench/scripts/run_infer.sh llm.eval_gpt4_1106_preview_llm HEAD CodeActAgent 300
 ```

--- a/evaluation/benchmarks/browsing_delegation/run_infer.py
+++ b/evaluation/benchmarks/browsing_delegation/run_infer.py
--- a/evaluation/benchmarks/browsing_delegation/scripts/run_infer.sh
+++ b/evaluation/benchmarks/browsing_delegation/scripts/run_infer.sh
@@ -28,7 +28,7 @@ echo "MODEL_CONFIG: $MODEL_CONFIG"

 EVAL_NOTE="$AGENT_VERSION"

-COMMAND="poetry run python evaluation/benchmarks/browsing_delegation/run_infer.py \
+COMMAND="poetry run python evaluation/browsing_delegation/run_infer.py \
  --agent-cls $AGENT \
  --llm-config $MODEL_CONFIG \
  --max-iterations 1 \
--- a/evaluation/benchmarks/discoverybench/README.md
+++ b/evaluation/benchmarks/discoverybench/README.md
@@ -16,7 +16,7 @@
 2. Execute the bash script to start DiscoveryBench Evaluation

 ```
-./evaluation/benchmarks/discoverybench/scripts/run_infer.sh [YOUR MODEL CONFIG]
+./evaluation/discoverybench/scripts/run_infer.sh [YOUR MODEL CONFIG]
 ```
 Replace `[YOUR MODEL CONFIG]` with any model the model that you have set up in `config.toml`

@@ -27,7 +27,7 @@ When the `run_infer.sh` script is started, it will automatically pull the latest


 ```
-./evaluation/benchmarks/discoverybench/scripts/run_infer.sh [MODEL_CONFIG] [GIT_COMMIT] [AGENT] [EVAL_LIMIT] [NUM_WORKERS]
+./evaluation/discoverybench/scripts/run_infer.sh [MODEL_CONFIG] [GIT_COMMIT] [AGENT] [EVAL_LIMIT] [NUM_WORKERS]
 ```

 - `MODEL_CONFIG`: Name of the model you want to evaluate with
--- a/evaluation/benchmarks/discoverybench/eval_utils/README.md
+++ b/evaluation/benchmarks/discoverybench/eval_utils/README.md
--- a/evaluation/benchmarks/discoverybench/eval_utils/init.py
+++ b/evaluation/benchmarks/discoverybench/eval_utils/init.py
--- a/evaluation/benchmarks/discoverybench/eval_utils/eval_w_subhypo_gen.py
+++ b/evaluation/benchmarks/discoverybench/eval_utils/eval_w_subhypo_gen.py
--- a/evaluation/benchmarks/discoverybench/eval_utils/lm_utils.py
+++ b/evaluation/benchmarks/discoverybench/eval_utils/lm_utils.py
--- a/evaluation/benchmarks/discoverybench/eval_utils/openai_helpers.py
+++ b/evaluation/benchmarks/discoverybench/eval_utils/openai_helpers.py
--- a/evaluation/benchmarks/discoverybench/eval_utils/openai_semantic_gen_prompts.py
+++ b/evaluation/benchmarks/discoverybench/eval_utils/openai_semantic_gen_prompts.py
--- a/evaluation/benchmarks/discoverybench/eval_utils/response_parser.py
+++ b/evaluation/benchmarks/discoverybench/eval_utils/response_parser.py
--- a/evaluation/benchmarks/discoverybench/run_infer.py
+++ b/evaluation/benchmarks/discoverybench/run_infer.py
@@ -5,10 +5,10 @@ import os
 import git
 import pandas as pd

-from evaluation.benchmarks.discoverybench.eval_utils.eval_w_subhypo_gen import (
+from evaluation.discoverybench.eval_utils.eval_w_subhypo_gen import (
    run_eval_gold_vs_gen_NL_hypo_workflow,
 )
-from evaluation.benchmarks.discoverybench.eval_utils.response_parser import (
+from evaluation.discoverybench.eval_utils.response_parser import (
    extract_gen_hypo_from_logs,
 )
 from evaluation.utils.shared import (
--- a/evaluation/benchmarks/discoverybench/scripts/run_infer.sh
+++ b/evaluation/benchmarks/discoverybench/scripts/run_infer.sh
@@ -29,7 +29,7 @@ echo "AGENT: $AGENT"
 echo "AGENT_VERSION: $AGENT_VERSION"
 echo "MODEL_CONFIG: $MODEL_CONFIG"

-COMMAND="poetry run python evaluation/benchmarks/discoverybench/run_infer.py \
+COMMAND="poetry run python evaluation/discoverybench/run_infer.py \
  --agent-cls $AGENT \
  --llm-config $MODEL_CONFIG \
  --max-iterations 10 \
--- a/evaluation/benchmarks/gaia/README.md
+++ b/evaluation/benchmarks/gaia/README.md
@@ -10,11 +10,11 @@ Please follow instruction [here](../README.md#setup) to setup your local develop
 We are using the GAIA dataset hosted on [Hugging Face](https://huggingface.co/datasets/gaia-benchmark/GAIA).
 Please accept the terms and make sure to have logged in on your computer by `huggingface-cli login` before running the evaluation.

-Following is the basic command to start the evaluation. Here we are evaluating on the validation set for the `2023_all` split. You can adjust `./evaluation/benchmarks/gaia/scripts/run_infer.sh` to change the subset you want to evaluate on.
+Following is the basic command to start the evaluation. Here we are evaluating on the validation set for the `2023_all` split. You can adjust `./evaluation/gaia/scripts/run_infer.sh` to change the subset you want to evaluate on.

 ```bash
-./evaluation/benchmarks/gaia/scripts/run_infer.sh [model_config] [git-version] [agent] [eval_limit] [gaia_subset]
-# e.g., ./evaluation/benchmarks/gaia/scripts/run_infer.sh eval_gpt4_1106_preview 0.6.2 CodeActAgent 300
+./evaluation/gaia/scripts/run_infer.sh [model_config] [git-version] [agent] [eval_limit] [gaia_subset]
+# e.g., ./evaluation/gaia/scripts/run_infer.sh eval_gpt4_1106_preview 0.6.2 CodeActAgent 300
 ```

 where `model_config` is mandatory, while `git-version`, `agent`, `eval_limit` and `gaia_subset` are optional.
@@ -35,13 +35,13 @@ to `CodeActAgent`.
 For example,

 ```bash
-./evaluation/benchmarks/gaia/scripts/run_infer.sh eval_gpt4_1106_preview 0.6.2 CodeActAgent 10
+./evaluation/gaia/scripts/run_infer.sh eval_gpt4_1106_preview 0.6.2 CodeActAgent 10
 ```

 ## Get score

 Then you can get stats by running the following command:
 ```bash
-python ./evaluation/benchmarks/gaia/get_score.py \
+python ./evaluation/gaia/get_score.py \
 --file <path_to/output.json>
 ```
--- a/evaluation/benchmarks/gaia/get_score.py
+++ b/evaluation/benchmarks/gaia/get_score.py
--- a/evaluation/benchmarks/gaia/run_infer.py
+++ b/evaluation/benchmarks/gaia/run_infer.py
@@ -7,7 +7,7 @@ import huggingface_hub
 import pandas as pd
 from datasets import load_dataset

-from evaluation.benchmarks.gaia.scorer import question_scorer
+from evaluation.gaia.scorer import question_scorer
 from evaluation.utils.shared import (
    EvalMetadata,
    EvalOutput,
--- a/evaluation/benchmarks/gaia/scorer.py
+++ b/evaluation/benchmarks/gaia/scorer.py
--- a/evaluation/benchmarks/gaia/scripts/run_infer.sh
+++ b/evaluation/benchmarks/gaia/scripts/run_infer.sh
@@ -35,7 +35,7 @@ echo "AGENT_VERSION: $AGENT_VERSION"
 echo "MODEL_CONFIG: $MODEL_CONFIG"
 echo "LEVELS: $LEVELS"

-COMMAND="poetry run python ./evaluation/benchmarks/gaia/run_infer.py \
+COMMAND="poetry run python ./evaluation/gaia/run_infer.py \
  --agent-cls $AGENT \
  --llm-config $MODEL_CONFIG \
  --max-iterations 30 \
--- a/evaluation/benchmarks/gorilla/README.md
+++ b/evaluation/benchmarks/gorilla/README.md
@@ -11,7 +11,7 @@ Please follow instruction [here](../README.md#setup) to setup your local develop
 Make sure your Docker daemon is running, then run this bash script:

 ```bash
-./evaluation/benchmarks/gorilla/scripts/run_infer.sh [model_config] [git-version] [agent] [eval_limit] [hubs]
+./evaluation/gorilla/scripts/run_infer.sh [model_config] [git-version] [agent] [eval_limit] [hubs]
 ```

 where `model_config` is mandatory, while all other arguments are optional.
@@ -35,5 +35,5 @@ Note: in order to use `eval_limit`, you must also set `agent`; in order to use `
 For example,

 ```bash
-./evaluation/benchmarks/gorilla/scripts/run_infer.sh llm 0.6.2 CodeActAgent 10 th
+./evaluation/gorilla/scripts/run_infer.sh llm 0.6.2 CodeActAgent 10 th
 ```
--- a/evaluation/benchmarks/gorilla/ast_eval_hf.py
+++ b/evaluation/benchmarks/gorilla/ast_eval_hf.py
--- a/evaluation/benchmarks/gorilla/ast_eval_tf.py
+++ b/evaluation/benchmarks/gorilla/ast_eval_tf.py
--- a/evaluation/benchmarks/gorilla/ast_eval_th.py
+++ b/evaluation/benchmarks/gorilla/ast_eval_th.py
--- a/evaluation/benchmarks/gorilla/run_infer.py
+++ b/evaluation/benchmarks/gorilla/run_infer.py
@@ -5,7 +5,7 @@ import os
 import pandas as pd
 import requests

-from evaluation.benchmarks.gorilla.utils import encode_question, get_data_for_hub
+from evaluation.gorilla.utils import encode_question, get_data_for_hub
 from evaluation.utils.shared import (
    EvalMetadata,
    EvalOutput,
--- a/evaluation/benchmarks/gorilla/scripts/run_infer.sh
+++ b/evaluation/benchmarks/gorilla/scripts/run_infer.sh
@@ -33,7 +33,7 @@ echo "AGENT_VERSION: $AGENT_VERSION"
 echo "MODEL_CONFIG: $MODEL_CONFIG"
 echo "HUBS: $HUBS"

-COMMAND="poetry run python evaluation/benchmarks/gorilla/run_infer.py \
+COMMAND="poetry run python evaluation/gorilla/run_infer.py \
  --agent-cls $AGENT \
  --llm-config $MODEL_CONFIG \
  --max-iterations 30 \
--- a/evaluation/benchmarks/gorilla/utils.py
+++ b/evaluation/benchmarks/gorilla/utils.py
--- a/evaluation/benchmarks/gpqa/README.md
+++ b/evaluation/benchmarks/gpqa/README.md
@@ -23,7 +23,7 @@ Please follow instruction [here](../README.md#setup) to setup your local develop
 'gpqa_main', 'gqpa_diamond', 'gpqa_experts', 'gpqa_extended' -- data split options
 From the root of the OpenHands repo, run the following command:
 ```bash
-./evaluation/benchmarks/gpqa/scripts/run_infer.sh [model_config_name] [git-version] [num_samples_eval] [data_split] [AgentClass]
+./evaluation/gpqa/scripts/run_infer.sh [model_config_name] [git-version] [num_samples_eval] [data_split] [AgentClass]
 ```
 You can replace `model_config_name` with any model you set up in `config.toml`.

--- a/evaluation/benchmarks/gpqa/init.py
+++ b/evaluation/benchmarks/gpqa/init.py
--- a/evaluation/benchmarks/gpqa/run_infer.py
+++ b/evaluation/benchmarks/gpqa/run_infer.py
--- a/evaluation/benchmarks/gpqa/scripts/run_infer.sh
+++ b/evaluation/benchmarks/gpqa/scripts/run_infer.sh
@@ -33,7 +33,7 @@ echo "AGENT: $AGENT"
 echo "AGENT_VERSION: $AGENT_VERSION"
 echo "MODEL_CONFIG: $MODEL_CONFIG"

-COMMAND="poetry run python evaluation/benchmarks/gpqa/run_infer.py \
+COMMAND="poetry run python evaluation/gpqa/run_infer.py \
  --agent-cls $AGENT \
  --llm-config $MODEL_CONFIG \
  --max-iterations 10 \
--- a/evaluation/benchmarks/humanevalfix/README.md
+++ b/evaluation/benchmarks/humanevalfix/README.md
@@ -9,7 +9,7 @@ Please follow instruction [here](../README.md#setup) to setup your local develop
 ## Run Inference on HumanEvalFix

 ```bash
-./evaluation/benchmarks/humanevalfix/scripts/run_infer.sh eval_gpt4_1106_preview
+./evaluation/humanevalfix/scripts/run_infer.sh eval_gpt4_1106_preview
 ```

 You can replace `eval_gpt4_1106_preview` with any model you set up in `config.toml`.
--- a/evaluation/benchmarks/humanevalfix/init.py
+++ b/evaluation/benchmarks/humanevalfix/init.py
--- a/evaluation/benchmarks/humanevalfix/run_infer.py
+++ b/evaluation/benchmarks/humanevalfix/run_infer.py
--- a/evaluation/benchmarks/humanevalfix/scripts/run_infer.sh
+++ b/evaluation/benchmarks/humanevalfix/scripts/run_infer.sh
@@ -64,7 +64,7 @@ echo "AGENT: $AGENT"
 echo "AGENT_VERSION: $AGENT_VERSION"
 echo "MODEL_CONFIG: $MODEL_CONFIG"

-COMMAND="poetry run python evaluation/benchmarks/humanevalfix/run_infer.py \
+COMMAND="poetry run python evaluation/humanevalfix/run_infer.py \
  --agent-cls $AGENT \
  --llm-config $MODEL_CONFIG \
  --max-iterations 10 \
--- a/evaluation/integration_tests/run_infer.py
+++ b/evaluation/integration_tests/run_infer.py
@@ -48,19 +48,13 @@ def get_config(
            # use default base_container_image
            enable_auto_lint=True,
            use_host_network=False,
-            timeout=300,
-            # Add platform to the sandbox config to solve issue 4401
-            platform='linux/amd64',
+            timeout=100,
            api_key=os.environ.get('ALLHANDS_API_KEY', None),
            remote_runtime_api_url=os.environ.get('SANDBOX_REMOTE_RUNTIME_API_URL'),
-            keep_runtime_alive=False,
-            remote_runtime_init_timeout=3600,
        ),
        # do not mount workspace
        workspace_base=None,
        workspace_mount_path=None,
-        # debug
-        debug=True,
    )
    config.set_llm_config(
        update_llm_config_for_completions_logging(
@@ -113,37 +107,31 @@ def process_instance(
    # =============================================
    # create sandbox and run the agent
    # =============================================
+
    runtime: Runtime = create_runtime(config)
    call_async_from_sync(runtime.connect)
-    try:
-        test_class.initialize_runtime(runtime)

-        # Here's how you can run the agent (similar to the `main` function) and get the final task state
-        state: State | None = asyncio.run(
-            run_controller(
-                config=config,
-                initial_user_action=MessageAction(content=instruction),
-                runtime=runtime,
-                fake_user_response_fn=FAKE_RESPONSES[metadata.agent_class],
-            )
+    test_class.initialize_runtime(runtime)
+
+    # Here's how you can run the agent (similar to the `main` function) and get the final task state
+    state: State | None = asyncio.run(
+        run_controller(
+            config=config,
+            initial_user_action=MessageAction(content=instruction),
+            runtime=runtime,
+            fake_user_response_fn=FAKE_RESPONSES[metadata.agent_class],
        )
-        if state is None:
-            raise ValueError('State should not be None.')
+    )
+    if state is None:
+        raise ValueError('State should not be None.')

-        # # =============================================
-        # # result evaluation
-        # # =============================================
+    # # =============================================
+    # # result evaluation
+    # # =============================================

-        histories = state.history
-
-        # some basic check
-        logger.info(f'Total events in history: {len(histories)}')
-        assert len(histories) > 0, 'History should not be empty'
-
-        test_result: TestResult = test_class.verify_result(runtime, histories)
-        metrics = state.metrics.get() if state.metrics else None
-    finally:
-        runtime.close()
+    histories = [event_to_dict(event) for event in state.history]
+    test_result: TestResult = test_class.verify_result(runtime, histories)
+    metrics = state.metrics.get() if state.metrics else None

    # Save the output
    output = EvalOutput(
@@ -151,7 +139,7 @@ def process_instance(
        instance=instance.to_dict(),
        instruction=instruction,
        metadata=metadata,
-        history=[event_to_dict(event) for event in histories],
+        history=histories,
        metrics=metrics,
        error=state.last_error if state and state.last_error else None,
        test_result=test_result.model_dump(),
--- a/evaluation/integration_tests/tests/t05_simple_browsing.py
+++ b/evaluation/integration_tests/tests/t05_simple_browsing.py
@@ -108,8 +108,6 @@ class Test(BaseIntegrationTest):

    @classmethod
    def verify_result(cls, runtime: Runtime, histories: list[Event]) -> TestResult:
-        from openhands.core.logger import openhands_logger as logger
-
        # check if the "The answer is OpenHands is all you need!" is in any message
        message_actions = [
            event
@@ -118,29 +116,19 @@ class Test(BaseIntegrationTest):
                event, (MessageAction, AgentFinishAction, AgentDelegateObservation)
            )
        ]
-        logger.debug(f'Total message-like events: {len(message_actions)}')
-
        for event in message_actions:
-            try:
-                if isinstance(event, AgentDelegateObservation):
-                    content = event.content
-                elif isinstance(event, AgentFinishAction):
-                    content = event.outputs.get('content', '')
-                elif isinstance(event, MessageAction):
-                    content = event.content
-                else:
-                    logger.warning(f'Unexpected event type: {type(event)}')
-                    continue
+            if isinstance(event, AgentDelegateObservation):
+                content = event.content
+            elif isinstance(event, AgentFinishAction):
+                content = event.outputs.get('content', '')
+            elif isinstance(event, MessageAction):
+                content = event.content
+            else:
+                raise ValueError(f'Unknown event type: {type(event)}')

-                if 'OpenHands is all you need!' in content:
-                    return TestResult(success=True)
-            except Exception as e:
-                logger.error(f'Error processing event: {e}')
-
-        logger.debug(
-            f'Total messages: {len(message_actions)}. Messages: {message_actions}'
-        )
+            if 'OpenHands is all you need!' in content:
+                return TestResult(success=True)
        return TestResult(
            success=False,
-            reason=f'The answer is not found in any message. Total messages: {len(message_actions)}.',
+            reason=f'The answer is not found in any message. Total messages: {len(message_actions)}. Messages: {message_actions}',
        )
--- a/evaluation/integration_tests/tests/t06_github_pr_browsing.py
+++ b/evaluation/integration_tests/tests/t06_github_pr_browsing.py
@@ -14,9 +14,7 @@ class Test(BaseIntegrationTest):

    @classmethod
    def verify_result(cls, runtime: Runtime, histories: list[Event]) -> TestResult:
-        from openhands.core.logger import openhands_logger as logger
-
-        # check if the license information is in any message
+        # check if the "The answer is OpenHands is all you need!" is in any message
        message_actions = [
            event
            for event in histories
@@ -24,35 +22,23 @@ class Test(BaseIntegrationTest):
                event, (MessageAction, AgentFinishAction, AgentDelegateObservation)
            )
        ]
-        logger.info(f'Total message-like events: {len(message_actions)}')
-
        for event in message_actions:
-            try:
-                if isinstance(event, AgentDelegateObservation):
-                    content = event.content
-                elif isinstance(event, AgentFinishAction):
-                    content = event.outputs.get('content', '')
-                    if event.thought:
-                        content += f'\n\n{event.thought}'
-                elif isinstance(event, MessageAction):
-                    content = event.content
-                else:
-                    logger.warning(f'Unexpected event type: {type(event)}')
-                    continue
+            if isinstance(event, AgentDelegateObservation):
+                content = event.content
+            elif isinstance(event, AgentFinishAction):
+                content = event.outputs.get('content', '')
+            elif isinstance(event, MessageAction):
+                content = event.content
+            else:
+                raise ValueError(f'Unknown event type: {type(event)}')

-                if (
-                    'non-commercial' in content
-                    or 'MIT' in content
-                    or 'Apache 2.0' in content
-                ):
-                    return TestResult(success=True)
-            except Exception as e:
-                logger.error(f'Error processing event: {e}')
-
-        logger.debug(
-            f'Total messages: {len(message_actions)}. Messages: {message_actions}'
-        )
+            if (
+                'non-commercial' in content
+                or 'MIT' in content
+                or 'Apache 2.0' in content
+            ):
+                return TestResult(success=True)
        return TestResult(
            success=False,
-            reason=f'The answer is not found in any message. Total messages: {len(message_actions)}.',
+            reason=f'The answer is not found in any message. Total messages: {len(message_actions)}. Messages: {message_actions}',
        )
--- a/evaluation/benchmarks/logic_reasoning/.cache_program/facts.kfb
+++ b/evaluation/benchmarks/logic_reasoning/.cache_program/facts.kfb
--- a/evaluation/benchmarks/logic_reasoning/.cache_program/rules.krb
+++ b/evaluation/benchmarks/logic_reasoning/.cache_program/rules.krb
--- a/evaluation/benchmarks/logic_reasoning/Dockerfile
+++ b/evaluation/benchmarks/logic_reasoning/Dockerfile
--- a/evaluation/benchmarks/logic_reasoning/README.md
+++ b/evaluation/benchmarks/logic_reasoning/README.md
@@ -10,5 +10,5 @@ Please follow instruction [here](../README.md#setup) to setup your local develop
 The following code will run inference on the first example of the ProofWriter dataset,

 ```bash
-./evaluation/benchmarks/logic_reasoning/scripts/run_infer.sh eval_gpt4_1106_preview_llm ProofWriter
+./evaluation/logic_reasoning/scripts/run_infer.sh eval_gpt4_1106_preview_llm ProofWriter
 ```
--- a/evaluation/benchmarks/logic_reasoning/init.py
+++ b/evaluation/benchmarks/logic_reasoning/init.py
--- a/evaluation/benchmarks/logic_reasoning/instruction.txt
+++ b/evaluation/benchmarks/logic_reasoning/instruction.txt
--- a/evaluation/benchmarks/logic_reasoning/logic_inference.py
+++ b/evaluation/benchmarks/logic_reasoning/logic_inference.py
--- a/evaluation/benchmarks/logic_reasoning/run_infer.py
+++ b/evaluation/benchmarks/logic_reasoning/run_infer.py
--- a/evaluation/benchmarks/logic_reasoning/scripts/run_infer.sh
+++ b/evaluation/benchmarks/logic_reasoning/scripts/run_infer.sh
@@ -34,7 +34,7 @@ echo "AGENT: $AGENT"
 echo "AGENT_VERSION: $AGENT_VERSION"
 echo "MODEL_CONFIG: $MODEL_CONFIG"

-COMMAND="poetry run python evaluation/benchmarks/logic_reasoning/run_infer.py \
+COMMAND="poetry run python evaluation/logic_reasoning/run_infer.py \
  --agent-cls $AGENT \
  --llm-config $MODEL_CONFIG \
  --dataset $DATASET \
--- a/evaluation/benchmarks/miniwob/Dockerfile
+++ b/evaluation/benchmarks/miniwob/Dockerfile
--- a/evaluation/benchmarks/miniwob/README.md
+++ b/evaluation/benchmarks/miniwob/README.md
@@ -13,7 +13,7 @@ Access with browser the above MiniWoB URLs and see if they load correctly.
 ## Run Evaluation

 ```sh
-./evaluation/benchmarks/miniwob/scripts/run_infer.sh llm.claude-35-sonnet-eval
+./evaluation/miniwob/scripts/run_infer.sh llm.claude-35-sonnet-eval
 ```

 ### Run Inference on `RemoteRuntime` (experimental)
@@ -21,13 +21,13 @@ Access with browser the above MiniWoB URLs and see if they load correctly.
 This is in limited beta. Contact Xingyao over slack if you want to try this out!

 ```bash
-./evaluation/benchmarks/miniwob/scripts/run_infer.sh [model_config] [git-version] [agent] [note] [eval_limit] [num_workers]
+./evaluation/miniwob/scripts/run_infer.sh [model_config] [git-version] [agent] [note] [eval_limit] [num_workers]

 # Example - This runs evaluation on BrowsingAgent for 125 instances on miniwob, with 2 workers running in parallel
 export ALLHANDS_API_KEY="YOUR-API-KEY"
 export RUNTIME=remote
 export SANDBOX_REMOTE_RUNTIME_API_URL="https://runtime.eval.all-hands.dev"
-./evaluation/benchmarks/miniwob/scripts/run_infer.sh llm.eval HEAD BrowsingAgent "" 125 2
+./evaluation/miniwob/scripts/run_infer.sh llm.eval HEAD BrowsingAgent "" 125 2
 ```

 Results will be in `evaluation/evaluation_outputs/outputs/miniwob/`
@@ -35,7 +35,7 @@ Results will be in `evaluation/evaluation_outputs/outputs/miniwob/`
 To calculate the average reward, run:

 ```sh
-poetry run python evaluation/benchmarks/miniwob/get_success_rate.py evaluation/evaluation_outputs/outputs/miniwob/SOME_AGENT/EXP_NAME/output.jsonl
+poetry run python evaluation/miniwob/get_success_rate.py evaluation/evaluation_outputs/outputs/miniwob/SOME_AGENT/EXP_NAME/output.jsonl
 ```

 ## Submit your evaluation results
--- a/evaluation/benchmarks/miniwob/get_avg_reward.py
+++ b/evaluation/benchmarks/miniwob/get_avg_reward.py
--- a/evaluation/benchmarks/miniwob/run_infer.py
+++ b/evaluation/benchmarks/miniwob/run_infer.py
--- a/evaluation/benchmarks/miniwob/scripts/run_infer.sh
+++ b/evaluation/benchmarks/miniwob/scripts/run_infer.sh
@@ -33,7 +33,7 @@ echo "MODEL_CONFIG: $MODEL_CONFIG"

 EVAL_NOTE="${AGENT_VERSION}_${NOTE}"

-COMMAND="export PYTHONPATH=evaluation/benchmarks/miniwob:\$PYTHONPATH && poetry run python evaluation/benchmarks/miniwob/run_infer.py \
+COMMAND="export PYTHONPATH=evaluation/miniwob:\$PYTHONPATH && poetry run python evaluation/miniwob/run_infer.py \
  --agent-cls $AGENT \
  --llm-config $MODEL_CONFIG \
  --max-iterations 10 \
--- a/evaluation/benchmarks/mint/.gitignore
+++ b/evaluation/benchmarks/mint/.gitignore
--- a/evaluation/benchmarks/mint/Dockerfile
+++ b/evaluation/benchmarks/mint/Dockerfile
--- a/Show More
+++ b/Show More
Author	SHA1	Message	Date
Graham Neubig	b49071ddce	Merge branch 'main' into openhands-fix-issue-5112	2024-11-22 09:11:22 -05:00
openhands	c6ee4089bd	Fix pr #5118 : Fix issue #5112 : [Bug]: "Push to GitHub" shows up even if there's no repo connected	2024-11-19 12:31:03 +00:00
openhands	8f261ed73e	Fix issue #5112 : [Bug]: "Push to GitHub" shows up even if there's no repo connected	2024-11-19 06:03:47 +00:00