style: Fix linting in test_listen.py

fix: Convert ResolverOutput to dict in response
style: Format imports in listen.py
2026-04-29 03:00:45 -04:00 · 2024-11-16 14:24:11 +00:00 · 2024-11-16 14:19:59 +00:00 · 2024-11-16 13:54:26 +00:00 · 2024-11-16 13:51:38 +00:00 · 2024-11-16 13:49:22 +00:00
257 changed files with 13420 additions and 4345 deletions
--- a/.github/workflows/ghcr-build.yml
+++ b/.github/workflows/ghcr-build.yml
@@ -286,7 +286,6 @@ jobs:
          image_name=ghcr.io/${{ github.repository_owner }}/runtime:${{ env.RELEVANT_SHA }}-${{ matrix.base_image }}
          image_name=$(echo $image_name | tr '[:upper:]' '[:lower:]')

-          SKIP_CONTAINER_LOGS=true \
          TEST_RUNTIME=eventstream \
          SANDBOX_USER_ID=$(id -u) \
          SANDBOX_RUNTIME_CONTAINER_IMAGE=$image_name \
@@ -364,7 +363,6 @@ jobs:
          image_name=ghcr.io/${{ github.repository_owner }}/runtime:${{ env.RELEVANT_SHA }}-${{ matrix.base_image }}
          image_name=$(echo $image_name | tr '[:upper:]' '[:lower:]')

-          SKIP_CONTAINER_LOGS=true \
          TEST_RUNTIME=eventstream \
          SANDBOX_USER_ID=$(id -u) \
          SANDBOX_RUNTIME_CONTAINER_IMAGE=$image_name \
--- a/.github/workflows/lint-fix.yml
+++ b/.github/workflows/lint-fix.yml
@@ -0,0 +1,61 @@
+name: Lint Fix
+
+on:
+  pull_request:
+    types: [labeled]
+
+jobs:
+  lint-fix:
+    if: github.event.label.name == 'lint-fix'
+    name: Fix linting issues
+    runs-on: ubuntu-latest
+    permissions:
+      contents: write
+      pull-requests: write
+    steps:
+      - uses: actions/checkout@v4
+        with:
+          ref: ${{ github.head_ref }}
+          repository: ${{ github.event.pull_request.head.repo.full_name }}
+          fetch-depth: 0
+          token: ${{ secrets.GITHUB_TOKEN }}
+
+      # Frontend lint fixes
+      - name: Install Node.js 20
+        uses: actions/setup-node@v4
+        with:
+          node-version: 20
+      - name: Install frontend dependencies
+        run: |
+          cd frontend
+          npm install --frozen-lockfile
+      - name: Fix frontend lint issues
+        run: |
+          cd frontend
+          npm run lint:fix
+
+      # Python lint fixes
+      - name: Set up python
+        uses: actions/setup-python@v5
+        with:
+          python-version: 3.12
+          cache: 'pip'
+      - name: Install pre-commit
+        run: pip install pre-commit==3.7.0
+      - name: Fix python lint issues
+        run: |
+          pre-commit run --files openhands/**/* evaluation/**/* tests/**/* --config ./dev_config/python/.pre-commit-config.yaml
+
+      # Commit and push changes if any
+      - name: Check for changes
+        id: git-check
+        run: |
+          git diff --quiet || echo "changes=true" >> $GITHUB_OUTPUT
+      - name: Commit and push if there are changes
+        if: steps.git-check.outputs.changes == 'true'
+        run: |
+          git config --local user.email "openhands@all-hands.dev"
+          git config --local user.name "OpenHands Bot"
+          git add -A
+          git commit -m "🤖 Auto-fix linting issues"
+          git push
--- a/.github/workflows/openhands-resolver.yml
+++ b/.github/workflows/openhands-resolver.yml
@@ -1,15 +1,269 @@
-name: Resolve Issues with OpenHands
+name: Auto-Fix Tagged Issue with OpenHands

 on:
+  workflow_call:
+    inputs:
+      max_iterations:
+        required: false
+        type: number
+        default: 50
+      macro:
+        required: false
+        type: string
+        default: "@openhands-agent"
+    secrets:
+      LLM_MODEL:
+        required: true
+      LLM_API_KEY:
+        required: true
+      LLM_BASE_URL:
+        required: false
+      PAT_TOKEN:
+        required: true
+      PAT_USERNAME:
+        required: true
+
  issues:
    types: [labeled]
  pull_request:
    types: [labeled]
+  issue_comment:
+    types: [created]
+  pull_request_review_comment:
+    types: [created]
+  pull_request_review:
+    types: [submitted]
+
+permissions:
+  contents: write
+  pull-requests: write
+  issues: write

 jobs:
-  call-openhands-resolver:
-    uses: All-Hands-AI/openhands-resolver/.github/workflows/openhands-resolver.yml@main
-    if: github.event.label.name == 'fix-me'
-    with:
-      max_iterations: 50
-    secrets: inherit
+
+  auto-fix:
+    if: |
+      github.event_name == 'workflow_call' ||
+      github.event.label.name == 'fix-me' ||
+      github.event.label.name == 'fix-me-experimental' ||
+
+      (
+        ((github.event_name == 'issue_comment' || github.event_name == 'pull_request_review_comment') &&
+        startsWith(github.event.comment.body, inputs.macro || '@openhands-agent') &&
+        (github.event.comment.author_association == 'OWNER' || github.event.comment.author_association == 'COLLABORATOR' || github.event.comment.author_association == 'MEMBER')
+        ) ||
+
+        (github.event_name == 'pull_request_review' &&
+        startsWith(github.event.review.body, inputs.macro || '@openhands-agent') &&
+        (github.event.review.author_association == 'OWNER' || github.event.review.author_association == 'COLLABORATOR' || github.event.review.author_association == 'MEMBER')
+        )
+      )
+    runs-on: ubuntu-latest
+    steps:
+      - name: Checkout repository
+        uses: actions/checkout@v4
+
+      - name: Set up Python
+        uses: actions/setup-python@v5
+        with:
+          python-version: "3.12"
+
+      - name: Get latest versions and create requirements.txt
+        run: |
+          python -m pip index versions openhands-ai > openhands_versions.txt
+          OPENHANDS_VERSION=$(head -n 1 openhands_versions.txt | awk '{print $2}' | tr -d '()')
+          echo "openhands-ai==${OPENHANDS_VERSION}" >> requirements.txt
+          cat requirements.txt
+
+      - name: Cache pip dependencies
+        if: github.event.label.name != 'fix-me-experimental'
+        uses: actions/cache@v3
+        with:
+          path: ${{ env.pythonLocation }}/lib/python3.12/site-packages/*
+          key: ${{ runner.os }}-pip-openhands-resolver-${{ hashFiles('requirements.txt') }}
+          restore-keys: |
+            ${{ runner.os }}-pip-openhands-resolver-${{ hashFiles('requirements.txt') }}
+
+      - name: Check required environment variables
+        env:
+          LLM_MODEL: ${{ secrets.LLM_MODEL }}
+          LLM_API_KEY: ${{ secrets.LLM_API_KEY }}
+          LLM_BASE_URL: ${{ secrets.LLM_BASE_URL }}
+          PAT_TOKEN: ${{ secrets.PAT_TOKEN }}
+          PAT_USERNAME: ${{ secrets.PAT_USERNAME }}
+        run: |
+          required_vars=("LLM_MODEL" "LLM_API_KEY" "PAT_TOKEN" "PAT_USERNAME")
+          for var in "${required_vars[@]}"; do
+            if [ -z "${!var}" ]; then
+              echo "Error: Required environment variable $var is not set."
+              exit 1
+            fi
+          done
+
+      - name: Set environment variables
+        run: |
+          if [ -n "${{ github.event.review.body }}" ]; then
+            echo "ISSUE_NUMBER=${{ github.event.pull_request.number }}" >> $GITHUB_ENV
+            echo "ISSUE_TYPE=pr" >> $GITHUB_ENV
+          elif [ -n "${{ github.event.issue.pull_request }}" ]; then
+            echo "ISSUE_NUMBER=${{ github.event.issue.number }}" >> $GITHUB_ENV
+            echo "ISSUE_TYPE=pr" >> $GITHUB_ENV
+          elif [ -n "${{ github.event.pull_request.number }}" ]; then
+            echo "ISSUE_NUMBER=${{ github.event.pull_request.number }}" >> $GITHUB_ENV
+            echo "ISSUE_TYPE=pr" >> $GITHUB_ENV
+          else
+            echo "ISSUE_NUMBER=${{ github.event.issue.number }}" >> $GITHUB_ENV
+            echo "ISSUE_TYPE=issue" >> $GITHUB_ENV
+          fi
+
+          if [ -n "${{ github.event.review.body }}" ]; then
+            echo "COMMENT_ID=${{ github.event.review.id || 'None' }}" >> $GITHUB_ENV
+          else
+            echo "COMMENT_ID=${{ github.event.comment.id || 'None' }}" >> $GITHUB_ENV
+          fi
+
+          echo "MAX_ITERATIONS=${{ inputs.max_iterations || 50 }}" >> $GITHUB_ENV
+          echo "SANDBOX_ENV_GITHUB_TOKEN=${{ secrets.GITHUB_TOKEN }}" >> $GITHUB_ENV
+
+      - name: Comment on issue with start message
+        uses: actions/github-script@v7
+        with:
+          github-token: ${{secrets.GITHUB_TOKEN}}
+          script: |
+            const issueType = process.env.ISSUE_TYPE;
+            github.rest.issues.createComment({
+              issue_number: ${{ env.ISSUE_NUMBER }},
+              owner: context.repo.owner,
+              repo: context.repo.repo,
+              body: `[OpenHands](https://github.com/All-Hands-AI/OpenHands) started fixing the ${issueType}! You can monitor the progress [here](https://github.com/${context.repo.owner}/${context.repo.repo}/actions/runs/${context.runId}).`
+            });
+
+      - name: Install OpenHands
+        run: |
+          if [ "${{ github.event.label.name }}" == "fix-me-experimental" ]; then
+            python -m pip install --upgrade pip
+            pip install git+https://github.com/all-hands-ai/openhands.git
+          else
+            python -m pip install --upgrade -r requirements.txt
+          fi
+
+      - name: Attempt to resolve issue
+        env:
+          GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
+          GITHUB_USERNAME: ${{ secrets.PAT_USERNAME }}
+          LLM_MODEL: ${{ secrets.LLM_MODEL }}
+          LLM_API_KEY: ${{ secrets.LLM_API_KEY }}
+          LLM_BASE_URL: ${{ secrets.LLM_BASE_URL }}
+          PYTHONPATH: ""
+        run: |
+          cd /tmp && python -m openhands.resolver.resolve_issue \
+            --repo ${{ github.repository }} \
+            --issue-number ${{ env.ISSUE_NUMBER }} \
+            --issue-type ${{ env.ISSUE_TYPE }} \
+            --max-iterations ${{ env.MAX_ITERATIONS }} \
+            --comment-id ${{ env.COMMENT_ID }}
+
+      - name: Check resolution result
+        id: check_result
+        run: |
+          if cd /tmp && grep -q '"success":true' output/output.jsonl; then
+            echo "RESOLUTION_SUCCESS=true" >> $GITHUB_OUTPUT
+          else
+            echo "RESOLUTION_SUCCESS=false" >> $GITHUB_OUTPUT
+          fi
+
+      - name: Upload output.jsonl as artifact
+        uses: actions/upload-artifact@v4
+        if: always() # Upload even if the previous steps fail
+        with:
+          name: resolver-output
+          path: /tmp/output/output.jsonl
+          retention-days: 30 # Keep the artifact for 30 days
+
+      - name: Create draft PR or push branch
+        if: always() # Create PR or branch even if the previous steps fail
+        env:
+          GITHUB_TOKEN: ${{ secrets.PAT_TOKEN }}
+          GITHUB_USERNAME: ${{ secrets.PAT_USERNAME }}
+          LLM_MODEL: ${{ secrets.LLM_MODEL }}
+          LLM_API_KEY: ${{ secrets.LLM_API_KEY }}
+          LLM_BASE_URL: ${{ secrets.LLM_BASE_URL }}
+          PYTHONPATH: ""
+        run: |
+          if [ "${{ steps.check_result.outputs.RESOLUTION_SUCCESS }}" == "true" ]; then
+            cd /tmp && python -m openhands.resolver.send_pull_request \
+              --issue-number ${{ env.ISSUE_NUMBER }} \
+              --pr-type draft | tee pr_result.txt && \
+              grep "draft created" pr_result.txt | sed 's/.*\///g' > pr_number.txt
+          else
+            cd /tmp && python -m openhands.resolver.send_pull_request \
+              --issue-number ${{ env.ISSUE_NUMBER }} \
+              --pr-type branch \
+              --send-on-failure | tee branch_result.txt && \
+              grep "branch created" branch_result.txt | sed 's/.*\///g; s/.expand=1//g' > branch_name.txt
+          fi
+
+      - name: Comment on issue
+        uses: actions/github-script@v7
+        if: always() # Comment on issue even if the previous steps fail
+        with:
+          github-token: ${{secrets.GITHUB_TOKEN}}
+          script: |
+            const fs = require('fs');
+            const issueNumber = ${{ env.ISSUE_NUMBER }};
+            const success = ${{ steps.check_result.outputs.RESOLUTION_SUCCESS }};
+
+            let prNumber = '';
+            let branchName = '';
+            let logContent = '';
+            const noChangesMessage = `No changes to commit for issue #${issueNumber}. Skipping commit.`;
+
+            try {
+              if (success){
+                logContent = fs.readFileSync('/tmp/pr_result.txt', 'utf8').trim();
+              } else {
+                logContent = fs.readFileSync('/tmp/branch_result.txt', 'utf8').trim();
+              }
+            } catch (error) {
+              console.error('Error reading results file:', error);
+            }
+
+            try {
+              if (success) {
+                prNumber = fs.readFileSync('/tmp/pr_number.txt', 'utf8').trim();
+              } else {
+                branchName = fs.readFileSync('/tmp/branch_name.txt', 'utf8').trim();
+              }
+            } catch (error) {
+              console.error('Error reading file:', error);
+            }
+
+            if (logContent.includes(noChangesMessage)) {
+              github.rest.issues.createComment({
+                issue_number: issueNumber,
+                owner: context.repo.owner,
+                repo: context.repo.repo,
+                body: `The workflow to fix this issue encountered an error. Openhands failed to create any code changes.`
+              });
+            } else if (success && prNumber) {
+              github.rest.issues.createComment({
+                issue_number: issueNumber,
+                owner: context.repo.owner,
+                repo: context.repo.repo,
+                body: `A potential fix has been generated and a draft PR #${prNumber} has been created. Please review the changes.`
+              });
+            } else if (!success && branchName) {
+              github.rest.issues.createComment({
+                issue_number: issueNumber,
+                owner: context.repo.owner,
+                repo: context.repo.repo,
+                body: `An attempt was made to automatically fix this issue, but it was unsuccessful. A branch named '${branchName}' has been created with the attempted changes. You can view the branch [here](https://github.com/${context.repo.owner}/${context.repo.repo}/tree/${branchName}). Manual intervention may be required.`
+              });
+            } else {
+              github.rest.issues.createComment({
+                issue_number: issueNumber,
+                owner: context.repo.owner,
+                repo: context.repo.repo,
+                body: `The workflow to fix this issue encountered an error. Please check the [workflow logs](https://github.com/${context.repo.owner}/${context.repo.repo}/actions/runs/${context.runId}) for more information.`
+              });
+            }
--- a/.gitignore
+++ b/.gitignore
@@ -176,6 +176,9 @@ evaluation/gorilla/data
 evaluation/toolqa/data
 evaluation/scienceagentbench/benchmark

+# openhands resolver
+output/
+
 # frontend

 # dependencies
--- a/CONTRIBUTING.md
+++ b/CONTRIBUTING.md
@@ -92,3 +92,32 @@ You may also check out previous PRs in the [PR list](https://github.com/All-Hand

 If your changes are user-facing (e.g. a new feature in the UI, a change in behavior, or a bugfix)
 please include a short message that we can add to our changelog.
+
+## How to Make Effective Contributions
+
+### Opening Issues
+
+If you notice any bugs or have any feature requests please open them via the [issues page](https://github.com/All-Hands-AI/OpenHands/issues). We will triage based on how critical the bug is or how potentially useful the improvement is, discuss, and implement the ones that the community has interest/effort for.
+
+Further, if you see an issue you like, please leave a "thumbs-up" or a comment, which will help us prioritize.
+
+### Making Pull Requests
+
+We're generally happy to consider all PRs, with the evaluation process varying based on the type of change:
+
+#### For Small Improvements
+
+Small improvements with few downsides are typically reviewed and approved quickly.
+One thing to check when making changes is to ensure that all continuous integration tests pass, which you can check before getting a review.
+
+#### For Core Agent Changes
+
+We need to be more careful with changes to the core agent, as it is imperative to maintain high quality. These PRs are evaluated based on three key metrics:
+
+1. **Accuracy**
+2. **Efficiency**
+3. **Code Complexity**
+
+If it improves accuracy, efficiency, or both with only a minimal change to code quality, that's great we're happy to merge it in!
+If there are bigger tradeoffs (e.g. helping efficiency a lot and hurting accuracy a little) we might want to put it behind a feature flag.
+Either way, please feel free to discuss on github issues or slack, and we will give guidance and preliminary feedback.
--- a/Development.md
+++ b/Development.md
@@ -38,7 +38,9 @@ make build
 ```

 ### 3. Configuring the Language Model
-OpenHands supports a diverse array of Language Models (LMs) through the powerful [litellm](https://docs.litellm.ai) library. By default, we've chosen the mighty GPT-4 from OpenAI as our go-to model, but the world is your oyster! You can unleash the potential of Anthropic's suave Claude, the enigmatic Llama, or any other LM that piques your interest.
+OpenHands supports a diverse array of Language Models (LMs) through the powerful [litellm](https://docs.litellm.ai) library.
+By default, we've chosen Claude Sonnet 3.5 as our go-to model, but the world is your oyster! You can unleash the
+potential of any other LM that piques your interest.

 To configure the LM of your choice, run:

@@ -52,10 +54,7 @@ To configure the LM of your choice, run:
   Environment variables > config.toml variables > default variables

 **Note on Alternative Models:**
-Some alternative models may prove more challenging to tame than others. Fear not, brave adventurer! We shall soon unveil LLM-specific documentation to guide you on your quest.
-And if you've already mastered the art of wielding a model other than OpenAI's GPT, we encourage you to share your setup instructions with us by creating instructions and adding it [to our documentation](https://github.com/All-Hands-AI/OpenHands/tree/main/docs/modules/usage/llms).
-
-For a full list of the LM providers and models available, please consult the [litellm documentation](https://docs.litellm.ai/docs/providers).
+See [our documentation](https://docs.all-hands.dev/modules/usage/llms) for recommended models.

 ### 4. Running the application
 #### Option A: Run the Full Application
@@ -98,9 +97,10 @@ poetry run pytest ./tests/unit/test_*.py
 2. Update the poetry.lock file via `poetry lock --no-update`

 ### 9. Use existing Docker image
-To reduce build time (e.g., if no changes were made to the client-runtime component), you can use an existing Docker container image. Follow these steps:
-1. Set the SANDBOX_RUNTIME_CONTAINER_IMAGE environment variable to the desired Docker image.
-2. Example: export SANDBOX_RUNTIME_CONTAINER_IMAGE=ghcr.io/all-hands-ai/runtime:0.13-nikolaik
+To reduce build time (e.g., if no changes were made to the client-runtime component), you can use an existing Docker container image by
+setting the SANDBOX_RUNTIME_CONTAINER_IMAGE environment variable to the desired Docker image.
+
+Example: `export SANDBOX_RUNTIME_CONTAINER_IMAGE=ghcr.io/all-hands-ai/runtime:0.14-nikolaik`

 ## Develop inside Docker container

--- a/ISSUE_TRIAGE.md
+++ b/ISSUE_TRIAGE.md
@@ -6,9 +6,9 @@ These are the procedures and guidelines on how issues are triaged in this repo b
 * Issues may be tagged with what it relates to (**backend**, **frontend**, **agent quality**, etc.)

 ## Severity
-* **Low**: Minor issues, single user report
-* **Medium**: Affecting multiple users
-* **Critical**: Affecting all users or potential security issues
+* **Low**: Minor issues or affecting single user.
+* **Medium**: Affecting multiple users.
+* **Critical**: Affecting all users or potential security issues.

 ## Effort
 * Issues may be estimated with effort required (**small effort**, **medium effort**, **large effort**)
@@ -17,9 +17,9 @@ These are the procedures and guidelines on how issues are triaged in this repo b
 * Issues with low implementation difficulty may be tagged with **good first issue**

 ## Not Enough Information
-* User is asked to provide more information (logs, how to reproduce, etc.) when the issue is not clear
-* If an issue is unclear and the author does not provide more information or respond to a request, the issue may be closed as **not planned** (Usually after a week)
+* User is asked to provide more information (logs, how to reproduce, etc.) when the issue is not clear.
+* If an issue is unclear and the author does not provide more information or respond to a request, the issue may be closed as **not planned** (Usually after a week).

 ## Multiple Requests/Fixes in One Issue
-* These issues will be narrowed down to one request/fix so the issue is more easily tracked and fixed
-* Issues may be broken down into multiple issues if required
+* These issues will be narrowed down to one request/fix so the issue is more easily tracked and fixed.
+* Issues may be broken down into multiple issues if required.
--- a/README.md
+++ b/README.md
@@ -38,15 +38,16 @@ See the [Installation](https://docs.all-hands.dev/modules/usage/installation) gu
 system requirements and more information.

 ```bash
-docker pull docker.all-hands.dev/all-hands-ai/runtime:0.13-nikolaik
+docker pull docker.all-hands.dev/all-hands-ai/runtime:0.14-nikolaik

 docker run -it --pull=always \
-    -e SANDBOX_RUNTIME_CONTAINER_IMAGE=docker.all-hands.dev/all-hands-ai/runtime:0.13-nikolaik \
+    -e SANDBOX_RUNTIME_CONTAINER_IMAGE=docker.all-hands.dev/all-hands-ai/runtime:0.14-nikolaik \
+    -e LOG_ALL_EVENTS=true \
    -v /var/run/docker.sock:/var/run/docker.sock \
    -p 3000:3000 \
    --add-host host.docker.internal:host-gateway \
    --name openhands-app \
-    docker.all-hands.dev/all-hands-ai/openhands:0.13
+    docker.all-hands.dev/all-hands-ai/openhands:0.14
 ```

 You'll find OpenHands running at [http://localhost:3000](http://localhost:3000)!
@@ -60,7 +61,7 @@ works best, but you have [many options](https://docs.all-hands.dev/modules/usage
 You can also [connect OpenHands to your local filesystem](https://docs.all-hands.dev/modules/usage/runtimes),
 run OpenHands in a scriptable [headless mode](https://docs.all-hands.dev/modules/usage/how-to/headless-mode),
 interact with it via a [friendly CLI](https://docs.all-hands.dev/modules/usage/how-to/cli-mode),
-or run it on tagged issues with [a github action](https://github.com/All-Hands-AI/OpenHands-resolver).
+or run it on tagged issues with [a github action](https://github.com/All-Hands-AI/OpenHands/blob/main/openhands/resolver/README.md).

 Visit [Installation](https://docs.all-hands.dev/modules/usage/installation) for more information and setup instructions.

--- a/compose.yml
+++ b/compose.yml
@@ -7,7 +7,7 @@ services:
    image: openhands:latest
    container_name: openhands-app-${DATE:-}
    environment:
-      - SANDBOX_RUNTIME_CONTAINER_IMAGE=${SANDBOX_RUNTIME_CONTAINER_IMAGE:-ghcr.io/all-hands-ai/runtime:0.13-nikolaik}
+      - SANDBOX_RUNTIME_CONTAINER_IMAGE=${SANDBOX_RUNTIME_CONTAINER_IMAGE:-ghcr.io/all-hands-ai/runtime:0.14-nikolaik}
      - SANDBOX_USER_ID=${SANDBOX_USER_ID:-1234}
      - WORKSPACE_MOUNT_PATH=${WORKSPACE_BASE:-$PWD/workspace}
    ports:
--- a/containers/dev/compose.yml
+++ b/containers/dev/compose.yml
@@ -11,7 +11,7 @@ services:
      - BACKEND_HOST=${BACKEND_HOST:-"0.0.0.0"}
      - SANDBOX_API_HOSTNAME=host.docker.internal
      #
-      - SANDBOX_RUNTIME_CONTAINER_IMAGE=${SANDBOX_RUNTIME_CONTAINER_IMAGE:-ghcr.io/all-hands-ai/runtime:0.13-nikolaik}
+      - SANDBOX_RUNTIME_CONTAINER_IMAGE=${SANDBOX_RUNTIME_CONTAINER_IMAGE:-ghcr.io/all-hands-ai/runtime:0.14-nikolaik}
      - SANDBOX_USER_ID=${SANDBOX_USER_ID:-1234}
      - WORKSPACE_MOUNT_PATH=${WORKSPACE_BASE:-$PWD/workspace}
    ports:
--- a/docs/modules/usage/about.md
+++ b/docs/modules/usage/about.md
@@ -1,6 +1,6 @@
-# 📚 Misc
+# About OpenHands

-## ⭐️ Research Strategy
+## Research Strategy

 Achieving full replication of production-grade applications with LLMs is a complex endeavor. Our strategy involves:

@@ -9,34 +9,11 @@ Achieving full replication of production-grade applications with LLMs is a compl
 3. **Task Planning:** Developing capabilities for bug detection, codebase management, and optimization
 4. **Evaluation:** Establishing comprehensive evaluation metrics to better understand and improve our models

-## 🚧 Default Agent
+## Default Agent

 Our default Agent is currently the [CodeActAgent](agents), which is capable of generating code and handling files.

-## 🤝 How to Contribute
-
-OpenHands is a community-driven project, and we welcome contributions from everyone. Whether you're a developer, a researcher, or simply enthusiastic about advancing the field of software engineering with AI, there are many ways to get involved:
-
- **Code Contributions:** Help us develop the core functionalities, frontend interface, or sandboxing solutions
- **Research and Evaluation:** Contribute to our understanding of LLMs in software engineering, participate in evaluating the models, or suggest improvements
- **Feedback and Testing:** Use the OpenHands toolset, report bugs, suggest features, or provide feedback on usability
-
-For details, please check [this document](https://github.com/All-Hands-AI/OpenHands/blob/main/CONTRIBUTING.md).
-
-## 🤖 Join Our Community
-
-We have both Slack workspace for the collaboration on building OpenHands and Discord server for discussion about anything related, e.g., this project, LLM, agent, etc.
-
- [Slack workspace](https://join.slack.com/t/opendevin/shared_invite/zt-2oikve2hu-UDxHeo8nsE69y6T7yFX_BA)
- [Discord server](https://discord.gg/ESHStjSjD4)
-
-If you would love to contribute, feel free to join our community. Let's simplify software engineering together!
-
-🐚 **Code less, make more with OpenHands.**
-
-[![Star History Chart](https://api.star-history.com/svg?repos=All-Hands-AI/OpenHands&type=Date)](https://star-history.com/#All-Hands-AI/OpenHands&Date)
-
-## 🛠️ Built With
+## Built With

 OpenHands is built using a combination of powerful frameworks and libraries, providing a robust foundation for its development. Here are the key technologies used in the project:

@@ -44,6 +21,9 @@ OpenHands is built using a combination of powerful frameworks and libraries, pro

 Please note that the selection of these technologies is in progress, and additional technologies may be added or existing ones may be removed as the project evolves. We strive to adopt the most suitable and efficient tools to enhance the capabilities of OpenHands.

-## 📜 License
+## Licensing, Contributing, Community Servers

-Distributed under the MIT License. See [our license](https://github.com/All-Hands-AI/OpenHands/blob/main/LICENSE) for more information.
+Distributed under MIT [License](https://github.com/All-Hands-AI/OpenHands/blob/main/LICENSE).
+
+For guides on how to contribute to OpenHands, joining our Discord and Slack servers
+[check out the OpenHands README.](https://github.com/All-Hands-AI/OpenHands?tab=readme-ov-file#-how-to-contribute)
--- a/docs/modules/usage/how-to/cli-mode.md
+++ b/docs/modules/usage/how-to/cli-mode.md
@@ -50,7 +50,7 @@ LLM_API_KEY="sk_test_12345"
 ```bash
 docker run -it \
    --pull=always \
-    -e SANDBOX_RUNTIME_CONTAINER_IMAGE=docker.all-hands.dev/all-hands-ai/runtime:0.13-nikolaik \
+    -e SANDBOX_RUNTIME_CONTAINER_IMAGE=docker.all-hands.dev/all-hands-ai/runtime:0.14-nikolaik \
    -e SANDBOX_USER_ID=$(id -u) \
    -e WORKSPACE_MOUNT_PATH=$WORKSPACE_BASE \
    -e LLM_API_KEY=$LLM_API_KEY \
@@ -59,7 +59,7 @@ docker run -it \
    -v /var/run/docker.sock:/var/run/docker.sock \
    --add-host host.docker.internal:host-gateway \
    --name openhands-app-$(date +%Y%m%d%H%M%S) \
-    docker.all-hands.dev/all-hands-ai/openhands:0.13 \
+    docker.all-hands.dev/all-hands-ai/openhands:0.14 \
    python -m openhands.core.cli
 ```

--- a/docs/modules/usage/how-to/custom-sandbox-guide.md
+++ b/docs/modules/usage/how-to/custom-sandbox-guide.md
@@ -62,25 +62,3 @@ Run OpenHands by running ```make run``` in the top level directory.
 ## Technical Explanation

 Please refer to [custom docker image section of the runtime documentation](https://docs.all-hands.dev/modules/usage/architecture/runtime#advanced-how-openhands-builds-and-maintains-od-runtime-images) for more details.
-
-## Troubleshooting / Errors
-
-### Error: ```useradd: UID 1000 is not unique```
-
-If you see this error in the console output it is because OpenHands is trying to create the openhands user in the sandbox with a UID of 1000, however this UID is already being used in the image (for some reason). To fix this change the sandbox_user_id field in the config.toml file to a different value:
-
-```toml
-[core]
-workspace_base="./workspace"
-run_as_openhands=true
-sandbox_base_container_image="custom_image"
-sandbox_user_id="1001"
-```
-
-### Port use errors
-
-If you see an error about a port being in use or unavailable, try deleting all running Docker Containers (run `docker ps` and `docker rm` relevant containers) and then re-running ```make run``` .
-
-## Discuss
-
-For other issues or questions join the [Slack](https://join.slack.com/t/opendevin/shared_invite/zt-2oikve2hu-UDxHeo8nsE69y6T7yFX_BA) or [Discord](https://discord.gg/ESHStjSjD4) and ask!
--- a/docs/modules/usage/how-to/github-action.md
+++ b/docs/modules/usage/how-to/github-action.md
@@ -12,4 +12,5 @@ To use the OpenHands GitHub Action in the OpenHands repository, an OpenHands mai

 ## Installing the Action in a New Repository

-To install the OpenHands GitHub Action in your own repository, follow the [directions in the OpenHands Resolver repo](https://github.com/All-Hands-AI/OpenHands-resolver?tab=readme-ov-file#using-the-github-actions-workflow).
+To install the OpenHands GitHub Action in your own repository, follow
+the [README for the OpenHands Resolver](https://github.com/All-Hands-AI/OpenHands/blob/main/openhands/resolver/README.md).
--- a/docs/modules/usage/how-to/headless-mode.md
+++ b/docs/modules/usage/how-to/headless-mode.md
@@ -44,15 +44,16 @@ LLM_API_KEY="sk_test_12345"
 ```bash
 docker run -it \
    --pull=always \
-    -e SANDBOX_RUNTIME_CONTAINER_IMAGE=docker.all-hands.dev/all-hands-ai/runtime:0.13-nikolaik \
+    -e SANDBOX_RUNTIME_CONTAINER_IMAGE=docker.all-hands.dev/all-hands-ai/runtime:0.14-nikolaik \
    -e SANDBOX_USER_ID=$(id -u) \
    -e WORKSPACE_MOUNT_PATH=$WORKSPACE_BASE \
    -e LLM_API_KEY=$LLM_API_KEY \
    -e LLM_MODEL=$LLM_MODEL \
+    -e LOG_ALL_EVENTS=true \
    -v $WORKSPACE_BASE:/opt/workspace_base \
    -v /var/run/docker.sock:/var/run/docker.sock \
    --add-host host.docker.internal:host-gateway \
    --name openhands-app-$(date +%Y%m%d%H%M%S) \
-    docker.all-hands.dev/all-hands-ai/openhands:0.13 \
+    docker.all-hands.dev/all-hands-ai/openhands:0.14 \
    python -m openhands.core.main -t "write a bash script that prints hi"
 ```
--- a/docs/modules/usage/installation.mdx
+++ b/docs/modules/usage/installation.mdx
@@ -11,15 +11,16 @@
 The easiest way to run OpenHands is in Docker.

 ```bash
-docker pull docker.all-hands.dev/all-hands-ai/runtime:0.13-nikolaik
+docker pull docker.all-hands.dev/all-hands-ai/runtime:0.14-nikolaik

 docker run -it --rm --pull=always \
-    -e SANDBOX_RUNTIME_CONTAINER_IMAGE=docker.all-hands.dev/all-hands-ai/runtime:0.13-nikolaik \
+    -e SANDBOX_RUNTIME_CONTAINER_IMAGE=docker.all-hands.dev/all-hands-ai/runtime:0.14-nikolaik \
+    -e LOG_ALL_EVENTS=true \
    -v /var/run/docker.sock:/var/run/docker.sock \
    -p 3000:3000 \
    --add-host host.docker.internal:host-gateway \
    --name openhands-app \
-    docker.all-hands.dev/all-hands-ai/openhands:0.13
+    docker.all-hands.dev/all-hands-ai/openhands:0.14
 ```

 You can also run OpenHands in a scriptable [headless mode](https://docs.all-hands.dev/modules/usage/how-to/headless-mode), as an [interactive CLI](https://docs.all-hands.dev/modules/usage/how-to/cli-mode), or using the [OpenHands GitHub Action](https://docs.all-hands.dev/modules/usage/how-to/github-action).
--- a/docs/modules/usage/llms/litellm-proxy.md
+++ b/docs/modules/usage/llms/litellm-proxy.md
@@ -0,0 +1,20 @@
+# LiteLLM Proxy
+
+OpenHands supports using the [LiteLLM proxy](https://docs.litellm.ai/docs/proxy/quick_start) to access various LLM providers.
+
+## Configuration
+
+To use LiteLLM proxy with OpenHands, you need to:
+
+1. Set up a LiteLLM proxy server (see [LiteLLM documentation](https://docs.litellm.ai/docs/proxy/quick_start))
+2. When running OpenHands, you'll need to set the following in the OpenHands UI through the Settings:
+  * Enable `Advanced Options`
+  * `Custom Model` to the prefix `litellm_proxy/` + the model you will be using (e.g. `litellm_proxy/anthropic.claude-3-5-sonnet-20241022-v2:0`)
+  * `Base URL` to your LiteLLM proxy URL (e.g. `https://your-litellm-proxy.com`)
+  * `API Key` to your LiteLLM proxy API key
+
+## Supported Models
+
+The supported models depend on your LiteLLM proxy configuration. OpenHands supports any model that your LiteLLM proxy is configured to handle.
+
+Refer to your LiteLLM proxy configuration for the list of available models and their names.
--- a/docs/modules/usage/llms/llms.md
+++ b/docs/modules/usage/llms/llms.md
@@ -4,11 +4,11 @@ OpenHands can connect to any LLM supported by LiteLLM. However, it requires a po

 ## Model Recommendations

-Based on a recent evaluation of language models for coding tasks (using the SWE-bench dataset), we can provide some recommendations for model selection. The full analysis can be found in [this blog article](https://www.all-hands.dev/blog/evaluation-of-llms-as-coding-agents-on-swe-bench-at-30x-speed).
+Based on our evaluations of language models for coding tasks (using the SWE-bench dataset), we can provide some recommendations for model selection. Some analyses can be found in [this blog article comparing LLMs](https://www.all-hands.dev/blog/evaluation-of-llms-as-coding-agents-on-swe-bench-at-30x-speed) and [this blog article with some more recent results](https://www.all-hands.dev/blog/openhands-codeact-21-an-open-state-of-the-art-software-development-agent).

 When choosing a model, consider both the quality of outputs and the associated costs. Here's a summary of the findings:

- Claude 3.5 Sonnet is the best by a fair amount, achieving a 27% resolve rate with the default agent in OpenHands.
+- Claude 3.5 Sonnet is the best by a fair amount, achieving a 53% resolve rate on SWE-Bench Verified with the default agent in OpenHands.
 - GPT-4o lags behind, and o1-mini actually performed somewhat worse than GPT-4o. We went in and analyzed the results a little, and briefly it seemed like o1 was sometimes "overthinking" things, performing extra environment configuration tasks when it could just go ahead and finish the task.
 - Finally, the strongest open models were Llama 3.1 405 B and deepseek-v2.5, and they performed reasonably, even besting some of the closed models.

@@ -63,6 +63,7 @@ We have a few guides for running OpenHands with specific model providers:
 - [Azure](llms/azure-llms)
 - [Google](llms/google-llms)
 - [Groq](llms/groq)
+- [LiteLLM Proxy](llms/litellm-proxy)
 - [OpenAI](llms/openai-llms)
 - [OpenRouter](llms/openrouter)

--- a/docs/modules/usage/runtimes.md
+++ b/docs/modules/usage/runtimes.md
@@ -49,7 +49,7 @@ but seems to work well on most systems.

 ## All Hands Runtime
 The All Hands Runtime is currently in beta. You can request access by joining
-the #remote-runtime-limited-beta channel on Slack (see the README for an invite).
+the #remote-runtime-limited-beta channel on Slack ([see the README](https://github.com/All-Hands-AI/OpenHands?tab=readme-ov-file#-join-our-community) for an invite).

 To use the All Hands Runtime, set the following environment variables when
 starting OpenHands:
@@ -59,14 +59,14 @@ docker run # ...
    -e RUNTIME=remote \
    -e SANDBOX_REMOTE_RUNTIME_API_URL="https://runtime.app.all-hands.dev" \
    -e SANDBOX_API_KEY="your-all-hands-api-key" \
-    -e SANDBOX_KEEP_REMOTE_RUNTIME_ALIVE="true" \
+    -e SANDBOX_KEEP_RUNTIME_ALIVE="true" \
    # ...
 ```

 ## Modal Runtime
 Our partners at [Modal](https://modal.com/) have also provided a runtime for OpenHands.

-To use the Modal Runtime, create an account, and then [create an API key](https://modal.com/settings)
+To use the Modal Runtime, create an account, and then [create an API key.](https://modal.com/settings)

 You'll then need to set the following environment variables when starting OpenHands:
 ```bash
--- a/docs/sidebars.ts
+++ b/docs/sidebars.ts
@@ -76,6 +76,11 @@ const sidebars: SidebarsConfig = {
                  label: 'Groq',
                  id: 'usage/llms/groq',
                },
+                {
+                  type: 'doc',
+                  label: 'LiteLLM Proxy',
+                  id: 'usage/llms/litellm-proxy',
+                },
                {
                  type: 'doc',
                  label: 'OpenAI',
--- a/evaluation/EDA/game.py
+++ b/evaluation/EDA/game.py
@@ -87,9 +87,7 @@ class Q20Game:
        # others
        bingo, anwser_reply = self.judge_winner(response)
        if bingo:
-            return (
-                'You are bingo! quit now, run: <execute_bash> exit </execute_bash>.\n'
-            )
+            return 'You are bingo! Use the "finish" tool to finish the interaction.\n'
        if self.curr_turn == self.num_turns - 2:
            anwser_reply += " You must guess now, what's it?"
        return anwser_reply
--- a/evaluation/aider_bench/README.md
+++ b/evaluation/aider_bench/README.md
@@ -56,6 +56,20 @@ You can update the arguments in the script
 ./evaluation/aider_bench/scripts/run_infer.sh eval_gpt35_turbo HEAD CodeActAgent 100 1 "1,3,10"
 ```

+### Run Inference on `RemoteRuntime` (experimental)
+
+This is in limited beta. Contact Xingyao over slack if you want to try this out!
+
+```bash
+./evaluation/aider_bench/scripts/run_infer.sh [model_config] [git-version] [agent] [eval_limit] [eval-num-workers] [eval_ids]
+
+# Example - This runs evaluation on CodeActAgent for 133 instances on aider_bench test set, with 2 workers running in parallel
+export ALLHANDS_API_KEY="YOUR-API-KEY"
+export RUNTIME=remote
+export SANDBOX_REMOTE_RUNTIME_API_URL="https://runtime.eval.all-hands.dev"
+./evaluation/aider_bench/scripts/run_infer.sh llm.eval HEAD CodeActAgent 133 2
+```
+
 ## Summarize Results

 ```bash
--- a/evaluation/aider_bench/run_infer.py
+++ b/evaluation/aider_bench/run_infer.py
@@ -58,6 +58,9 @@ def get_config(
            use_host_network=False,
            timeout=100,
            api_key=os.environ.get('ALLHANDS_API_KEY', None),
+            remote_runtime_api_url=os.environ.get('SANDBOX_REMOTE_RUNTIME_API_URL'),
+            keep_runtime_alive=False,
+            remote_runtime_init_timeout=1800,
        ),
        # do not mount workspace
        workspace_base=None,
--- a/evaluation/biocoder/run_infer.py
+++ b/evaluation/biocoder/run_infer.py
@@ -40,7 +40,7 @@ AGENT_CLS_TO_FAKE_USER_RESPONSE_FN = {
 }

 AGENT_CLS_TO_INST_SUFFIX = {
-    'CodeActAgent': 'When you think you have fixed the issue through code changes, please run the following command: <execute_bash> exit </execute_bash>.\n'
+    'CodeActAgent': 'When you think you have fixed the issue through code changes, please finish the interaction using the "finish" tool.\n'
 }

 FILE_EXT_MAP = {
--- a/evaluation/bird/README.md
+++ b/evaluation/bird/README.md
--- a/evaluation/bird/run_infer.py
+++ b/evaluation/bird/run_infer.py
@@ -40,7 +40,7 @@ from openhands.utils.async_utils import call_async_from_sync
 def codeact_user_response(state: State) -> str:
    msg = (
        'Please continue working on the task on whatever approach you think is suitable.\n'
-        'If you think you have completed the SQL, please run the following command: <execute_bash> exit </execute_bash>.\n'
+        'If you think you have completed the SQL, please finish the interaction using the "finish" tool.\n'
        'IMPORTANT: YOU SHOULD NEVER ASK FOR HUMAN HELP OR USE THE INTERNET TO SOLVE THIS TASK.\n'
    )
    if state.history:
@@ -54,7 +54,7 @@ def codeact_user_response(state: State) -> str:
            # let the agent know that it can give up when it has tried 3 times
            return (
                msg
-                + 'If you want to give up, run: <execute_bash> exit </execute_bash>.\n'
+                + 'If you want to give up, use the "finish" tool to finish the interaction.\n'
            )
    return msg

@@ -64,7 +64,7 @@ AGENT_CLS_TO_FAKE_USER_RESPONSE_FN = {
 }

 AGENT_CLS_TO_INST_SUFFIX = {
-    'CodeActAgent': 'When you think you have fixed the issue through code changes, please run the following command: <execute_bash> exit </execute_bash>.\n'
+    'CodeActAgent': 'When you think you have fixed the issue through code changes, please finish the interaction using the "finish" tool.\n'
 }


--- a/evaluation/discoverybench/run_infer.py
+++ b/evaluation/discoverybench/run_infer.py
@@ -55,7 +55,7 @@ AGENT_CLS_TO_FAKE_USER_RESPONSE_FN = {
 }

 AGENT_CLS_TO_INST_SUFFIX = {
-    'CodeActAgent': 'When you think you have fixed the issue through code changes, please run the following command: <execute_bash> exit </execute_bash>.\n'
+    'CodeActAgent': 'When you think you have fixed the issue through code changes, please finish the interaction using the "finish" tool.\n'
 }


--- a/evaluation/gorilla/run_infer.py
+++ b/evaluation/gorilla/run_infer.py
@@ -33,7 +33,7 @@ AGENT_CLS_TO_FAKE_USER_RESPONSE_FN = {
 }

 AGENT_CLS_TO_INST_SUFFIX = {
-    'CodeActAgent': 'When you think you have completed the request, please run the following command: <execute_bash> exit </execute_bash>.\n'
+    'CodeActAgent': 'When you think you have completed the request, please finish the interaction using the "finish" tool.\n'
 }


--- a/evaluation/gpqa/run_infer.py
+++ b/evaluation/gpqa/run_infer.py
@@ -87,11 +87,10 @@ def gpqa_codeact_user_response(
    msg = (
        'Please continue working on the task on whatever approach you think is suitable.\n'
        'Feel free to use all tools for calculations and solving the problem, and web-search for finding relevant facts during the process if needed\n'
-        'If you have finished reporting the answer in the expected format, (and only once that is done), please run the following command to submit: <execute_bash> exit </execute_bash>.\n'
+        'If you have finished reporting the answer in the expected format, (and only once that is done), please use the "finish" tool to finish the interaction.\n'
        'Again you are being told a million times to first report the answer in the requested format (see again below for reference) before exiting. DO NOT EXIT WITHOUT REPORTING THE ANSWER FIRST.\n'
        'That is, when you have decided on the answer report in the following format:\n'
        f'{ACTION_FORMAT}\n'
-        '<execute_bash> exit </execute_bash>\n'
        'IMPORTANT: YOU SHOULD NEVER ASK FOR HUMAN HELP TO SOLVE THIS TASK.\n'
    )
    return msg
@@ -100,7 +99,7 @@ def gpqa_codeact_user_response(
 AGENT_CLS_TO_FAKE_USER_RESPONSE_FN = {'CodeActAgent': gpqa_codeact_user_response}

 AGENT_CLS_TO_INST_SUFFIX = {
-    'CodeActAgent': '\n\n SUPER IMPORTANT: When you think you have solved the question, first report it back to the user in the requested format. Only once that is done, in the next turn, please run the following command: <execute_bash> exit </execute_bash>.\n'
+    'CodeActAgent': '\n\n SUPER IMPORTANT: When you think you have solved the question, first report it back to the user in the requested format. Only once that is done, in the next turn, please finish the interaction using the "finish" tool.\n'
 }


@@ -205,12 +204,11 @@ Additional Instructions:
 - Do not try to solve the question in a single step. Break it down into smaller steps.
 - You should ONLY interact with the environment provided to you AND NEVER ASK FOR HUMAN HELP.

- SUPER IMPORTANT: When you have reported the answer to the user in the requested format, (and only once that is done) in the next turn, please run the following command: <execute_bash> exit </execute_bash>.
+- SUPER IMPORTANT: When you have reported the answer to the user in the requested format, (and only once that is done) in the next turn, please finish the interaction using the "finish" tool.
 - Again you are being told a million times to first report the answer in the requested format (see again below for reference) before exiting. DO NOT EXIT WITHOUT REPORTING THE ANSWER FIRST.
    That is, when you have decided on the answer report in the following format:

 {ACTION_FORMAT}
-<execute_bash> exit </execute_bash>

 Again do not quit without reporting the answer first.
 Ok now its time to start solving the question. Good luck!
--- a/evaluation/humanevalfix/README.md
+++ b/evaluation/humanevalfix/README.md
@@ -23,7 +23,7 @@ For each problem, OpenHands is given a set number of iterations to fix the faili
 ```
 {
    "task_id": "Python/2",
-    "instruction": "Please fix the function in Python__2.py such that all test cases pass.\nEnvironment has been set up for you to start working. You may assume all necessary tools are installed.\n\n# Problem Statement\ndef truncate_number(number: float) -> float:\n    return number % 1.0 + 1.0\n\n\n\n\n\n\ndef check(truncate_number):\n    assert truncate_number(3.5) == 0.5\n    assert abs(truncate_number(1.33) - 0.33) < 1e-6\n    assert abs(truncate_number(123.456) - 0.456) < 1e-6\n\ncheck(truncate_number)\n\nIMPORTANT: You should ONLY interact with the environment provided to you AND NEVER ASK FOR HUMAN HELP.\nYou should NOT modify any existing test case files. If needed, you can add new test cases in a NEW file to reproduce the issue.\nYou SHOULD INCLUDE PROPER INDENTATION in your edit commands.\nWhen you think you have fixed the issue through code changes, please run the following command: <execute_bash> exit </execute_bash>.\n",
+    "instruction": "Please fix the function in Python__2.py such that all test cases pass.\nEnvironment has been set up for you to start working. You may assume all necessary tools are installed.\n\n# Problem Statement\ndef truncate_number(number: float) -> float:\n    return number % 1.0 + 1.0\n\n\n\n\n\n\ndef check(truncate_number):\n    assert truncate_number(3.5) == 0.5\n    assert abs(truncate_number(1.33) - 0.33) < 1e-6\n    assert abs(truncate_number(123.456) - 0.456) < 1e-6\n\ncheck(truncate_number)\n\nIMPORTANT: You should ONLY interact with the environment provided to you AND NEVER ASK FOR HUMAN HELP.\nYou should NOT modify any existing test case files. If needed, you can add new test cases in a NEW file to reproduce the issue.\nYou SHOULD INCLUDE PROPER INDENTATION in your edit commands.\nWhen you think you have fixed the issue through code changes, please finish the interaction using the "finish" tool.\n",
    "metadata": {
        "agent_class": "CodeActAgent",
        "model_name": "gpt-4",
@@ -38,10 +38,10 @@ For each problem, OpenHands is given a set number of iterations to fix the faili
                "id": 27,
                "timestamp": "2024-05-22T20:57:24.688651",
                "source": "user",
-                "message": "Please fix the function in Python__2.py such that all test cases pass.\nEnvironment has been set up for you to start working. You may assume all necessary tools are installed.\n\n# Problem Statement\ndef truncate_number(number: float) -> float:\n    return number % 1.0 + 1.0\n\n\n\n\n\n\ndef check(truncate_number):\n    assert truncate_number(3.5) == 0.5\n    assert abs(truncate_number(1.33) - 0.33) < 1e-6\n    assert abs(truncate_number(123.456) - 0.456) < 1e-6\n\ncheck(truncate_number)\n\nIMPORTANT: You should ONLY interact with the environment provided to you AND NEVER ASK FOR HUMAN HELP.\nYou should NOT modify any existing test case files. If needed, you can add new test cases in a NEW file to reproduce the issue.\nYou SHOULD INCLUDE PROPER INDENTATION in your edit commands.\nWhen you think you have fixed the issue through code changes, please run the following command: <execute_bash> exit </execute_bash>.\n",
+                "message": "Please fix the function in Python__2.py such that all test cases pass.\nEnvironment has been set up for you to start working. You may assume all necessary tools are installed.\n\n# Problem Statement\ndef truncate_number(number: float) -> float:\n    return number % 1.0 + 1.0\n\n\n\n\n\n\ndef check(truncate_number):\n    assert truncate_number(3.5) == 0.5\n    assert abs(truncate_number(1.33) - 0.33) < 1e-6\n    assert abs(truncate_number(123.456) - 0.456) < 1e-6\n\ncheck(truncate_number)\n\nIMPORTANT: You should ONLY interact with the environment provided to you AND NEVER ASK FOR HUMAN HELP.\nYou should NOT modify any existing test case files. If needed, you can add new test cases in a NEW file to reproduce the issue.\nYou SHOULD INCLUDE PROPER INDENTATION in your edit commands.\nWhen you think you have fixed the issue through code changes, please finish the interaction using the "finish" tool.\n",
                "action": "message",
                "args": {
-                    "content": "Please fix the function in Python__2.py such that all test cases pass.\nEnvironment has been set up for you to start working. You may assume all necessary tools are installed.\n\n# Problem Statement\ndef truncate_number(number: float) -> float:\n    return number % 1.0 + 1.0\n\n\n\n\n\n\ndef check(truncate_number):\n    assert truncate_number(3.5) == 0.5\n    assert abs(truncate_number(1.33) - 0.33) < 1e-6\n    assert abs(truncate_number(123.456) - 0.456) < 1e-6\n\ncheck(truncate_number)\n\nIMPORTANT: You should ONLY interact with the environment provided to you AND NEVER ASK FOR HUMAN HELP.\nYou should NOT modify any existing test case files. If needed, you can add new test cases in a NEW file to reproduce the issue.\nYou SHOULD INCLUDE PROPER INDENTATION in your edit commands.\nWhen you think you have fixed the issue through code changes, please run the following command: <execute_bash> exit </execute_bash>.\n",
+                    "content": "Please fix the function in Python__2.py such that all test cases pass.\nEnvironment has been set up for you to start working. You may assume all necessary tools are installed.\n\n# Problem Statement\ndef truncate_number(number: float) -> float:\n    return number % 1.0 + 1.0\n\n\n\n\n\n\ndef check(truncate_number):\n    assert truncate_number(3.5) == 0.5\n    assert abs(truncate_number(1.33) - 0.33) < 1e-6\n    assert abs(truncate_number(123.456) - 0.456) < 1e-6\n\ncheck(truncate_number)\n\nIMPORTANT: You should ONLY interact with the environment provided to you AND NEVER ASK FOR HUMAN HELP.\nYou should NOT modify any existing test case files. If needed, you can add new test cases in a NEW file to reproduce the issue.\nYou SHOULD INCLUDE PROPER INDENTATION in your edit commands.\nWhen you think you have fixed the issue through code changes, please finish the interaction using the "finish" tool.\n",
                    "wait_for_response": false
                }
            },
--- a/evaluation/humanevalfix/run_infer.py
+++ b/evaluation/humanevalfix/run_infer.py
@@ -75,7 +75,7 @@ AGENT_CLS_TO_FAKE_USER_RESPONSE_FN = {
 }

 AGENT_CLS_TO_INST_SUFFIX = {
-    'CodeActAgent': 'When you think you have fixed the issue through code changes, please run the following command: <execute_bash> exit </execute_bash>.\n'
+    'CodeActAgent': 'When you think you have fixed the issue through code changes, please finish the interaction using the "finish" tool.\n'
 }


--- a/evaluation/miniwob/README.md
+++ b/evaluation/miniwob/README.md
@@ -16,6 +16,20 @@ Access with browser the above MiniWoB URLs and see if they load correctly.
 ./evaluation/miniwob/scripts/run_infer.sh llm.claude-35-sonnet-eval
 ```

+### Run Inference on `RemoteRuntime` (experimental)
+
+This is in limited beta. Contact Xingyao over slack if you want to try this out!
+
+```bash
+./evaluation/miniwob/scripts/run_infer.sh [model_config] [git-version] [agent] [note] [eval_limit] [num_workers]
+
+# Example - This runs evaluation on BrowsingAgent for 125 instances on miniwob, with 2 workers running in parallel
+export ALLHANDS_API_KEY="YOUR-API-KEY"
+export RUNTIME=remote
+export SANDBOX_REMOTE_RUNTIME_API_URL="https://runtime.eval.all-hands.dev"
+./evaluation/miniwob/scripts/run_infer.sh llm.eval HEAD BrowsingAgent "" 125 2
+```
+
 Results will be in `evaluation/evaluation_outputs/outputs/miniwob/`

 To calculate the average reward, run:
--- a/evaluation/miniwob/get_avg_reward.py
+++ b/evaluation/miniwob/get_avg_reward.py
@@ -23,7 +23,7 @@ if __name__ == '__main__':
            data = json.loads(line)
            actual_num += 1
            total_cost += data['metrics']['accumulated_cost']
-            total_reward += data['test_result']
+            total_reward += data['test_result']['reward']

    avg_reward = total_reward / total_num
    print('Avg Reward: ', avg_reward)
--- a/evaluation/miniwob/run_infer.py
+++ b/evaluation/miniwob/run_infer.py
@@ -47,6 +47,7 @@ SUPPORTED_AGENT_CLS = {'BrowsingAgent', 'CodeActAgent'}

 AGENT_CLS_TO_FAKE_USER_RESPONSE_FN = {
    'CodeActAgent': codeact_user_response,
+    'BrowsingAgent': 'Continue the task. IMPORTANT: do not talk to the user until you have finished the task',
 }


@@ -66,7 +67,9 @@ def get_config(
            browsergym_eval_env=env_id,
            api_key=os.environ.get('ALLHANDS_API_KEY', None),
            remote_runtime_api_url=os.environ.get('SANDBOX_REMOTE_RUNTIME_API_URL'),
-            keep_remote_runtime_alive=False,
+            remote_runtime_init_timeout=1800,
+            keep_runtime_alive=False,
+            timeout=120,
        ),
        # do not mount workspace
        workspace_base=None,
--- a/evaluation/miniwob/scripts/run_infer.sh
+++ b/evaluation/miniwob/scripts/run_infer.sh
@@ -33,7 +33,7 @@ echo "MODEL_CONFIG: $MODEL_CONFIG"

 EVAL_NOTE="${AGENT_VERSION}_${NOTE}"

-COMMAND="poetry run python evaluation/miniwob/run_infer.py \
+COMMAND="export PYTHONPATH=evaluation/miniwob:\$PYTHONPATH && poetry run python evaluation/miniwob/run_infer.py \
  --agent-cls $AGENT \
  --llm-config $MODEL_CONFIG \
  --max-iterations 10 \
--- a/evaluation/mint/run_infer.py
+++ b/evaluation/mint/run_infer.py
@@ -70,7 +70,7 @@ AGENT_CLS_TO_FAKE_USER_RESPONSE_FN = {
 }

 AGENT_CLS_TO_INST_SUFFIX = {
-    'CodeActAgent': '\nIMPORTANT: When your answer is confirmed by the user to be correct, you can exit using the following command: <execute_bash> exit </execute_bash>.\n'
+    'CodeActAgent': 'IMPORTANT: When your answer is confirmed by the user to be correct, you can use the "finish" tool to finish the interaction.\n'
 }

 with open(os.path.join(os.path.dirname(__file__), 'requirements.txt'), 'r') as f:
--- a/evaluation/ml_bench/README.md
+++ b/evaluation/ml_bench/README.md
@@ -55,7 +55,7 @@ Here's an example of the evaluation output for a single task instance:
 {
  "instance_id": 3,
  "repo": "https://github.com/dmlc/dgl",
-  "instruction": "Please complete the Machine Learning task in the following repository: dgl\n\nThe task is: DGL Implementation of NGCF model\n\nI have a deep desire to embark on a journey brimming with knowledge and expertise. My objective is to train a cutting-edge NGCF Model, known for its unparalleled capabilities, on the illustrious dataset known as gowalla. To ensure swift execution, I kindly request your assistance in crafting the code, making use of the powerful GPU #3 and an embedding size of 32. Can you lend a helping hand to transform this dream into a reality?\n\nYou should create a script named `run.sh` under the specified path in the repo to run the task.\n\nYou can find the task repo at: /workspace/dgl/examples/pytorch/NGCF/NGCF\n\nYou should terminate the subprocess after running the task (e.g., call subprocess.Popen(args).wait()).When you think you have completed the task, please run the following command: <execute_bash> exit </execute_bash>.\n",
+  "instruction": "Please complete the Machine Learning task in the following repository: dgl\n\nThe task is: DGL Implementation of NGCF model\n\nI have a deep desire to embark on a journey brimming with knowledge and expertise. My objective is to train a cutting-edge NGCF Model, known for its unparalleled capabilities, on the illustrious dataset known as gowalla. To ensure swift execution, I kindly request your assistance in crafting the code, making use of the powerful GPU #3 and an embedding size of 32. Can you lend a helping hand to transform this dream into a reality?\n\nYou should create a script named `run.sh` under the specified path in the repo to run the task.\n\nYou can find the task repo at: /workspace/dgl/examples/pytorch/NGCF/NGCF\n\nYou should terminate the subprocess after running the task (e.g., call subprocess.Popen(args).wait()).When you think you have completed the task, please finish the interaction using the "finish" tool.\n",
  "metadata": {
    "agent_class": "CodeActAgent",
    "model_name": "gpt-4-1106-preview",
@@ -70,10 +70,10 @@ Here's an example of the evaluation output for a single task instance:
        "id": 0,
        "timestamp": "2024-05-26T17:40:41.060009",
        "source": "user",
-        "message": "Please complete the Machine Learning task in the following repository: dgl\n\nThe task is: DGL Implementation of NGCF model\n\nI have a deep desire to embark on a journey brimming with knowledge and expertise. My objective is to train a cutting-edge NGCF Model, known for its unparalleled capabilities, on the illustrious dataset known as gowalla. To ensure swift execution, I kindly request your assistance in crafting the code, making use of the powerful GPU #3 and an embedding size of 32. Can you lend a helping hand to transform this dream into a reality?\n\nYou should create a script named `run.sh` under the specified path in the repo to run the task.\n\nYou can find the task repo at: /workspace/dgl/examples/pytorch/NGCF/NGCF\n\nYou should terminate the subprocess after running the task (e.g., call subprocess.Popen(args).wait()).When you think you have completed the task, please run the following command: <execute_bash> exit </execute_bash>.\n",
+        "message": "Please complete the Machine Learning task in the following repository: dgl\n\nThe task is: DGL Implementation of NGCF model\n\nI have a deep desire to embark on a journey brimming with knowledge and expertise. My objective is to train a cutting-edge NGCF Model, known for its unparalleled capabilities, on the illustrious dataset known as gowalla. To ensure swift execution, I kindly request your assistance in crafting the code, making use of the powerful GPU #3 and an embedding size of 32. Can you lend a helping hand to transform this dream into a reality?\n\nYou should create a script named `run.sh` under the specified path in the repo to run the task.\n\nYou can find the task repo at: /workspace/dgl/examples/pytorch/NGCF/NGCF\n\nYou should terminate the subprocess after running the task (e.g., call subprocess.Popen(args).wait()).When you think you have completed the task, please finish the interaction using the "finish" tool.\n",
        "action": "message",
        "args": {
-          "content": "Please complete the Machine Learning task in the following repository: dgl\n\nThe task is: DGL Implementation of NGCF model\n\nI have a deep desire to embark on a journey brimming with knowledge and expertise. My objective is to train a cutting-edge NGCF Model, known for its unparalleled capabilities, on the illustrious dataset known as gowalla. To ensure swift execution, I kindly request your assistance in crafting the code, making use of the powerful GPU #3 and an embedding size of 32. Can you lend a helping hand to transform this dream into a reality?\n\nYou should create a script named `run.sh` under the specified path in the repo to run the task.\n\nYou can find the task repo at: /workspace/dgl/examples/pytorch/NGCF/NGCF\n\nYou should terminate the subprocess after running the task (e.g., call subprocess.Popen(args).wait()).When you think you have completed the task, please run the following command: <execute_bash> exit </execute_bash>.\n",
+          "content": "Please complete the Machine Learning task in the following repository: dgl\n\nThe task is: DGL Implementation of NGCF model\n\nI have a deep desire to embark on a journey brimming with knowledge and expertise. My objective is to train a cutting-edge NGCF Model, known for its unparalleled capabilities, on the illustrious dataset known as gowalla. To ensure swift execution, I kindly request your assistance in crafting the code, making use of the powerful GPU #3 and an embedding size of 32. Can you lend a helping hand to transform this dream into a reality?\n\nYou should create a script named `run.sh` under the specified path in the repo to run the task.\n\nYou can find the task repo at: /workspace/dgl/examples/pytorch/NGCF/NGCF\n\nYou should terminate the subprocess after running the task (e.g., call subprocess.Popen(args).wait()).When you think you have completed the task, please finish the interaction using the "finish" tool.\n",
          "wait_for_response": false
        }
      },
--- a/evaluation/ml_bench/run_infer.py
+++ b/evaluation/ml_bench/run_infer.py
@@ -52,7 +52,7 @@ AGENT_CLS_TO_FAKE_USER_RESPONSE_FN = {
 }

 AGENT_CLS_TO_INST_SUFFIX = {
-    'CodeActAgent': 'When you think you have completed the task, please run the following command: <execute_bash> exit </execute_bash>.\n'
+    'CodeActAgent': 'When you think you have completed the task, please finish the interaction using the "finish" tool.\n'
 }

 ID2CONDA = {
--- a/evaluation/scienceagentbench/run_infer.py
+++ b/evaluation/scienceagentbench/run_infer.py
@@ -72,7 +72,7 @@ def get_config(
            timeout=300,
            api_key=os.environ.get('ALLHANDS_API_KEY', None),
            remote_runtime_api_url=os.environ.get('SANDBOX_REMOTE_RUNTIME_API_URL'),
-            keep_remote_runtime_alive=False,
+            keep_runtime_alive=False,
        ),
        # do not mount workspace
        workspace_base=None,
--- a/evaluation/swe_bench/eval_infer.py
+++ b/evaluation/swe_bench/eval_infer.py
@@ -1,6 +1,7 @@
 import os
 import tempfile
 import time
+from functools import partial

 import pandas as pd
 from swebench.harness.grading import get_eval_report
@@ -83,7 +84,7 @@ def get_config(instance: pd.Series) -> AppConfig:
            timeout=1800,
            api_key=os.environ.get('ALLHANDS_API_KEY', None),
            remote_runtime_api_url=os.environ.get('SANDBOX_REMOTE_RUNTIME_API_URL'),
-            remote_runtime_init_timeout=1800,
+            remote_runtime_init_timeout=3600,
        ),
        # do not mount workspace
        workspace_base=None,
@@ -94,13 +95,28 @@ def get_config(instance: pd.Series) -> AppConfig:

 def process_instance(
    instance: pd.Series,
-    metadata: EvalMetadata | None = None,
+    metadata: EvalMetadata,
    reset_logger: bool = True,
+    log_dir: str | None = None,
 ) -> EvalOutput:
+    """
+    Evaluate agent performance on a SWE-bench problem instance.
+
+    Note that this signature differs from the expected input to `run_evaluation`. Use
+    `functools.partial` to provide optional arguments before passing to the evaluation harness.
+
+    Args:
+        log_dir (str | None, default=None): Path to directory where log files will be written. Must
+        be provided if `reset_logger` is set.
+
+    Raises:
+        AssertionError: if the `reset_logger` flag is set without a provided log directory.
+    """
    # Setup the logger properly, so you can run multi-processing to parallelize the evaluation
    if reset_logger:
-        global output_file
-        log_dir = output_file.replace('.jsonl', '.logs')
+        assert (
+            log_dir is not None
+        ), "Can't reset logger without a provided log directory."
        os.makedirs(log_dir, exist_ok=True)
        reset_logger_for_multiprocessing(logger, instance.instance_id, log_dir)
    else:
@@ -127,6 +143,7 @@ def process_instance(
        return EvalOutput(
            instance_id=instance_id,
            test_result=instance['test_result'],
+            metadata=metadata,
        )

    runtime = create_runtime(config)
@@ -176,6 +193,7 @@ def process_instance(
            return EvalOutput(
                instance_id=instance_id,
                test_result=instance['test_result'],
+                metadata=metadata,
            )
        elif 'APPLY_PATCH_PASS' in apply_patch_output:
            logger.info(f'[{instance_id}] {APPLY_PATCH_PASS}:\n{apply_patch_output}')
@@ -245,23 +263,29 @@ def process_instance(
                        test_output_path = os.path.join(log_dir, 'test_output.txt')
                        with open(test_output_path, 'w') as f:
                            f.write(test_output)
-
-                        _report = get_eval_report(
-                            test_spec=test_spec,
-                            prediction={
-                                'model_patch': model_patch,
-                                'instance_id': instance_id,
-                            },
-                            log_path=test_output_path,
-                            include_tests_status=True,
-                        )
-                        report = _report[instance_id]
-                        logger.info(
-                            f"[{instance_id}] report: {report}\nResult for {instance_id}: resolved: {report['resolved']}"
-                        )
-                        instance['test_result']['report']['resolved'] = report[
-                            'resolved'
-                        ]
+                        try:
+                            _report = get_eval_report(
+                                test_spec=test_spec,
+                                prediction={
+                                    'model_patch': model_patch,
+                                    'instance_id': instance_id,
+                                },
+                                log_path=test_output_path,
+                                include_tests_status=True,
+                            )
+                            report = _report[instance_id]
+                            logger.info(
+                                f"[{instance_id}] report: {report}\nResult for {instance_id}: resolved: {report['resolved']}"
+                            )
+                            instance['test_result']['report']['resolved'] = report[
+                                'resolved'
+                            ]
+                        except Exception as e:
+                            logger.error(
+                                f'[{instance_id}] Error when getting eval report: {e}'
+                            )
+                            instance['test_result']['report']['resolved'] = False
+                            instance['test_result']['report']['error_eval'] = True
            else:
                logger.info(f'[{instance_id}] Error when starting eval:\n{obs.content}')
                instance['test_result']['report']['error_eval'] = True
@@ -269,6 +293,7 @@ def process_instance(
            return EvalOutput(
                instance_id=instance_id,
                test_result=instance['test_result'],
+                metadata=metadata,
            )
        else:
            logger.info(
@@ -336,7 +361,7 @@ if __name__ == '__main__':

    if 'model_patch' not in predictions.columns:
        predictions['model_patch'] = predictions['test_result'].apply(
-            lambda x: x['git_patch']
+            lambda x: x.get('git_patch', '')
        )
    assert {'instance_id', 'model_patch'}.issubset(
        set(predictions.columns)
@@ -355,12 +380,26 @@ if __name__ == '__main__':
    output_file = args.input_file.replace('.jsonl', '.swebench_eval.jsonl')
    instances = prepare_dataset(predictions, output_file, args.eval_n_limit)

+    # If possible, load the relevant metadata to avoid issues with `run_evaluation`.
+    metadata: EvalMetadata | None = None
+    metadata_filepath = os.path.join(os.path.dirname(args.input_file), 'metadata.json')
+    if os.path.exists(metadata_filepath):
+        with open(metadata_filepath, 'r') as metadata_file:
+            data = metadata_file.read()
+            metadata = EvalMetadata.model_validate_json(data)
+
+    # The evaluation harness constrains the signature of `process_instance_func` but we need to
+    # pass extra information. Build a new function object to avoid issues with multiprocessing.
+    process_instance_func = partial(
+        process_instance, log_dir=output_file.replace('.jsonl', '.logs')
+    )
+
    run_evaluation(
        instances,
-        metadata=None,
+        metadata=metadata,
        output_file=output_file,
        num_workers=args.eval_num_workers,
-        process_instance_func=process_instance,
+        process_instance_func=process_instance_func,
    )

    # Load evaluated predictions & print number of resolved predictions
--- a/evaluation/swe_bench/examples/example_agent_output.jsonl
+++ b/evaluation/swe_bench/examples/example_agent_output.jsonl
--- a/evaluation/swe_bench/prompt.py
+++ b/evaluation/swe_bench/prompt.py
@@ -1,6 +1,6 @@
 CODEACT_SWE_PROMPT = """Now, you're going to solve this issue on your own. Your terminal session has started and you're in the repository's root directory. You can use any bash commands or the special interface to help you. Edit all the files you need to and run any checks or tests that you want.
 Remember, YOU CAN ONLY ENTER ONE COMMAND AT A TIME. You should always wait for feedback after every command.
-When you're satisfied with all of the changes you've made, you can run the following command: <execute_bash> exit </execute_bash>.
+When you're satisfied with all of the changes you've made, you can use the "finish" tool to finish the interaction.
 Note however that you cannot use any interactive session commands (e.g. vim) in this environment, but you can write scripts and run them. E.g. you can write a python script and then run it with `python <script_name>.py`.

 NOTE ABOUT THE EDIT COMMAND: Indentation really matters! When editing a file, make sure to insert appropriate indentation before each line!
--- a/evaluation/swe_bench/run_infer.py
+++ b/evaluation/swe_bench/run_infer.py
@@ -36,8 +36,8 @@ from openhands.events.action import CmdRunAction, MessageAction
 from openhands.events.observation import CmdOutputObservation, ErrorObservation
 from openhands.events.serialization.event import event_to_dict
 from openhands.runtime.base import Runtime
-from openhands.runtime.utils.shutdown_listener import sleep_if_should_continue
 from openhands.utils.async_utils import call_async_from_sync
+from openhands.utils.shutdown_listener import sleep_if_should_continue

 USE_HINT_TEXT = os.environ.get('USE_HINT_TEXT', 'false').lower() == 'true'
 USE_INSTANCE_IMAGE = os.environ.get('USE_INSTANCE_IMAGE', 'false').lower() == 'true'
@@ -146,7 +146,7 @@ def get_config(
            api_key=os.environ.get('ALLHANDS_API_KEY', None),
            remote_runtime_api_url=os.environ.get('SANDBOX_REMOTE_RUNTIME_API_URL'),
            keep_remote_runtime_alive=False,
-            remote_runtime_init_timeout=1800,
+            remote_runtime_init_timeout=3600,
        ),
        # do not mount workspace
        workspace_base=None,
@@ -534,5 +534,10 @@ if __name__ == '__main__':
            instances[col] = instances[col].apply(lambda x: str(x))

    run_evaluation(
-        instances, metadata, output_file, args.eval_num_workers, process_instance
+        instances,
+        metadata,
+        output_file,
+        args.eval_num_workers,
+        process_instance,
+        timeout_seconds=120 * 60,  # 2 hour PER instance should be more than enough
    )
--- a/evaluation/toolqa/run_infer.py
+++ b/evaluation/toolqa/run_infer.py
@@ -34,7 +34,7 @@ AGENT_CLS_TO_FAKE_USER_RESPONSE_FN = {
 }

 AGENT_CLS_TO_INST_SUFFIX = {
-    'CodeActAgent': 'When you think you have completed the request, please run the following command: <execute_bash> exit </execute_bash>.\n'
+    'CodeActAgent': 'When you think you have completed the request, please finish the interaction using the "finish" tool.\n'
 }


--- a/evaluation/utils/shared.py
+++ b/evaluation/utils/shared.py
@@ -137,7 +137,7 @@ def codeact_user_response(
            # let the agent know that it can give up when it has tried 3 times
            return (
                msg
-                + 'If you want to give up, run: <execute_bash> exit </execute_bash>.\n'
+                + 'If you want to give up, use the "finish" tool to finish the interaction.\n'
            )
    return msg

@@ -346,6 +346,7 @@ def run_evaluation(
            f'model {metadata.llm_config.model}, max iterations {metadata.max_iterations}.\n'
        )
    else:
+        logger.warning('Running evaluation without metadata.')
        logger.info(f'Evaluation started with {num_workers} workers.')

    total_instances = len(dataset)
--- a/frontend/tests/clear-session.test.ts
+++ b/frontend/tests/clear-session.test.ts
@@ -0,0 +1,40 @@
+import { describe, it, expect, beforeEach, vi } from "vitest";
+import { clearSession } from "../src/utils/clear-session";
+import store from "../src/store";
+import { initialState as browserInitialState } from "../src/state/browserSlice";
+
+describe("clearSession", () => {
+  beforeEach(() => {
+    // Mock localStorage
+    const localStorageMock = {
+      getItem: vi.fn(),
+      setItem: vi.fn(),
+      removeItem: vi.fn(),
+      clear: vi.fn(),
+    };
+    vi.stubGlobal("localStorage", localStorageMock);
+
+    // Set initial browser state to non-default values
+    store.dispatch({
+      type: "browser/setUrl",
+      payload: "https://example.com",
+    });
+    store.dispatch({
+      type: "browser/setScreenshotSrc",
+      payload: "base64screenshot",
+    });
+  });
+
+  it("should clear localStorage and reset browser state", () => {
+    clearSession();
+
+    // Verify localStorage items were removed
+    expect(localStorage.removeItem).toHaveBeenCalledWith("token");
+    expect(localStorage.removeItem).toHaveBeenCalledWith("repo");
+
+    // Verify browser state was reset
+    const state = store.getState();
+    expect(state.browser.url).toBe(browserInitialState.url);
+    expect(state.browser.screenshotSrc).toBe(browserInitialState.screenshotSrc);
+  });
+});
--- a/frontend/tests/components/chat/chat-interface.test.tsx
+++ b/frontend/tests/components/chat/chat-interface.test.tsx
@@ -16,14 +16,19 @@ describe("Empty state", () => {
    send: vi.fn(),
  }));

-  const { useSocket: useSocketMock } = vi.hoisted(() => ({
-    useSocket: vi.fn(() => ({ send: sendMock, runtimeActive: true })),
+  const { useWsClient: useWsClientMock } = vi.hoisted(() => ({
+    useWsClient: vi.fn(() => ({ send: sendMock, runtimeActive: true })),
  }));

  beforeAll(() => {
+    vi.mock("@remix-run/react", async (importActual) => ({
+      ...(await importActual<typeof import("@remix-run/react")>()),
+      useRouteLoaderData: vi.fn(() => ({})),
+    }));
+
    vi.mock("#/context/socket", async (importActual) => ({
-      ...(await importActual<typeof import("#/context/socket")>()),
-      useSocket: useSocketMock,
+      ...(await importActual<typeof import("#/context/ws-client-provider")>()),
+      useWsClient: useWsClientMock,
    }));
  });

@@ -77,7 +82,7 @@ describe("Empty state", () => {
    "should load the a user message to the input when selecting",
    async () => {
      // this is to test that the message is in the UI before the socket is called
-      useSocketMock.mockImplementation(() => ({
+      useWsClientMock.mockImplementation(() => ({
        send: sendMock,
        runtimeActive: false, // mock an inactive runtime setup
      }));
@@ -106,7 +111,7 @@ describe("Empty state", () => {
  it.fails(
    "should send the message to the socket only if the runtime is active",
    async () => {
-      useSocketMock.mockImplementation(() => ({
+      useWsClientMock.mockImplementation(() => ({
        send: sendMock,
        runtimeActive: false, // mock an inactive runtime setup
      }));
@@ -123,7 +128,7 @@ describe("Empty state", () => {
      await user.click(displayedSuggestions[0]);
      expect(sendMock).not.toHaveBeenCalled();

-      useSocketMock.mockImplementation(() => ({
+      useWsClientMock.mockImplementation(() => ({
        send: sendMock,
        runtimeActive: true, // mock an active runtime setup
      }));
--- a/frontend/tests/hooks/use-rate.test.ts
+++ b/frontend/tests/hooks/use-rate.test.ts
@@ -0,0 +1,93 @@
+import { act, renderHook } from "@testing-library/react";
+import { afterEach, beforeEach, describe, expect, it, vi } from "vitest";
+import { useRate } from "#/utils/use-rate";
+
+describe("useRate", () => {
+  beforeEach(() => {
+    vi.useFakeTimers();
+  });
+
+  afterEach(() => {
+    vi.useRealTimers();
+  });
+
+  it("should initialize", () => {
+    const { result } = renderHook(() => useRate());
+
+    expect(result.current.items).toHaveLength(0);
+    expect(result.current.rate).toBeNull();
+    expect(result.current.lastUpdated).toBeNull();
+    expect(result.current.isUnderThreshold).toBe(true);
+  });
+
+  it("should handle the case of a single element", () => {
+    const { result } = renderHook(() => useRate());
+
+    act(() => {
+      result.current.record(123);
+    });
+
+    expect(result.current.items).toHaveLength(1);
+    expect(result.current.lastUpdated).not.toBeNull();
+  });
+
+  it("should return the difference between the last two elements", () => {
+    const { result } = renderHook(() => useRate());
+
+    vi.setSystemTime(500);
+    act(() => {
+      result.current.record(4);
+    });
+
+    vi.advanceTimersByTime(500);
+    act(() => {
+      result.current.record(9);
+    });
+
+    expect(result.current.items).toHaveLength(2);
+    expect(result.current.rate).toBe(5);
+    expect(result.current.lastUpdated).toBe(1000);
+  });
+
+  it("should update isUnderThreshold after [threshold]ms of no activity", () => {
+    const { result } = renderHook(() => useRate({ threshold: 500 }));
+
+    expect(result.current.isUnderThreshold).toBe(true);
+
+    act(() => {
+       // not sure if fake timers is buggy with intervals,
+       // but I need to call it twice to register
+      vi.advanceTimersToNextTimer();
+      vi.advanceTimersToNextTimer();
+    });
+
+    expect(result.current.isUnderThreshold).toBe(false);
+  });
+
+  it("should return an isUnderThreshold boolean", () => {
+    const { result } = renderHook(() => useRate({ threshold: 500 }));
+
+    vi.setSystemTime(500);
+    act(() => {
+      result.current.record(400);
+    });
+    act(() => {
+      result.current.record(1000);
+    });
+
+    expect(result.current.isUnderThreshold).toBe(false);
+
+    act(() => {
+      result.current.record(1500);
+    });
+
+    expect(result.current.isUnderThreshold).toBe(true);
+
+    act(() => {
+      vi.advanceTimersToNextTimer();
+      vi.advanceTimersToNextTimer();
+    });
+
+    expect(result.current.isUnderThreshold).toBe(false);
+  });
+});
--- a/frontend/tests/hooks/use-terminal.test.tsx
+++ b/frontend/tests/hooks/use-terminal.test.tsx
@@ -2,8 +2,9 @@ import { beforeAll, describe, expect, it, vi } from "vitest";
 import { render } from "@testing-library/react";
 import { afterEach } from "node:test";
 import { useTerminal } from "#/hooks/useTerminal";
-import { SocketProvider } from "#/context/socket";
 import { Command } from "#/state/commandSlice";
+import { WsClientProvider } from "#/context/ws-client-provider";
+import { ReactNode } from "react";

 interface TestTerminalComponentProps {
  commands: Command[];
@@ -18,6 +19,17 @@ function TestTerminalComponent({
  return <div ref={ref} />;
 }

+interface WrapperProps {
+  children: ReactNode;
+}
+
+
+function Wrapper({children}: WrapperProps) {
+  return (
+    <WsClientProvider enabled={true} token="NO_JWT" ghToken="NO_GITHUB" settings={null}>{children}</WsClientProvider>
+  )
+}
+
 describe("useTerminal", () => {
  const mockTerminal = vi.hoisted(() => ({
    loadAddon: vi.fn(),
@@ -50,7 +62,7 @@ describe("useTerminal", () => {

  it("should render", () => {
    render(<TestTerminalComponent commands={[]} secrets={[]} />, {
-      wrapper: SocketProvider,
+      wrapper: Wrapper,
    });
  });

@@ -61,7 +73,7 @@ describe("useTerminal", () => {
    ];

    render(<TestTerminalComponent commands={commands} secrets={[]} />, {
-      wrapper: SocketProvider,
+      wrapper: Wrapper,
    });

    expect(mockTerminal.writeln).toHaveBeenNthCalledWith(1, "echo hello");
@@ -85,7 +97,7 @@ describe("useTerminal", () => {
        secrets={[secret, anotherSecret]}
      />,
      {
-        wrapper: SocketProvider,
+        wrapper: Wrapper,
      },
    );

--- a/frontend/tests/utils/extractModelAndProvider.test.ts
+++ b/frontend/tests/utils/extractModelAndProvider.test.ts
@@ -59,9 +59,9 @@ describe("extractModelAndProvider", () => {
      separator: "/",
    });

-    expect(extractModelAndProvider("claude-3-5-sonnet-20241022")).toEqual({
+    expect(extractModelAndProvider("claude-3-5-sonnet-20240620")).toEqual({
      provider: "anthropic",
-      model: "claude-3-5-sonnet-20241022",
+      model: "claude-3-5-sonnet-20240620",
      separator: "/",
    });

--- a/frontend/tests/utils/organizeModelsAndProviders.test.ts
+++ b/frontend/tests/utils/organizeModelsAndProviders.test.ts
@@ -15,7 +15,7 @@ test("organizeModelsAndProviders", () => {
    "gpt-4o",
    "together-ai-21.1b-41b",
    "gpt-4o-mini",
-    "claude-3-5-sonnet-20241022",
+    "anthropic/claude-3-5-sonnet-20241022",
    "claude-3-haiku-20240307",
    "claude-2",
    "claude-2.1",
--- a/frontend/package-lock.json
+++ b/frontend/package-lock.json
@@ -1,12 +1,12 @@
 {
  "name": "openhands-frontend",
-  "version": "0.13.0",
+  "version": "0.14.0",
  "lockfileVersion": 3,
  "requires": true,
  "packages": {
    "": {
      "name": "openhands-frontend",
-      "version": "0.13.0",
+      "version": "0.14.0",
      "dependencies": {
        "@monaco-editor/react": "^4.6.0",
        "@nextui-org/react": "^2.4.8",
@@ -26,7 +26,7 @@
        "isbot": "^5.1.17",
        "jose": "^5.9.4",
        "monaco-editor": "^0.52.0",
-        "posthog-js": "^1.176.0",
+        "posthog-js": "^1.184.1",
        "react": "^18.3.1",
        "react-dom": "^18.3.1",
        "react-highlight": "^0.15.0",
@@ -19749,9 +19749,9 @@
      "integrity": "sha512-1NNCs6uurfkVbeXG4S8JFT9t19m45ICnif8zWLd5oPSZ50QnwMfK+H3jv408d4jw/7Bttv5axS5IiHoLaVNHeQ=="
    },
    "node_modules/posthog-js": {
-      "version": "1.176.0",
-      "resolved": "https://registry.npmjs.org/posthog-js/-/posthog-js-1.176.0.tgz",
-      "integrity": "sha512-T5XKNtRzp7q6CGb7Vc7wAI76rWap9fiuDUPxPsyPBPDkreKya91x9RIsSapAVFafwD1AEin1QMczCmt9Le9BWw==",
+      "version": "1.184.1",
+      "resolved": "https://registry.npmjs.org/posthog-js/-/posthog-js-1.184.1.tgz",
+      "integrity": "sha512-q/1Kdard5SZnL2smrzeKcD+RuUi2PnbidiN4D3ThK20bNrhy5Z2heIy9SnRMvEiARY5lcQ7zxmDCAKPBKGSOtQ==",
      "dependencies": {
        "core-js": "^3.38.1",
        "fflate": "^0.4.8",
--- a/frontend/package.json
+++ b/frontend/package.json
@@ -1,6 +1,6 @@
 {
  "name": "openhands-frontend",
-  "version": "0.13.0",
+  "version": "0.14.0",
  "private": true,
  "type": "module",
  "engines": {
@@ -25,7 +25,7 @@
    "isbot": "^5.1.17",
    "jose": "^5.9.4",
    "monaco-editor": "^0.52.0",
-    "posthog-js": "^1.176.0",
+    "posthog-js": "^1.184.1",
    "react": "^18.3.1",
    "react-dom": "^18.3.1",
    "react-highlight": "^0.15.0",
--- a/frontend/public/config.json
+++ b/frontend/public/config.json
@@ -1,4 +1,5 @@
 {
  "APP_MODE": "oss",
-  "GITHUB_CLIENT_ID": ""
+  "GITHUB_CLIENT_ID": "",
+  "POSTHOG_CLIENT_KEY": "phc_3ESMmY9SgqEAGBB6sMGK5ayYHkeUuknH2vP6FmWH9RA"
 }
--- a/frontend/src/api/open-hands.ts
+++ b/frontend/src/api/open-hands.ts
@@ -8,6 +8,7 @@ import {
  GitHubAccessTokenResponse,
  ErrorResponse,
  GetConfigResponse,
+  GetVSCodeUrlResponse,
 } from "./open-hands.types";

 class OpenHands {
@@ -174,6 +175,21 @@ class OpenHands {
      true,
    );
  }
+
+  /**
+   * Get the VSCode URL
+   * @returns VSCode URL
+   */
+  static async getVSCodeUrl(): Promise<GetVSCodeUrlResponse> {
+    return request(`/api/vscode-url`, {}, false, false, 1);
+  }
+
+  static async getRuntimeId(): Promise<{ runtime_id: string }> {
+    const response = await request("/api/config");
+    const data = await response.json();
+
+    return data;
+  }
 }

 export default OpenHands;
--- a/frontend/src/api/open-hands.types.ts
+++ b/frontend/src/api/open-hands.types.ts
@@ -43,5 +43,11 @@ export interface Feedback {

 export interface GetConfigResponse {
  APP_MODE: "saas" | "oss";
-  GITHUB_CLIENT_ID: string | null;
+  GITHUB_CLIENT_ID: string;
+  POSTHOG_CLIENT_KEY: string;
+}
+
+export interface GetVSCodeUrlResponse {
+  vscode_url: string | null;
+  error?: string;
 }
--- a/frontend/src/assets/vscode-alt.svg
+++ b/frontend/src/assets/vscode-alt.svg
@@ -0,0 +1,57 @@
+<svg width="100" height="100" viewBox="0 0 100 100" fill="none" xmlns="http://www.w3.org/2000/svg">
+<g clip-path="url(#clip0)">
+<g filter="url(#filter0_d)">
+<mask id="mask0" mask-type="alpha" maskUnits="userSpaceOnUse" x="0" y="0" width="100" height="100">
+<path fill-rule="evenodd" clip-rule="evenodd" d="M70.9119 99.5723C72.4869 100.189 74.2828 100.15 75.8725 99.3807L96.4604 89.4231C98.624 88.3771 100 86.1762 100 83.7616V16.2392C100 13.8247 98.624 11.6238 96.4604 10.5774L75.8725 0.619067C73.7862 -0.389991 71.3446 -0.142885 69.5135 1.19527C69.252 1.38636 69.0028 1.59985 68.769 1.83502L29.3551 37.9795L12.1872 24.88C10.5891 23.6607 8.35365 23.7606 6.86938 25.1178L1.36302 30.1525C-0.452603 31.8127 -0.454583 34.6837 1.35854 36.3466L16.2471 50.0001L1.35854 63.6536C-0.454583 65.3164 -0.452603 68.1876 1.36302 69.8477L6.86938 74.8824C8.35365 76.2395 10.5891 76.34 12.1872 75.1201L29.3551 62.0207L68.769 98.1651C69.3925 98.7923 70.1246 99.2645 70.9119 99.5723ZM75.0152 27.1813L45.1092 50.0001L75.0152 72.8189V27.1813Z" fill="white"/>
+</mask>
+<g mask="url(#mask0)">
+<path d="M96.4614 10.593L75.8567 0.62085C73.4717 -0.533437 70.6215 -0.0465506 68.7498 1.83492L1.29834 63.6535C-0.515935 65.3164 -0.513852 68.1875 1.30281 69.8476L6.8125 74.8823C8.29771 76.2395 10.5345 76.339 12.1335 75.1201L93.3604 13.18C96.0854 11.102 100 13.0557 100 16.4939V16.2535C100 13.84 98.6239 11.64 96.4614 10.593Z" fill="#D9D9D9"/>
+<g filter="url(#filter1_d)">
+<path d="M96.4614 89.4074L75.8567 99.3797C73.4717 100.534 70.6215 100.047 68.7498 98.1651L1.29834 36.3464C-0.515935 34.6837 -0.513852 31.8125 1.30281 30.1524L6.8125 25.1177C8.29771 23.7605 10.5345 23.6606 12.1335 24.88L93.3604 86.8201C96.0854 88.8985 100 86.9447 100 83.5061V83.747C100 86.1604 98.6239 88.3603 96.4614 89.4074Z" fill="#E6E6E6"/>
+</g>
+<g filter="url(#filter2_d)">
+<path d="M75.8578 99.3807C73.4721 100.535 70.6219 100.047 68.75 98.1651C71.0564 100.483 75 98.8415 75 95.5631V4.43709C75 1.15852 71.0565 -0.483493 68.75 1.83492C70.6219 -0.0467614 73.4721 -0.534276 75.8578 0.618963L96.4583 10.5773C98.6229 11.6237 100 13.8246 100 16.2391V83.7616C100 86.1762 98.6229 88.3761 96.4583 89.4231L75.8578 99.3807Z" fill="white"/>
+</g>
+<g style="mix-blend-mode:overlay" opacity="0.25">
+<path style="mix-blend-mode:overlay" opacity="0.25" fill-rule="evenodd" clip-rule="evenodd" d="M70.8508 99.5723C72.4258 100.189 74.2218 100.15 75.8115 99.3807L96.4 89.4231C98.5635 88.3771 99.9386 86.1762 99.9386 83.7616V16.2391C99.9386 13.8247 98.5635 11.6239 96.4 10.5774L75.8115 0.618974C73.7252 -0.390085 71.2835 -0.142871 69.4525 1.19518C69.1909 1.38637 68.9418 1.59976 68.7079 1.83493L29.2941 37.9795L12.1261 24.88C10.528 23.6606 8.2926 23.7605 6.80833 25.1177L1.30198 30.1524C-0.51354 31.8126 -0.515625 34.6837 1.2975 36.3465L16.186 50L1.2975 63.6536C-0.515625 65.3164 -0.51354 68.1875 1.30198 69.8476L6.80833 74.8824C8.2926 76.2395 10.528 76.339 12.1261 75.1201L29.2941 62.0207L68.7079 98.1651C69.3315 98.7923 70.0635 99.2645 70.8508 99.5723ZM74.9542 27.1812L45.0481 50L74.9542 72.8188V27.1812Z" fill="url(#paint0_linear)"/>
+</g>
+</g>
+</g>
+</g>
+<defs>
+<filter id="filter0_d" x="-6.25" y="-4.16667" width="112.5" height="112.5" filterUnits="userSpaceOnUse" color-interpolation-filters="sRGB">
+<feFlood flood-opacity="0" result="BackgroundImageFix"/>
+<feColorMatrix in="SourceAlpha" type="matrix" values="0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 127 0"/>
+<feOffset dy="2.08333"/>
+<feGaussianBlur stdDeviation="3.125"/>
+<feColorMatrix type="matrix" values="0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0.15 0"/>
+<feBlend mode="normal" in2="BackgroundImageFix" result="effect1_dropShadow"/>
+<feBlend mode="normal" in="SourceGraphic" in2="effect1_dropShadow" result="shape"/>
+</filter>
+<filter id="filter1_d" x="-8.39436" y="15.6951" width="116.728" height="92.6376" filterUnits="userSpaceOnUse" color-interpolation-filters="sRGB">
+<feFlood flood-opacity="0" result="BackgroundImageFix"/>
+<feColorMatrix in="SourceAlpha" type="matrix" values="0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 127 0"/>
+<feOffset/>
+<feGaussianBlur stdDeviation="4.16667"/>
+<feColorMatrix type="matrix" values="0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0.25 0"/>
+<feBlend mode="overlay" in2="BackgroundImageFix" result="effect1_dropShadow"/>
+<feBlend mode="normal" in="SourceGraphic" in2="effect1_dropShadow" result="shape"/>
+</filter>
+<filter id="filter2_d" x="60.4167" y="-8.33346" width="47.9167" height="116.667" filterUnits="userSpaceOnUse" color-interpolation-filters="sRGB">
+<feFlood flood-opacity="0" result="BackgroundImageFix"/>
+<feColorMatrix in="SourceAlpha" type="matrix" values="0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 127 0"/>
+<feOffset/>
+<feGaussianBlur stdDeviation="4.16667"/>
+<feColorMatrix type="matrix" values="0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0.25 0"/>
+<feBlend mode="overlay" in2="BackgroundImageFix" result="effect1_dropShadow"/>
+<feBlend mode="normal" in="SourceGraphic" in2="effect1_dropShadow" result="shape"/>
+</filter>
+<linearGradient id="paint0_linear" x1="49.939" y1="-5.19792e-05" x2="49.939" y2="100.001" gradientUnits="userSpaceOnUse">
+<stop stop-color="white"/>
+<stop offset="1" stop-color="white" stop-opacity="0"/>
+</linearGradient>
+<clipPath id="clip0">
+<rect width="100" height="100" fill="white"/>
+</clipPath>
+</defs>
+</svg>
--- a/frontend/src/components/AgentControlBar.tsx
+++ b/frontend/src/components/AgentControlBar.tsx
@@ -6,7 +6,7 @@ import PlayIcon from "#/assets/play";
 import { generateAgentStateChangeEvent } from "#/services/agentStateService";
 import { RootState } from "#/store";
 import AgentState from "#/types/AgentState";
-import { useSocket } from "#/context/socket";
+import { useWsClient } from "#/context/ws-client-provider";

 const IgnoreTaskStateMap: Record<string, AgentState[]> = {
  [AgentState.PAUSED]: [
@@ -72,7 +72,7 @@ function ActionButton({
 }

 function AgentControlBar() {
-  const { send } = useSocket();
+  const { send } = useWsClient();
  const { curAgentState } = useSelector((state: RootState) => state.agent);

  const handleAction = (action: AgentState) => {
--- a/frontend/src/components/attach-image-label.tsx
+++ b/frontend/src/components/attach-image-label.tsx
@@ -1,4 +1,4 @@
-import Clip from "#/assets/clip.svg?react";
+import Clip from "#/icons/clip.svg?react";

 export function AttachImageLabel() {
  return (
--- a/frontend/src/components/chat-input.tsx
+++ b/frontend/src/components/chat-input.tsx
@@ -1,6 +1,6 @@
 import React from "react";
 import TextareaAutosize from "react-textarea-autosize";
-import ArrowSendIcon from "#/assets/arrow-send.svg?react";
+import ArrowSendIcon from "#/icons/arrow-send.svg?react";
 import { cn } from "#/utils/utils";

 interface ChatInputProps {
@@ -18,6 +18,7 @@ interface ChatInputProps {
  onBlur?: () => void;
  onImagePaste?: (files: File[]) => void;
  className?: React.HTMLAttributes<HTMLDivElement>["className"];
+  buttonClassName?: React.HTMLAttributes<HTMLButtonElement>["className"];
 }

 export function ChatInput({
@@ -35,6 +36,7 @@ export function ChatInput({
  onBlur,
  onImagePaste,
  className,
+  buttonClassName,
 }: ChatInputProps) {
  const textareaRef = React.useRef<HTMLTextAreaElement>(null);
  const [isDraggingOver, setIsDraggingOver] = React.useState(false);
@@ -100,7 +102,7 @@ export function ChatInput({
  return (
    <div
      data-testid="chat-input"
-      className="flex items-end justify-end grow gap-1 min-h-6"
+      className="flex items-end justify-end grow gap-1 min-h-6 w-full"
    >
      <TextareaAutosize
        ref={textareaRef}
@@ -128,7 +130,7 @@ export function ChatInput({
        )}
      />
      {showButton && (
-        <>
+        <div className={buttonClassName}>
          {button === "submit" && (
            <button
              aria-label="Send"
@@ -152,7 +154,7 @@ export function ChatInput({
              <div className="w-[10px] h-[10px] bg-white" />
            </button>
          )}
-        </>
+        </div>
      )}
    </div>
  );
--- a/frontend/src/components/chat-interface.tsx
+++ b/frontend/src/components/chat-interface.tsx
@@ -1,7 +1,7 @@
 import { useDispatch, useSelector } from "react-redux";
 import React from "react";
 import posthog from "posthog-js";
-import { useSocket } from "#/context/socket";
+import { useRouteLoaderData } from "@remix-run/react";
 import { convertImageToBase64 } from "#/utils/convert-image-to-base-64";
 import { ChatMessage } from "./chat-message";
 import { FeedbackActions } from "./feedback-actions";
@@ -21,18 +21,28 @@ import { ContinueButton } from "./continue-button";
 import { ScrollToBottomButton } from "./scroll-to-bottom-button";
 import { Suggestions } from "./suggestions";
 import { SUGGESTIONS } from "#/utils/suggestions";
-import BuildIt from "#/assets/build-it.svg?react";
+import BuildIt from "#/icons/build-it.svg?react";
+import {
+  useWsClient,
+  WsClientProviderStatus,
+} from "#/context/ws-client-provider";
+import OpenHands from "#/api/open-hands";
+import { clientLoader } from "#/routes/_oh";
+import { downloadWorkspace } from "#/utils/download-workspace";
+import { SuggestionItem } from "./suggestion-item";

 const isErrorMessage = (
  message: Message | ErrorMessage,
 ): message is ErrorMessage => "error" in message;

 export function ChatInterface() {
-  const { send } = useSocket();
+  const { send, status, isLoadingMessages } = useWsClient();
+
  const dispatch = useDispatch();
  const scrollRef = React.useRef<HTMLDivElement>(null);
  const { scrollDomToBottom, onChatBodyScroll, hitBottom } =
    useScrollToBottom(scrollRef);
+  const rootLoaderData = useRouteLoaderData<typeof clientLoader>("routes/_oh");

  const { messages } = useSelector((state: RootState) => state.chat);
  const { curAgentState } = useSelector((state: RootState) => state.agent);
@@ -42,6 +52,24 @@ export function ChatInterface() {
  >("positive");
  const [feedbackModalIsOpen, setFeedbackModalIsOpen] = React.useState(false);
  const [messageToSend, setMessageToSend] = React.useState<string | null>(null);
+  const [isDownloading, setIsDownloading] = React.useState(false);
+
+  React.useEffect(() => {
+    if (status === WsClientProviderStatus.ACTIVE) {
+      try {
+        OpenHands.getRuntimeId().then(({ runtime_id }) => {
+          // eslint-disable-next-line no-console
+          console.log(
+            "Runtime ID: %c%s",
+            "background: #444; color: #ffeb3b; font-weight: bold; padding: 2px 4px; border-radius: 4px;",
+            runtime_id,
+          );
+        });
+      } catch (e) {
+        console.warn("Runtime ID not available in this environment");
+      }
+    }
+  }, [status]);

  const handleSendMessage = async (content: string, files: File[]) => {
    posthog.capture("user_message_sent", {
@@ -72,6 +100,17 @@ export function ChatInterface() {
    setFeedbackPolarity(polarity);
  };

+  const handleDownloadWorkspace = async () => {
+    setIsDownloading(true);
+    try {
+      await downloadWorkspace();
+    } catch (error) {
+      // TODO: Handle error
+    } finally {
+      setIsDownloading(false);
+    }
+  };
+
  return (
    <div className="h-full flex flex-col justify-between">
      {messages.length === 0 && (
@@ -101,29 +140,64 @@ export function ChatInterface() {
        onScroll={(e) => onChatBodyScroll(e.currentTarget)}
        className="flex flex-col grow overflow-y-auto overflow-x-hidden px-4 pt-4 gap-2"
      >
-        {messages.map((message, index) =>
-          isErrorMessage(message) ? (
-            <ErrorMessage
-              key={index}
-              id={message.id}
-              message={message.message}
-            />
-          ) : (
-            <ChatMessage
-              key={index}
-              type={message.sender}
-              message={message.content}
-            >
-              {message.imageUrls.length > 0 && (
-                <ImageCarousel size="small" images={message.imageUrls} />
-              )}
-              {messages.length - 1 === index &&
-                message.sender === "assistant" &&
-                curAgentState === AgentState.AWAITING_USER_CONFIRMATION && (
-                  <ConfirmationButtons />
+        {isLoadingMessages && (
+          <div className="flex justify-center">
+            <div className="w-6 h-6 border-2 border-t-[4px] border-primary-500 rounded-full animate-spin" />
+          </div>
+        )}
+
+        {!isLoadingMessages &&
+          messages.map((message, index) =>
+            isErrorMessage(message) ? (
+              <ErrorMessage
+                key={index}
+                id={message.id}
+                message={message.message}
+              />
+            ) : (
+              <ChatMessage
+                key={index}
+                type={message.sender}
+                message={message.content}
+              >
+                {message.imageUrls.length > 0 && (
+                  <ImageCarousel size="small" images={message.imageUrls} />
                )}
-            </ChatMessage>
-          ),
+                {messages.length - 1 === index &&
+                  message.sender === "assistant" &&
+                  curAgentState === AgentState.AWAITING_USER_CONFIRMATION && (
+                    <ConfirmationButtons />
+                  )}
+              </ChatMessage>
+            ),
+          )}
+
+        {(curAgentState === AgentState.AWAITING_USER_INPUT ||
+          curAgentState === AgentState.FINISHED) && (
+          <div className="flex flex-col gap-2 mb-2">
+            {rootLoaderData?.ghToken ? (
+              <SuggestionItem
+                suggestion={{
+                  label: "Push to GitHub",
+                  value:
+                    "Please push the changes to GitHub and open a pull request.",
+                }}
+                onClick={(value) => {
+                  handleSendMessage(value, []);
+                }}
+              />
+            ) : (
+              <SuggestionItem
+                suggestion={{
+                  label: !isDownloading
+                    ? "Download .zip"
+                    : "Downloading, please wait...",
+                  value: "Download .zip",
+                }}
+                onClick={handleDownloadWorkspace}
+              />
+            )}
+          </div>
        )}
      </div>

--- a/frontend/src/components/chat/ConfirmationButtons.tsx
+++ b/frontend/src/components/chat/ConfirmationButtons.tsx
@@ -5,7 +5,7 @@ import RejectIcon from "#/assets/reject";
 import { I18nKey } from "#/i18n/declaration";
 import AgentState from "#/types/AgentState";
 import { generateAgentStateChangeEvent } from "#/services/agentStateService";
-import { useSocket } from "#/context/socket";
+import { useWsClient } from "#/context/ws-client-provider";

 interface ActionTooltipProps {
  type: "confirm" | "reject";
@@ -37,7 +37,7 @@ function ActionTooltip({ type, onClick }: ActionTooltipProps) {

 function ConfirmationButtons() {
  const { t } = useTranslation();
-  const { send } = useSocket();
+  const { send } = useWsClient();

  const handleStateChange = (state: AgentState) => {
    const event = generateAgentStateChangeEvent(state);
--- a/frontend/src/components/event-handler.tsx
+++ b/frontend/src/components/event-handler.tsx
@@ -0,0 +1,191 @@
+import React from "react";
+import {
+  useFetcher,
+  useLoaderData,
+  useRouteLoaderData,
+} from "@remix-run/react";
+import { useDispatch, useSelector } from "react-redux";
+import toast from "react-hot-toast";
+
+import posthog from "posthog-js";
+import {
+  useWsClient,
+  WsClientProviderStatus,
+} from "#/context/ws-client-provider";
+import { ErrorObservation } from "#/types/core/observations";
+import { addErrorMessage, addUserMessage } from "#/state/chatSlice";
+import {
+  getCloneRepoCommand,
+  getGitHubTokenCommand,
+} from "#/services/terminalService";
+import {
+  clearFiles,
+  clearInitialQuery,
+  clearSelectedRepository,
+  setImportedProjectZip,
+} from "#/state/initial-query-slice";
+import { clientLoader as appClientLoader } from "#/routes/_oh.app";
+import store, { RootState } from "#/store";
+import { createChatMessage } from "#/services/chatService";
+import { clientLoader as rootClientLoader } from "#/routes/_oh";
+import { isGitHubErrorReponse } from "#/api/github";
+import OpenHands from "#/api/open-hands";
+import { base64ToBlob } from "#/utils/base64-to-blob";
+import { setCurrentAgentState } from "#/state/agentSlice";
+import AgentState from "#/types/AgentState";
+import { getSettings } from "#/services/settings";
+import { generateAgentStateChangeEvent } from "#/services/agentStateService";
+
+interface ServerError {
+  error: boolean | string;
+  message: string;
+  [key: string]: unknown;
+}
+
+const isServerError = (data: object): data is ServerError => "error" in data;
+
+const isErrorObservation = (data: object): data is ErrorObservation =>
+  "observation" in data && data.observation === "error";
+
+export function EventHandler({ children }: React.PropsWithChildren) {
+  const { events, status, send } = useWsClient();
+  const statusRef = React.useRef<WsClientProviderStatus | null>(null);
+  const runtimeActive = status === WsClientProviderStatus.ACTIVE;
+  const fetcher = useFetcher();
+  const dispatch = useDispatch();
+  const { files, importedProjectZip, initialQuery } = useSelector(
+    (state: RootState) => state.initalQuery,
+  );
+  const { ghToken, repo } = useLoaderData<typeof appClientLoader>();
+
+  const sendInitialQuery = (query: string, base64Files: string[]) => {
+    const timestamp = new Date().toISOString();
+    send(createChatMessage(query, base64Files, timestamp));
+  };
+  const data = useRouteLoaderData<typeof rootClientLoader>("routes/_oh");
+  const userId = React.useMemo(() => {
+    if (data?.user && !isGitHubErrorReponse(data.user)) return data.user.id;
+    return null;
+  }, [data?.user]);
+  const userSettings = getSettings();
+
+  React.useEffect(() => {
+    if (!events.length) {
+      return;
+    }
+    const event = events[events.length - 1];
+    if (event.token) {
+      fetcher.submit({ token: event.token as string }, { method: "post" });
+      return;
+    }
+
+    if (isServerError(event)) {
+      if (event.error_code === 401) {
+        toast.error("Session expired.");
+        fetcher.submit({}, { method: "POST", action: "/end-session" });
+        return;
+      }
+
+      if (typeof event.error === "string") {
+        toast.error(event.error);
+      } else {
+        toast.error(event.message);
+      }
+      return;
+    }
+
+    if (event.type === "error") {
+      const message: string = `${event.message}`;
+      if (message.startsWith("Agent reached maximum")) {
+        // We set the agent state to paused here - if the user clicks resume, it auto updates the max iterations
+        send(generateAgentStateChangeEvent(AgentState.PAUSED));
+      }
+    }
+
+    if (isErrorObservation(event)) {
+      dispatch(
+        addErrorMessage({
+          id: event.extras?.error_id,
+          message: event.message,
+        }),
+      );
+    }
+  }, [events.length]);
+
+  React.useEffect(() => {
+    if (statusRef.current === status) {
+      return; // This is a check because of strict mode - if the status did not change, don't do anything
+    }
+    statusRef.current = status;
+
+    if (status === WsClientProviderStatus.ACTIVE) {
+      let additionalInfo = "";
+      if (ghToken && repo) {
+        send(getCloneRepoCommand(ghToken, repo));
+        additionalInfo = `Repository ${repo} has been cloned to /workspace. Please check the /workspace for files.`;
+        dispatch(clearSelectedRepository()); // reset selected repository; maybe better to move this to '/'?
+      }
+      // if there's an uploaded project zip, add it to the chat
+      else if (importedProjectZip) {
+        additionalInfo = `Files have been uploaded. Please check the /workspace for files.`;
+      }
+
+      if (initialQuery) {
+        if (additionalInfo) {
+          sendInitialQuery(`${initialQuery}\n\n[${additionalInfo}]`, files);
+        } else {
+          sendInitialQuery(initialQuery, files);
+        }
+        dispatch(clearFiles()); // reset selected files
+        dispatch(clearInitialQuery()); // reset initial query
+      }
+    }
+
+    if (status === WsClientProviderStatus.OPENING && initialQuery) {
+      dispatch(
+        addUserMessage({
+          content: initialQuery,
+          imageUrls: files,
+          timestamp: new Date().toISOString(),
+        }),
+      );
+    }
+
+    if (status === WsClientProviderStatus.STOPPED) {
+      store.dispatch(setCurrentAgentState(AgentState.STOPPED));
+    }
+  }, [status]);
+
+  React.useEffect(() => {
+    if (runtimeActive && userId && ghToken) {
+      // Export if the user valid, this could happen mid-session so it is handled here
+      send(getGitHubTokenCommand(ghToken));
+    }
+  }, [userId, ghToken, runtimeActive]);
+
+  React.useEffect(() => {
+    (async () => {
+      if (runtimeActive && importedProjectZip) {
+        // upload files action
+        try {
+          const blob = base64ToBlob(importedProjectZip);
+          const file = new File([blob], "imported-project.zip", {
+            type: blob.type,
+          });
+          await OpenHands.uploadFiles([file]);
+          dispatch(setImportedProjectZip(null));
+        } catch (error) {
+          toast.error("Failed to upload project files.");
+        }
+      }
+    })();
+  }, [runtimeActive, importedProjectZip]);
+
+  React.useEffect(() => {
+    if (userSettings.LLM_API_KEY) {
+      posthog.capture("user_activated");
+    }
+  }, [userSettings.LLM_API_KEY]);
+
+  return children;
+}
--- a/frontend/src/components/file-explorer/FileExplorer.tsx
+++ b/frontend/src/components/file-explorer/FileExplorer.tsx
@@ -12,6 +12,7 @@ import { useTranslation } from "react-i18next";
 import { twMerge } from "tailwind-merge";
 import AgentState from "#/types/AgentState";
 import { setRefreshID } from "#/state/codeSlice";
+import { addAssistantMessage } from "#/state/chatSlice";
 import IconButton from "../IconButton";
 import ExplorerTree from "./ExplorerTree";
 import toast from "#/utils/toast";
@@ -20,6 +21,7 @@ import { I18nKey } from "#/i18n/declaration";
 import OpenHands from "#/api/open-hands";
 import { useFiles } from "#/context/files";
 import { isOpenHandsErrorResponse } from "#/api/open-hands.utils";
+import VSCodeIcon from "#/assets/vscode-alt.svg?react";

 interface ExplorerActionsProps {
  onRefresh: () => void;
@@ -168,6 +170,35 @@ function FileExplorer({ error, isOpen, onToggle }: FileExplorerProps) {
    }
  };

+  const handleVSCodeClick = async (e: React.MouseEvent) => {
+    e.preventDefault();
+    try {
+      const response = await OpenHands.getVSCodeUrl();
+      if (response.vscode_url) {
+        dispatch(
+          addAssistantMessage(
+            "You opened VS Code. Please inform the agent of any changes you made to the workspace or environment. To avoid conflicts, it's best to pause the agent before making any changes.",
+          ),
+        );
+        window.open(response.vscode_url, "_blank");
+      } else {
+        toast.error(
+          `open-vscode-error-${new Date().getTime()}`,
+          t(I18nKey.EXPLORER$VSCODE_SWITCHING_ERROR_MESSAGE, {
+            error: response.error,
+          }),
+        );
+      }
+    } catch (exp_error) {
+      toast.error(
+        `open-vscode-error-${new Date().getTime()}`,
+        t(I18nKey.EXPLORER$VSCODE_SWITCHING_ERROR_MESSAGE, {
+          error: String(exp_error),
+        }),
+      );
+    }
+  };
+
  React.useEffect(() => {
    refreshWorkspace();
  }, [curAgentState]);
@@ -210,7 +241,7 @@ function FileExplorer({ error, isOpen, onToggle }: FileExplorerProps) {
          !isOpen ? "w-12" : "w-60",
        )}
      >
-        <div className="flex flex-col relative h-full px-3 py-2">
+        <div className="flex flex-col relative h-full px-3 py-2 overflow-hidden">
          <div className="sticky top-0 bg-neutral-800">
            <div
              className={twMerge(
@@ -232,7 +263,7 @@ function FileExplorer({ error, isOpen, onToggle }: FileExplorerProps) {
            </div>
          </div>
          {!error && (
-            <div className="overflow-auto flex-grow">
+            <div className="overflow-auto flex-grow min-h-0">
              <div style={{ display: !isOpen ? "none" : "block" }}>
                <ExplorerTree files={paths} />
              </div>
@@ -243,6 +274,27 @@ function FileExplorer({ error, isOpen, onToggle }: FileExplorerProps) {
              <p className="text-neutral-300 text-sm">{error}</p>
            </div>
          )}
+          {isOpen && (
+            <button
+              type="button"
+              onClick={handleVSCodeClick}
+              disabled={
+                curAgentState === AgentState.INIT ||
+                curAgentState === AgentState.LOADING
+              }
+              className={twMerge(
+                "mt-auto mb-2 w-full h-10 text-white rounded flex items-center justify-center gap-2 transition-colors",
+                curAgentState === AgentState.INIT ||
+                  curAgentState === AgentState.LOADING
+                  ? "bg-neutral-600 cursor-not-allowed"
+                  : "bg-[#4465DB] hover:bg-[#3451C7]",
+              )}
+              aria-label="Open in VS Code"
+            >
+              <VSCodeIcon width={20} height={20} />
+              Open in VS Code
+            </button>
+          )}
        </div>
        <input
          data-testid="file-input"
--- a/frontend/src/components/github-repositories-suggestion-box.tsx
+++ b/frontend/src/components/github-repositories-suggestion-box.tsx
@@ -10,32 +10,8 @@ import { GitHubRepositorySelector } from "#/routes/_oh._index/github-repo-select
 import ModalButton from "./buttons/ModalButton";
 import GitHubLogo from "#/assets/branding/github-logo.svg?react";

-interface GitHubAuthProps {
-  onConnectToGitHub: () => void;
-  repositories: GitHubRepository[];
-  isLoggedIn: boolean;
-}
-
-function GitHubAuth({
-  onConnectToGitHub,
-  repositories,
-  isLoggedIn,
-}: GitHubAuthProps) {
-  if (isLoggedIn) {
-    return <GitHubRepositorySelector repositories={repositories} />;
-  }
-
-  return (
-    <ModalButton
-      text="Connect to GitHub"
-      icon={<GitHubLogo width={20} height={20} />}
-      className="bg-[#791B80] w-full"
-      onClick={onConnectToGitHub}
-    />
-  );
-}
-
 interface GitHubRepositoriesSuggestionBoxProps {
+  handleSubmit: () => void;
  repositories: Awaited<
    ReturnType<typeof retrieveAllGitHubUserRepositories>
  > | null;
@@ -44,6 +20,7 @@ interface GitHubRepositoriesSuggestionBoxProps {
 }

 export function GitHubRepositoriesSuggestionBox({
+  handleSubmit,
  repositories,
  gitHubAuthUrl,
  user,
@@ -70,16 +47,26 @@ export function GitHubRepositoriesSuggestionBox({
    );
  }

+  const isLoggedIn = !!user && !isGitHubErrorReponse(user);
+
  return (
    <>
      <SuggestionBox
        title="Open a Repo"
        content={
-          <GitHubAuth
-            isLoggedIn={!!user && !isGitHubErrorReponse(user)}
-            repositories={repositories || []}
-            onConnectToGitHub={handleConnectToGitHub}
-          />
+          isLoggedIn ? (
+            <GitHubRepositorySelector
+              onSelect={handleSubmit}
+              repositories={repositories || []}
+            />
+          ) : (
+            <ModalButton
+              text="Connect to GitHub"
+              icon={<GitHubLogo width={20} height={20} />}
+              className="bg-[#791B80] w-full"
+              onClick={handleConnectToGitHub}
+            />
+          )
        }
      />
      {connectToGitHubModalOpen && (
--- a/frontend/src/components/image-preview.tsx
+++ b/frontend/src/components/image-preview.tsx
@@ -1,4 +1,4 @@
-import CloseIcon from "#/assets/close.svg?react";
+import CloseIcon from "#/icons/close.svg?react";
 import { cn } from "#/utils/utils";

 interface ImagePreviewProps {
--- a/frontend/src/components/interactive-chat-box.tsx
+++ b/frontend/src/components/interactive-chat-box.tsx
@@ -56,14 +56,9 @@ export function InteractiveChatBox({
      <div
        className={cn(
          "flex items-end gap-1",
-          "bg-neutral-700 border border-neutral-600 rounded-lg px-2 py-[10px]",
+          "bg-neutral-700 border border-neutral-600 rounded-lg px-2",
          "transition-colors duration-200",
          "hover:border-neutral-500 focus-within:border-neutral-500",
-          "group relative",
-          "before:pointer-events-none before:absolute before:inset-0 before:rounded-lg before:transition-colors",
-          "before:border-2 before:border-dashed before:border-transparent",
-          "[&:has(*:focus-within)]:before:border-neutral-500/50",
-          "[&:has(*[data-dragging-over='true'])]:before:border-neutral-500/50",
        )}
      >
        <UploadImageInput onUpload={handleUpload} />
@@ -76,6 +71,8 @@ export function InteractiveChatBox({
          onStop={onStop}
          value={value}
          onImagePaste={handleUpload}
+          className="py-[10px]"
+          buttonClassName="py-[10px]"
        />
      </div>
    </div>
--- a/frontend/src/components/markdown/list.tsx
+++ b/frontend/src/components/markdown/list.tsx
@@ -4,8 +4,8 @@ import { ExtraProps } from "react-markdown";
 // Custom component to render <ul> in markdown
 export function ul({
  children,
-}: React.ClassAttributes<HTMLElement> &
-  React.HTMLAttributes<HTMLElement> &
+}: React.ClassAttributes<HTMLUListElement> &
+  React.HTMLAttributes<HTMLUListElement> &
  ExtraProps) {
  return <ul className="list-disc ml-5 pl-2 whitespace-normal">{children}</ul>;
 }
@@ -13,10 +13,13 @@ export function ul({
 // Custom component to render <ol> in markdown
 export function ol({
  children,
-}: React.ClassAttributes<HTMLElement> &
-  React.HTMLAttributes<HTMLElement> &
+  start,
+}: React.ClassAttributes<HTMLOListElement> &
+  React.OlHTMLAttributes<HTMLOListElement> &
  ExtraProps) {
  return (
-    <ol className="list-decimal ml-5 pl-2 whitespace-normal">{children}</ol>
+    <ol className="list-decimal ml-5 pl-2 whitespace-normal" start={start}>
+      {children}
+    </ol>
  );
 }
--- a/frontend/src/components/modals/AccountSettingsModal.tsx
+++ b/frontend/src/components/modals/AccountSettingsModal.tsx
@@ -1,7 +1,10 @@
 import { useFetcher, useRouteLoaderData } from "@remix-run/react";
 import React from "react";
 import { useTranslation } from "react-i18next";
-import { BaseModalTitle } from "./confirmation-modals/BaseModal";
+import {
+  BaseModalDescription,
+  BaseModalTitle,
+} from "./confirmation-modals/BaseModal";
 import ModalBody from "./ModalBody";
 import ModalButton from "../buttons/ModalButton";
 import FormFieldset from "../form/FormFieldset";
@@ -87,6 +90,17 @@ function AccountSettingsModal({
            type="password"
            defaultValue={data?.ghToken ?? ""}
          />
+          <BaseModalDescription>
+            {t(I18nKey.CONNECT_TO_GITHUB_MODAL$GET_YOUR_TOKEN)}{" "}
+            <a
+              href="https://github.com/settings/tokens/new?description=openhands-app&scopes=repo,user,workflow"
+              target="_blank"
+              rel="noreferrer noopener"
+              className="text-[#791B80] underline"
+            >
+              {t(I18nKey.CONNECT_TO_GITHUB_MODAL$HERE)}
+            </a>
+          </BaseModalDescription>
          {gitHubError && (
            <p className="text-danger text-xs">
              {t(I18nKey.ACCOUNT_SETTINGS_MODAL$GITHUB_TOKEN_INVALID)}
--- a/frontend/src/components/modals/LoadingProject.tsx
+++ b/frontend/src/components/modals/LoadingProject.tsx
@@ -1,5 +1,5 @@
 import { useTranslation } from "react-i18next";
-import LoadingSpinnerOuter from "#/assets/loading-outer.svg?react";
+import LoadingSpinnerOuter from "#/icons/loading-outer.svg?react";
 import { cn } from "#/utils/utils";
 import ModalBody from "./ModalBody";
 import { I18nKey } from "#/i18n/declaration";
--- a/frontend/src/components/project-menu/ProjectMenuCard.tsx
+++ b/frontend/src/components/project-menu/ProjectMenuCard.tsx
@@ -2,17 +2,17 @@ import React from "react";
 import { useDispatch } from "react-redux";
 import toast from "react-hot-toast";
 import posthog from "posthog-js";
-import EllipsisH from "#/assets/ellipsis-h.svg?react";
+import EllipsisH from "#/icons/ellipsis-h.svg?react";
 import { ModalBackdrop } from "../modals/modal-backdrop";
 import { ConnectToGitHubModal } from "../modals/connect-to-github-modal";
 import { addUserMessage } from "#/state/chatSlice";
-import { useSocket } from "#/context/socket";
 import { createChatMessage } from "#/services/chatService";
 import { ProjectMenuCardContextMenu } from "./project.menu-card-context-menu";
 import { ProjectMenuDetailsPlaceholder } from "./project-menu-details-placeholder";
 import { ProjectMenuDetails } from "./project-menu-details";
 import { downloadWorkspace } from "#/utils/download-workspace";
 import { LoadingSpinner } from "../modals/LoadingProject";
+import { useWsClient } from "#/context/ws-client-provider";

 interface ProjectMenuCardProps {
  isConnectedToGitHub: boolean;
@@ -27,7 +27,7 @@ export function ProjectMenuCard({
  isConnectedToGitHub,
  githubData,
 }: ProjectMenuCardProps) {
-  const { send } = useSocket();
+  const { send } = useWsClient();
  const dispatch = useDispatch();

  const [contextMenuIsOpen, setContextMenuIsOpen] = React.useState(false);
--- a/frontend/src/components/project-menu/project-menu-details-placeholder.tsx
+++ b/frontend/src/components/project-menu/project-menu-details-placeholder.tsx
@@ -1,6 +1,6 @@
 import { useTranslation } from "react-i18next";
 import { cn } from "#/utils/utils";
-import CloudConnection from "#/assets/cloud-connection.svg?react";
+import CloudConnection from "#/icons/cloud-connection.svg?react";
 import { I18nKey } from "#/i18n/declaration";

 interface ProjectMenuDetailsPlaceholderProps {
--- a/frontend/src/components/project-menu/project-menu-details.tsx
+++ b/frontend/src/components/project-menu/project-menu-details.tsx
@@ -1,5 +1,5 @@
 import { useTranslation } from "react-i18next";
-import ExternalLinkIcon from "#/assets/external-link.svg?react";
+import ExternalLinkIcon from "#/icons/external-link.svg?react";
 import { formatTimeDelta } from "#/utils/format-time-delta";
 import { I18nKey } from "#/i18n/declaration";

--- a/frontend/src/components/scroll-to-bottom-button.tsx
+++ b/frontend/src/components/scroll-to-bottom-button.tsx
@@ -1,4 +1,4 @@
-import ArrowSendIcon from "#/assets/arrow-send.svg?react";
+import ArrowSendIcon from "#/icons/arrow-send.svg?react";

 interface ScrollToBottomButtonProps {
  onClick: () => void;
--- a/frontend/src/components/suggestion-bubble.tsx
+++ b/frontend/src/components/suggestion-bubble.tsx
@@ -1,5 +1,5 @@
-import Lightbulb from "#/assets/lightbulb.svg?react";
-import Refresh from "#/assets/refresh.svg?react";
+import Lightbulb from "#/icons/lightbulb.svg?react";
+import Refresh from "#/icons/refresh.svg?react";

 interface SuggestionBubbleProps {
  suggestion: string;
--- a/frontend/src/components/suggestion-item.tsx
+++ b/frontend/src/components/suggestion-item.tsx
@@ -7,12 +7,12 @@ interface SuggestionItemProps {

 export function SuggestionItem({ suggestion, onClick }: SuggestionItemProps) {
  return (
-    <li className="border border-neutral-600 rounded-xl hover:bg-neutral-700">
+    <li className="list-none border border-neutral-600 rounded-xl hover:bg-neutral-700">
      <button
        type="button"
        data-testid="suggestion"
        onClick={() => onClick(suggestion.value)}
-        className="text-[16px] leading-6 -tracking-[0.01em] text-center w-full p-4 font-semibold"
+        className="text-[16px] leading-6 -tracking-[0.01em] text-center w-full p-3 font-semibold"
      >
        {suggestion.label}
      </button>
--- a/frontend/src/components/upload-image-input.tsx
+++ b/frontend/src/components/upload-image-input.tsx
@@ -1,4 +1,4 @@
-import Clip from "#/assets/clip.svg?react";
+import Clip from "#/icons/clip.svg?react";

 interface UploadImageInputProps {
  onUpload: (files: File[]) => void;
@@ -11,7 +11,7 @@ export function UploadImageInput({ onUpload, label }: UploadImageInputProps) {
  };

  return (
-    <label className="cursor-pointer">
+    <label className="cursor-pointer py-[10px]">
      {label || <Clip data-testid="default-label" width={24} height={24} />}
      <input
        data-testid="upload-image-input"
--- a/frontend/src/components/user-avatar.tsx
+++ b/frontend/src/components/user-avatar.tsx
@@ -1,5 +1,5 @@
 import { LoadingSpinner } from "./modals/LoadingProject";
-import DefaultUserAvatar from "#/assets/default-user.svg?react";
+import DefaultUserAvatar from "#/icons/default-user.svg?react";
 import { cn } from "#/utils/utils";

 interface UserAvatarProps {
--- a/frontend/src/context/socket.tsx
+++ b/frontend/src/context/socket.tsx
@@ -1,146 +0,0 @@
-import React from "react";
-import { Data } from "ws";
-import posthog from "posthog-js";
-import EventLogger from "#/utils/event-logger";
-
-interface WebSocketClientOptions {
-  token: string | null;
-  onOpen?: (event: Event) => void;
-  onMessage?: (event: MessageEvent<Data>) => void;
-  onError?: (event: Event) => void;
-  onClose?: (event: Event) => void;
-}
-
-interface WebSocketContextType {
-  send: (data: string | ArrayBufferLike | Blob | ArrayBufferView) => void;
-  start: (options?: WebSocketClientOptions) => void;
-  stop: () => void;
-  setRuntimeIsInitialized: () => void;
-  runtimeActive: boolean;
-  isConnected: boolean;
-  events: Record<string, unknown>[];
-}
-
-const SocketContext = React.createContext<WebSocketContextType | undefined>(
-  undefined,
-);
-
-interface SocketProviderProps {
-  children: React.ReactNode;
-}
-
-function SocketProvider({ children }: SocketProviderProps) {
-  const wsRef = React.useRef<WebSocket | null>(null);
-  const [isConnected, setIsConnected] = React.useState(false);
-  const [runtimeActive, setRuntimeActive] = React.useState(false);
-  const [events, setEvents] = React.useState<Record<string, unknown>[]>([]);
-
-  const setRuntimeIsInitialized = () => {
-    setRuntimeActive(true);
-  };
-
-  const start = React.useCallback((options?: WebSocketClientOptions): void => {
-    if (wsRef.current) {
-      EventLogger.warning(
-        "WebSocket connection is already established, but a new one is starting anyways.",
-      );
-    }
-
-    const baseUrl =
-      import.meta.env.VITE_BACKEND_BASE_URL || window?.location.host;
-    const protocol = window.location.protocol === "https:" ? "wss:" : "ws:";
-    const sessionToken = options?.token || "NO_JWT"; // not allowed to be empty or duplicated
-    const ghToken = localStorage.getItem("ghToken") || "NO_GITHUB";
-
-    const ws = new WebSocket(`${protocol}//${baseUrl}/ws`, [
-      "openhands",
-      sessionToken,
-      ghToken,
-    ]);
-
-    ws.addEventListener("open", (event) => {
-      posthog.capture("socket_opened");
-      setIsConnected(true);
-      options?.onOpen?.(event);
-    });
-
-    ws.addEventListener("message", (event) => {
-      EventLogger.message(event);
-
-      setEvents((prevEvents) => [...prevEvents, JSON.parse(event.data)]);
-      options?.onMessage?.(event);
-    });
-
-    ws.addEventListener("error", (event) => {
-      posthog.capture("socket_error");
-      EventLogger.event(event, "SOCKET ERROR");
-      options?.onError?.(event);
-    });
-
-    ws.addEventListener("close", (event) => {
-      posthog.capture("socket_closed");
-      EventLogger.event(event, "SOCKET CLOSE");
-
-      setIsConnected(false);
-      setRuntimeActive(false);
-      wsRef.current = null;
-      options?.onClose?.(event);
-    });
-
-    wsRef.current = ws;
-  }, []);
-
-  const stop = React.useCallback((): void => {
-    if (wsRef.current) {
-      wsRef.current.close();
-      wsRef.current = null;
-    }
-  }, []);
-
-  const send = React.useCallback(
-    (data: string | ArrayBufferLike | Blob | ArrayBufferView) => {
-      if (!wsRef.current) {
-        EventLogger.error("WebSocket is not connected.");
-        return;
-      }
-      setEvents((prevEvents) => [...prevEvents, JSON.parse(data.toString())]);
-      wsRef.current.send(data);
-    },
-    [],
-  );
-
-  const value = React.useMemo(
-    () => ({
-      send,
-      start,
-      stop,
-      setRuntimeIsInitialized,
-      runtimeActive,
-      isConnected,
-      events,
-    }),
-    [
-      send,
-      start,
-      stop,
-      setRuntimeIsInitialized,
-      runtimeActive,
-      isConnected,
-      events,
-    ],
-  );
-
-  return (
-    <SocketContext.Provider value={value}>{children}</SocketContext.Provider>
-  );
-}
-
-function useSocket() {
-  const context = React.useContext(SocketContext);
-  if (context === undefined) {
-    throw new Error("useSocket must be used within a SocketProvider");
-  }
-  return context;
-}
-
-export { SocketProvider, useSocket };
--- a/frontend/src/context/ws-client-provider.tsx
+++ b/frontend/src/context/ws-client-provider.tsx
@@ -0,0 +1,208 @@
+import posthog from "posthog-js";
+import React from "react";
+import { Settings } from "#/services/settings";
+import ActionType from "#/types/ActionType";
+import EventLogger from "#/utils/event-logger";
+import AgentState from "#/types/AgentState";
+import { handleAssistantMessage } from "#/services/actions";
+import { useRate } from "#/utils/use-rate";
+
+const isOpenHandsMessage = (event: Record<string, unknown>) =>
+  event.action === "message";
+
+const RECONNECT_RETRIES = 5;
+
+export enum WsClientProviderStatus {
+  STOPPED,
+  OPENING,
+  ACTIVE,
+  ERROR,
+}
+
+interface UseWsClient {
+  status: WsClientProviderStatus;
+  isLoadingMessages: boolean;
+  events: Record<string, unknown>[];
+  send: (event: Record<string, unknown>) => void;
+}
+
+const WsClientContext = React.createContext<UseWsClient>({
+  status: WsClientProviderStatus.STOPPED,
+  isLoadingMessages: true,
+  events: [],
+  send: () => {
+    throw new Error("not connected");
+  },
+});
+
+interface WsClientProviderProps {
+  enabled: boolean;
+  token: string | null;
+  ghToken: string | null;
+  settings: Settings | null;
+}
+
+export function WsClientProvider({
+  enabled,
+  token,
+  ghToken,
+  settings,
+  children,
+}: React.PropsWithChildren<WsClientProviderProps>) {
+  const wsRef = React.useRef<WebSocket | null>(null);
+  const tokenRef = React.useRef<string | null>(token);
+  const ghTokenRef = React.useRef<string | null>(ghToken);
+  const closeRef = React.useRef<ReturnType<typeof setTimeout> | null>(null);
+  const [status, setStatus] = React.useState(WsClientProviderStatus.STOPPED);
+  const [events, setEvents] = React.useState<Record<string, unknown>[]>([]);
+  const [retryCount, setRetryCount] = React.useState(RECONNECT_RETRIES);
+
+  const messageRateHandler = useRate({ threshold: 500 });
+
+  function send(event: Record<string, unknown>) {
+    if (!wsRef.current) {
+      EventLogger.error("WebSocket is not connected.");
+      return;
+    }
+    wsRef.current.send(JSON.stringify(event));
+  }
+
+  function handleOpen() {
+    setRetryCount(RECONNECT_RETRIES);
+    setStatus(WsClientProviderStatus.OPENING);
+    const initEvent = {
+      action: ActionType.INIT,
+      args: settings,
+    };
+    send(initEvent);
+  }
+
+  function handleMessage(messageEvent: MessageEvent) {
+    const event = JSON.parse(messageEvent.data);
+    if (isOpenHandsMessage(event)) {
+      messageRateHandler.record(new Date().getTime());
+    }
+    setEvents((prevEvents) => [...prevEvents, event]);
+    if (event.extras?.agent_state === AgentState.INIT) {
+      setStatus(WsClientProviderStatus.ACTIVE);
+    }
+    if (
+      status !== WsClientProviderStatus.ACTIVE &&
+      event?.observation === "error"
+    ) {
+      setStatus(WsClientProviderStatus.ERROR);
+    }
+
+    handleAssistantMessage(event);
+  }
+
+  function handleClose() {
+    if (retryCount) {
+      setTimeout(() => {
+        setRetryCount(retryCount - 1);
+      }, 1000);
+    } else {
+      setStatus(WsClientProviderStatus.STOPPED);
+      setEvents([]);
+    }
+    wsRef.current = null;
+  }
+
+  function handleError(event: Event) {
+    posthog.capture("socket_error");
+    EventLogger.event(event, "SOCKET ERROR");
+    setStatus(WsClientProviderStatus.ERROR);
+  }
+
+  // Connect websocket
+  React.useEffect(() => {
+    let ws = wsRef.current;
+
+    // If disabled close any existing websockets...
+    if (!enabled || !retryCount) {
+      if (ws) {
+        ws.close();
+      }
+      wsRef.current = null;
+      return () => {};
+    }
+
+    // If there is no websocket or the tokens have changed or the current websocket is closed,
+    // create a new one
+    if (
+      !ws ||
+      (tokenRef.current && token !== tokenRef.current) ||
+      ghToken !== ghTokenRef.current ||
+      ws.readyState === WebSocket.CLOSED ||
+      ws.readyState === WebSocket.CLOSING
+    ) {
+      ws?.close();
+      const baseUrl =
+        import.meta.env.VITE_BACKEND_BASE_URL || window?.location.host;
+      const protocol = window.location.protocol === "https:" ? "wss:" : "ws:";
+      let wsUrl = `${protocol}//${baseUrl}/ws`;
+      if (events.length) {
+        wsUrl += `?latest_event_id=${events[events.length - 1].id}`;
+      }
+      ws = new WebSocket(wsUrl, [
+        "openhands",
+        token || "NO_JWT",
+        ghToken || "NO_GITHUB",
+      ]);
+    }
+    ws.addEventListener("open", handleOpen);
+    ws.addEventListener("message", handleMessage);
+    ws.addEventListener("error", handleError);
+    ws.addEventListener("close", handleClose);
+    wsRef.current = ws;
+    tokenRef.current = token;
+    ghTokenRef.current = ghToken;
+
+    return () => {
+      ws.removeEventListener("open", handleOpen);
+      ws.removeEventListener("message", handleMessage);
+      ws.removeEventListener("error", handleError);
+      ws.removeEventListener("close", handleClose);
+    };
+  }, [enabled, token, ghToken, retryCount]);
+
+  // Strict mode mounts and unmounts each component twice, so we have to wait in the destructor
+  // before actually closing the socket and cancel the operation if the component gets remounted.
+  React.useEffect(() => {
+    const timeout = closeRef.current;
+    if (timeout != null) {
+      clearTimeout(timeout);
+    }
+
+    return () => {
+      closeRef.current = setTimeout(() => {
+        const ws = wsRef.current;
+        if (ws) {
+          ws.removeEventListener("close", handleClose);
+          ws.close();
+        }
+      }, 100);
+    };
+  }, []);
+
+  const value = React.useMemo<UseWsClient>(
+    () => ({
+      status,
+      isLoadingMessages: messageRateHandler.isUnderThreshold,
+      events,
+      send,
+    }),
+    [status, messageRateHandler.isUnderThreshold, events],
+  );
+
+  return (
+    <WsClientContext.Provider value={value}>
+      {children}
+    </WsClientContext.Provider>
+  );
+}
+
+export function useWsClient() {
+  const context = React.useContext(WsClientContext);
+  return context;
+}
--- a/frontend/src/entry.client.tsx
+++ b/frontend/src/entry.client.tsx
@@ -10,18 +10,28 @@ import React, { startTransition, StrictMode } from "react";
 import { hydrateRoot } from "react-dom/client";
 import { Provider } from "react-redux";
 import posthog from "posthog-js";
-import { SocketProvider } from "./context/socket";
 import "./i18n";
 import store from "./store";
+import OpenHands from "./api/open-hands";

 function PosthogInit() {
+  const [key, setKey] = React.useState<string | null>(null);
+
  React.useEffect(() => {
-    posthog.init("phc_3ESMmY9SgqEAGBB6sMGK5ayYHkeUuknH2vP6FmWH9RA", {
-      api_host: "https://us.i.posthog.com",
-      person_profiles: "identified_only",
+    OpenHands.getConfig().then((config) => {
+      setKey(config.POSTHOG_CLIENT_KEY);
    });
  }, []);

+  React.useEffect(() => {
+    if (key) {
+      posthog.init(key, {
+        api_host: "https://us.i.posthog.com",
+        person_profiles: "identified_only",
+      });
+    }
+  }, [key]);
+
  return null;
 }

@@ -43,12 +53,10 @@ prepareApp().then(() =>
    hydrateRoot(
      document,
      <StrictMode>
-        <SocketProvider>
-          <Provider store={store}>
-            <RemixBrowser />
-            <PosthogInit />
-          </Provider>
-        </SocketProvider>
+        <Provider store={store}>
+          <RemixBrowser />
+          <PosthogInit />
+        </Provider>
      </StrictMode>,
    );
  }),
--- a/frontend/src/hooks/useTerminal.ts
+++ b/frontend/src/hooks/useTerminal.ts
@@ -4,7 +4,7 @@ import React from "react";
 import { Command } from "#/state/commandSlice";
 import { getTerminalCommand } from "#/services/terminalService";
 import { parseTerminalOutput } from "#/utils/parseTerminalOutput";
-import { useSocket } from "#/context/socket";
+import { useWsClient } from "#/context/ws-client-provider";

 /*
  NOTE: Tests for this hook are indirectly covered by the tests for the XTermTerminal component.
@@ -15,7 +15,7 @@ export const useTerminal = (
  commands: Command[] = [],
  secrets: string[] = [],
 ) => {
-  const { send } = useSocket();
+  const { send } = useWsClient();
  const terminal = React.useRef<Terminal | null>(null);
  const fitAddon = React.useRef<FitAddon | null>(null);
  const ref = React.useRef<HTMLDivElement>(null);
--- a/frontend/src/i18n/translation.json
+++ b/frontend/src/i18n/translation.json
@@ -535,7 +535,8 @@
    "pt": "Socket não inicializado",
    "ko-KR": "소켓이 초기화되지 않았습니다",
    "ar": "لم يتم تهيئة Socket",
-    "tr": "Soket başlatılmadı"
+    "tr": "Soket başlatılmadı",
+    "no": "Socket ikke initialisert"
  },
  "EXPLORER$UPLOAD_ERROR_MESSAGE": {
    "en": "Error uploading file",
@@ -548,7 +549,8 @@
    "pt": "Erro ao fazer upload do arquivo",
    "ko-KR": "파일 업로드 중 오류 발생",
    "ar": "خطأ في تحميل الملف",
-    "tr": "Dosya yüklenirken hata oluştu"
+    "tr": "Dosya yüklenirken hata oluştu",
+    "no": "Feil ved opplasting av fil"
  },
  "EXPLORER$LABEL_DROP_FILES": {
    "en": "Drop files here",
@@ -557,6 +559,7 @@
    "zh-TW": "將檔案拖曳至此",
    "es": "Suelta los archivos aquí",
    "fr": "Déposez les fichiers ici",
+    "no": "Slipp filer her",
    "it": "Trascina i file qui",
    "pt": "Solte os arquivos aqui",
    "ko-KR": "파일을 여기에 놓으세요",
@@ -574,7 +577,8 @@
    "pt": "Espaço de trabalho",
    "ko-KR": "작업 공간",
    "ar": "مساحة العمل",
-    "tr": "Çalışma alanı"
+    "tr": "Çalışma alanı",
+    "no": "Arbeidsområde"
  },
  "EXPLORER$EMPTY_WORKSPACE_MESSAGE": {
    "en": "No files in workspace",
@@ -587,7 +591,8 @@
    "pt": "Nenhum arquivo no espaço de trabalho",
    "ko-KR": "작업 공간에 파일이 없습니다",
    "ar": "لا توجد ملفات في مساحة العمل",
-    "tr": "Çalışma alanında dosya yok"
+    "tr": "Çalışma alanında dosya yok",
+    "no": "Ingen filer i arbeidsområdet"
  },
  "EXPLORER$LOADING_WORKSPACE_MESSAGE": {
    "en": "Loading workspace...",
@@ -600,7 +605,8 @@
    "pt": "Carregando espaço de trabalho...",
    "ko-KR": "작업 공간 로딩 중...",
    "ar": "جارٍ تحميل مساحة العمل...",
-    "tr": "Çalışma alanı yükleniyor..."
+    "tr": "Çalışma alanı yükleniyor...",
+    "no": "Laster arbeidsområde..."
  },
  "EXPLORER$REFRESH_ERROR_MESSAGE": {
    "en": "Error refreshing workspace",
@@ -613,7 +619,8 @@
    "pt": "Erro ao atualizar o espaço de trabalho",
    "ko-KR": "작업 공간 새로 고침 오류",
    "ar": "خطأ في تحديث مساحة العمل",
-    "tr": "Çalışma alanı yenilenirken hata oluştu"
+    "tr": "Çalışma alanı yenilenirken hata oluştu",
+    "no": "Feil ved oppdatering av arbeidsområde"
  },
  "EXPLORER$UPLOAD_SUCCESS_MESSAGE": {
    "en": "Successfully uploaded {{count}} file(s)",
@@ -626,7 +633,8 @@
    "pt": "{{count}} arquivo(s) carregado(s) com sucesso",
    "ko-KR": "{{count}}개의 파일을 성공적으로 업로드했습니다",
    "ar": "تم تحميل {{count}} ملف (ملفات) بنجاح",
-    "tr": "{{count}} dosya başarıyla yüklendi"
+    "tr": "{{count}} dosya başarıyla yüklendi",
+    "no": "Lastet opp {{count}} fil(er) vellykket"
  },
  "EXPLORER$NO_FILES_UPLOADED_MESSAGE": {
    "en": "No files were uploaded",
@@ -639,7 +647,8 @@
    "pt": "Nenhum arquivo foi carregado",
    "ko-KR": "업로드된 파일이 없습니다",
    "ar": "لم يتم تحميل أي ملفات",
-    "tr": "Hiçbir dosya yüklenmedi"
+    "tr": "Hiçbir dosya yüklenmedi",
+    "no": "Ingen filer ble lastet opp"
  },
  "EXPLORER$UPLOAD_PARTIAL_SUCCESS_MESSAGE": {
    "en": "{{count}} file(s) were skipped during upload",
@@ -652,7 +661,8 @@
    "pt": "{{count}} arquivo(s) foram ignorados durante o upload",
    "ko-KR": "업로드 중 {{count}}개의 파일이 건너뛰어졌습니다",
    "ar": "تم تخطي {{count}} ملف (ملفات) أثناء التحميل",
-    "tr": "Yükleme sırasında {{count}} dosya atlandı"
+    "tr": "Yükleme sırasında {{count}} dosya atlandı",
+    "no": "{{count}} fil(er) ble hoppet over under opplasting"
  },
  "EXPLORER$UPLOAD_UNEXPECTED_RESPONSE_MESSAGE": {
    "en": "Unexpected response structure from server",
@@ -665,7 +675,18 @@
    "pt": "Estrutura de resposta inesperada do servidor",
    "ko-KR": "서버로부터 예상치 못한 응답 구조",
    "ar": "بنية استجابة غير متوقعة من الخادم",
-    "tr": "Sunucudan beklenmeyen yanıt yapısı"
+    "tr": "Sunucudan beklenmeyen yanıt yapısı",
+    "no": "Uventet responsstruktur fra serveren"
+  },
+  "EXPLORER$VSCODE_SWITCHING_MESSAGE": {
+    "en": "Switching to VS Code in 3 seconds...\nImportant: Please inform the agent of any changes you make in VS Code. To avoid conflicts, wait for the assistant to complete its work before making your own changes.",
+    "zh-CN": "3 秒后切换到 VS Code\n重要提示：请告知 OpenHands 您在 VS Code 中进行的任何更改。为了避免冲突，请在 OpenHands 完成工作后再进行自己的更改。",
+    "zh-TW": "3 秒後切換到 VS Code\n重要提示：請告知 OpenHands 您在 VS Code 中進行的任何更改。為避免衝突，請在 OpenHands 完成工作後再進行自己的更改。"
+  },
+  "EXPLORER$VSCODE_SWITCHING_ERROR_MESSAGE": {
+    "en": "Error switching to VS Code: {{error}}",
+    "zh-CN": "切换到 VS Code 时发生错误: {{error}}",
+    "zh-TW": "切換到 VS Code 時發生錯誤: {{error}}"
  },
  "LOAD_SESSION$MODAL_TITLE": {
    "en": "Return to existing session?",
@@ -799,95 +820,325 @@
  },
  "FEEDBACK$EMAIL_PLACEHOLDER": {
    "en": "Enter your email address",
-    "es": "Ingresa tu correo electrónico"
+    "es": "Ingresa tu correo electrónico",
+    "zh-CN": "输入您的电子邮件地址",
+    "zh-TW": "輸入您的電子郵件地址",
+    "ko-KR": "이메일 주소를 입력하세요",
+    "no": "Skriv inn din e-postadresse",
+    "ar": "أدخل عنوان بريدك الإلكتروني",
+    "de": "Geben Sie Ihre E-Mail-Adresse ein",
+    "fr": "Entrez votre adresse e-mail",
+    "it": "Inserisci il tuo indirizzo email",
+    "pt": "Digite seu endereço de e-mail",
+    "tr": "E-posta adresinizi girin"
  },
  "FEEDBACK$PASSWORD_COPIED_MESSAGE": {
    "en": "Password copied to clipboard.",
-    "es": "Contraseña copiada al portapapeles."
+    "es": "Contraseña copiada al portapapeles.",
+    "zh-CN": "密码已复制到剪贴板。",
+    "zh-TW": "密碼已複製到剪貼板。",
+    "ko-KR": "비밀번호가 클립보드에 복사되었습니다.",
+    "no": "Passord kopiert til utklippstavlen.",
+    "ar": "تم نسخ كلمة المرور إلى الحافظة.",
+    "de": "Passwort in die Zwischenablage kopiert.",
+    "fr": "Mot de passe copié dans le presse-papiers.",
+    "it": "Password copiata negli appunti.",
+    "pt": "Senha copiada para a área de transferência.",
+    "tr": "Parola panoya kopyalandı."
  },
  "FEEDBACK$GO_TO_FEEDBACK": {
    "en": "Go to shared feedback",
-    "es": "Ir a feedback compartido"
+    "es": "Ir a feedback compartido",
+    "zh-CN": "转到共享反馈",
+    "zh-TW": "前往共享反饋",
+    "ko-KR": "공유된 피드백으로 이동",
+    "no": "Gå til delt tilbakemelding",
+    "ar": "الذهاب إلى التعليقات المشتركة",
+    "de": "Zum geteilten Feedback gehen",
+    "fr": "Aller aux commentaires partagés",
+    "it": "Vai al feedback condiviso",
+    "pt": "Ir para feedback compartilhado",
+    "tr": "Paylaşılan geri bildirimlere git"
  },
  "FEEDBACK$PASSWORD": {
    "en": "Password:",
-    "es": "Contraseña:"
+    "es": "Contraseña:",
+    "zh-CN": "密码：",
+    "zh-TW": "密碼：",
+    "ko-KR": "비밀번호:",
+    "no": "Passord:",
+    "ar": "كلمة المرور:",
+    "de": "Passwort:",
+    "fr": "Mot de passe :",
+    "it": "Password:",
+    "pt": "Senha:",
+    "tr": "Parola:"
  },
  "FEEDBACK$INVALID_EMAIL_FORMAT": {
    "en": "Invalid email format",
-    "es": "Formato de correo inválido"
+    "es": "Formato de correo inválido",
+    "zh-CN": "无效的电子邮件格式",
+    "zh-TW": "無效的電子郵件格式",
+    "ko-KR": "잘못된 이메일 형식",
+    "no": "Ugyldig e-postformat",
+    "ar": "تنسيق البريد الإلكتروني غير صالح",
+    "de": "Ungültiges E-Mail-Format",
+    "fr": "Format d'e-mail invalide",
+    "it": "Formato email non valido",
+    "pt": "Formato de e-mail inválido",
+    "tr": "Geçersiz e-posta biçimi"
  },
  "FEEDBACK$FAILED_TO_SHARE": {
    "en": "Failed to share, please contact the developers:",
-    "es": "Error al compartir, por favor contacta con los desarrolladores:"
+    "es": "Error al compartir, por favor contacta con los desarrolladores:",
+    "zh-CN": "分享失败，请联系开发人员：",
+    "zh-TW": "分享失敗，請聯繫開發人員：",
+    "ko-KR": "공유 실패, 개발자에게 문의하세요:",
+    "no": "Deling mislyktes, vennligst kontakt utviklerne:",
+    "ar": "فشل المشاركة، يرجى الاتصال بالمطورين:",
+    "de": "Teilen fehlgeschlagen, bitte kontaktieren Sie die Entwickler:",
+    "fr": "Échec du partage, veuillez contacter les développeurs :",
+    "it": "Condivisione fallita, contattare gli sviluppatori:",
+    "pt": "Falha ao compartilhar, entre em contato com os desenvolvedores:",
+    "tr": "Paylaşım başarısız, lütfen geliştiricilerle iletişime geçin:"
  },
  "FEEDBACK$COPY_LABEL": {
    "en": "Copy",
-    "es": "Copiar"
+    "es": "Copiar",
+    "zh-CN": "复制",
+    "zh-TW": "複製",
+    "ko-KR": "복사",
+    "no": "Kopier",
+    "ar": "نسخ",
+    "de": "Kopieren",
+    "fr": "Copier",
+    "it": "Copia",
+    "pt": "Copiar",
+    "tr": "Kopyala"
  },
  "FEEDBACK$SHARING_SETTINGS_LABEL": {
    "en": "Sharing settings",
-    "es": "Configuración de compartir"
+    "es": "Configuración de compartir",
+    "zh-CN": "共享设置",
+    "zh-TW": "共享設定",
+    "ko-KR": "공유 설정",
+    "no": "Delingsinnstillinger",
+    "ar": "إعدادات المشاركة",
+    "de": "Freigabeeinstellungen",
+    "fr": "Paramètres de partage",
+    "it": "Impostazioni di condivisione",
+    "pt": "Configurações de compartilhamento",
+    "tr": "Paylaşım ayarları"
  },
  "SECURITY$UNKNOWN_ANALYZER_LABEL":{
    "en": "Unknown security analyzer chosen",
-    "es": "Analizador de seguridad desconocido"
+    "es": "Analizador de seguridad desconocido",
+    "zh-CN": "选择了未知的安全分析器",
+    "zh-TW": "選擇了未知的安全分析器",
+    "ko-KR": "알 수 없는 보안 분석기가 선택되었습니다",
+    "no": "Ukjent sikkerhetsanalysator valgt",
+    "ar": "تم اختيار محلل أمان غير معروف",
+    "de": "Unbekannter Sicherheitsanalysator ausgewählt",
+    "fr": "Analyseur de sécurité inconnu choisi",
+    "it": "Analizzatore di sicurezza sconosciuto selezionato",
+    "pt": "Analisador de segurança desconhecido escolhido",
+    "tr": "Bilinmeyen güvenlik analizörü seçildi"
  },
  "INVARIANT$UPDATE_POLICY_LABEL": {
    "en": "Update Policy",
-    "es": "Actualizar política"
+    "es": "Actualizar política",
+    "zh-CN": "更新策略",
+    "zh-TW": "更新策略",
+    "ko-KR": "정책 업데이트",
+    "no": "Oppdater policy",
+    "ar": "تحديث السياسة",
+    "de": "Richtlinie aktualisieren",
+    "fr": "Mettre à jour la politique",
+    "it": "Aggiorna policy",
+    "pt": "Atualizar política",
+    "tr": "İlkeyi güncelle"
  },
  "INVARIANT$UPDATE_SETTINGS_LABEL": {
    "en": "Update Settings",
-    "es": "Actualizar configuración"
+    "es": "Actualizar configuración",
+    "zh-CN": "更新设置",
+    "zh-TW": "更新設定",
+    "ko-KR": "설정 업데이트",
+    "no": "Oppdater innstillinger",
+    "ar": "تحديث الإعدادات",
+    "de": "Einstellungen aktualisieren",
+    "fr": "Mettre à jour les paramètres",
+    "it": "Aggiorna impostazioni",
+    "pt": "Atualizar configurações",
+    "tr": "Ayarları güncelle"
  },
  "INVARIANT$SETTINGS_LABEL": {
    "en": "Settings",
-    "es": "Configuración"
+    "es": "Configuración",
+    "zh-CN": "设置",
+    "zh-TW": "設定",
+    "ko-KR": "설정",
+    "no": "Innstillinger",
+    "ar": "الإعدادات",
+    "de": "Einstellungen",
+    "fr": "Paramètres",
+    "it": "Impostazioni",
+    "pt": "Configurações",
+    "tr": "Ayarlar"
  },
  "INVARIANT$ASK_CONFIRMATION_RISK_SEVERITY_LABEL": {
    "en": "Ask for user confirmation on risk severity:",
-    "es": "Preguntar por confirmación del usuario sobre severidad del riesgo:"
+    "es": "Preguntar por confirmación del usuario sobre severidad del riesgo:",
+    "zh-CN": "询问用户确认风险等级：",
+    "zh-TW": "詢問用戶確認風險等級：",
+    "ko-KR": "위험 심각도에 대한 사용자 확인 요청:",
+    "no": "Be om brukerbekreftelse på risikoalvorlighet:",
+    "ar": "اطلب تأكيد المستخدم على مستوى الخطورة:",
+    "de": "Nach Benutzerbestätigung für Risikoschweregrad fragen:",
+    "fr": "Demander la confirmation de l'utilisateur sur la gravité du risque :",
+    "it": "Chiedi conferma all'utente sulla gravità del rischio:",
+    "pt": "Solicitar confirmação do usuário sobre a gravidade do risco:",
+    "tr": "Risk şiddeti için kullanıcı onayı iste:"
  },
  "INVARIANT$DONT_ASK_FOR_CONFIRMATION_LABEL": {
    "en": "Don't ask for confirmation",
-    "es": "No solicitar confirmación"
+    "es": "No solicitar confirmación",
+    "zh-CN": "不要请求确认",
+    "zh-TW": "不要請求確認",
+    "ko-KR": "확인 요청하지 않음",
+    "no": "Ikke spør om bekreftelse",
+    "ar": "لا تطلب التأكيد",
+    "de": "Nicht nach Bestätigung fragen",
+    "fr": "Ne pas demander de confirmation",
+    "it": "Non chiedere conferma",
+    "pt": "Não solicitar confirmação",
+    "tr": "Onay isteme"
  },
  "INVARIANT$INVARIANT_ANALYZER_LABEL": {
    "en": "Invariant Analyzer",
-    "es": "Analizador de invariantes"
+    "es": "Analizador de invariantes",
+    "zh-CN": "不变量分析器",
+    "zh-TW": "不變量分析器",
+    "ko-KR": "불변성 분석기",
+    "no": "Invariant-analysator",
+    "ar": "محلل الثوابت",
+    "de": "Invarianten-Analysator",
+    "fr": "Analyseur d'invariants",
+    "it": "Analizzatore di invarianti",
+    "pt": "Analisador de invariantes",
+    "tr": "Değişmez Analizörü"
  },
  "INVARIANT$INVARIANT_ANALYZER_MESSAGE": {
    "en": "Invariant Analyzer continuously monitors your OpenHands agent for security issues.",
-    "es": "Analizador de invariantes continuamente monitorea tu agente de OpenHands por problemas de seguridad."
+    "es": "Analizador de invariantes continuamente monitorea tu agente de OpenHands por problemas de seguridad.",
+    "zh-CN": "不变量分析器持续监控您的 OpenHands 代理的安全问题。",
+    "zh-TW": "不變量分析器持續監控您的 OpenHands 代理的安全問題。",
+    "ko-KR": "불변성 분석기는 OpenHands 에이전트의 보안 문제를 지속적으로 모니터링합니다.",
+    "no": "Invariant-analysatoren overvåker kontinuerlig OpenHands-agenten din for sikkerhetsproblemer.",
+    "ar": "يراقب محلل الثوابت وكيل OpenHands الخاص بك باستمرار للتحقق من المشاكل الأمنية.",
+    "de": "Der Invarianten-Analysator überwacht kontinuierlich Ihren OpenHands-Agenten auf Sicherheitsprobleme.",
+    "fr": "L'analyseur d'invariants surveille en permanence votre agent OpenHands pour détecter les problèmes de sécurité.",
+    "it": "L'analizzatore di invarianti monitora continuamente il tuo agente OpenHands per problemi di sicurezza.",
+    "pt": "O analisador de invariantes monitora continuamente seu agente OpenHands em busca de problemas de segurança.",
+    "tr": "Değişmez Analizörü, OpenHands ajanınızı güvenlik sorunları için sürekli olarak izler."
  },
  "INVARIANT$CLICK_TO_LEARN_MORE_LABEL": {
    "en": "Click to learn more",
-    "es": "Clic para aprender más"
+    "es": "Clic para aprender más",
+    "zh-CN": "点击了解更多",
+    "zh-TW": "點擊了解更多",
+    "ko-KR": "자세히 알아보기",
+    "no": "Klikk for å lære mer",
+    "ar": "انقر لمعرفة المزيد",
+    "de": "Klicken Sie, um mehr zu erfahren",
+    "fr": "Cliquez pour en savoir plus",
+    "it": "Clicca per saperne di più",
+    "pt": "Clique para saber mais",
+    "tr": "Daha fazla bilgi için tıklayın"
  },
  "INVARIANT$POLICY_LABEL": {
    "en": "Policy",
-    "es": "Política"
+    "es": "Política",
+    "zh-CN": "策略",
+    "zh-TW": "策略",
+    "ko-KR": "정책",
+    "no": "Policy",
+    "ar": "السياسة",
+    "de": "Richtlinie",
+    "fr": "Politique",
+    "it": "Policy",
+    "pt": "Política",
+    "tr": "İlke"
  },
  "INVARIANT$LOG_LABEL": {
    "en": "Logs",
-    "es": "Logs"
+    "es": "Logs",
+    "zh-CN": "日志",
+    "zh-TW": "日誌",
+    "ko-KR": "로그",
+    "no": "Logger",
+    "ar": "السجلات",
+    "de": "Protokolle",
+    "fr": "Journaux",
+    "it": "Log",
+    "pt": "Logs",
+    "tr": "Günlükler"
  },
  "INVARIANT$EXPORT_TRACE_LABEL": {
    "en": "Export Trace",
-    "es": "Exportar traza"
+    "es": "Exportar traza",
+    "zh-CN": "导出跟踪",
+    "zh-TW": "匯出追蹤",
+    "ko-KR": "추적 내보내기",
+    "no": "Eksporter sporing",
+    "ar": "تصدير التتبع",
+    "de": "Ablaufverfolgung exportieren",
+    "fr": "Exporter la trace",
+    "it": "Esporta traccia",
+    "pt": "Exportar rastreamento",
+    "tr": "İzlemeyi dışa aktar"
  },
  "INVARIANT$TRACE_EXPORTED_MESSAGE": {
    "en": "Trace exported",
-    "es": "Traza exportada"
+    "es": "Traza exportada",
+    "zh-CN": "跟踪已导出",
+    "zh-TW": "追蹤已匯出",
+    "ko-KR": "추적 내보내기 완료",
+    "no": "Sporing eksportert",
+    "ar": "تم تصدير التتبع",
+    "de": "Ablaufverfolgung exportiert",
+    "fr": "Trace exportée",
+    "it": "Traccia esportata",
+    "pt": "Rastreamento exportado",
+    "tr": "İzleme dışa aktarıldı"
  },
  "INVARIANT$POLICY_UPDATED_MESSAGE": {
    "en": "Policy updated",
-    "es": "Política actualizada"
+    "es": "Política actualizada",
+    "zh-CN": "策略已更新",
+    "zh-TW": "策略已更新",
+    "ko-KR": "정책이 업데이트되었습니다",
+    "no": "Policy oppdatert",
+    "ar": "تم تحديث السياسة",
+    "de": "Richtlinie aktualisiert",
+    "fr": "Politique mise à jour",
+    "it": "Policy aggiornata",
+    "pt": "Política atualizada",
+    "tr": "İlke güncellendi"
  },
  "INVARIANT$SETTINGS_UPDATED_MESSAGE": {
    "en": "Settings updated",
-    "es": "Configuración actualizada"
+    "es": "Configuración actualizada",
+    "zh-CN": "设置已更新",
+    "zh-TW": "設定已更新",
+    "ko-KR": "설정이 업데이트되었습니다",
+    "no": "Innstillinger oppdatert",
+    "ar": "تم تحديث الإعدادات",
+    "de": "Einstellungen aktualisiert",
+    "fr": "Paramètres mis à jour",
+    "it": "Impostazioni aggiornate",
+    "pt": "Configurações atualizadas",
+    "tr": "Ayarlar güncellendi"
  },
  "CHAT_INTERFACE$INITIALIZING_AGENT_LOADING_MESSAGE": {
    "en": "Starting up!",
@@ -1276,7 +1527,8 @@
    "pt": "Conversa de chat",
    "es": "Conversación de chat",
    "ar": "محادثة تلقيم",
-    "fr": "Conversation de chat"
+    "fr": "Conversation de chat",
+    "tr": "Sohbet Konuşması"
  },
  "CHAT_INTERFACE$UNKNOWN_SENDER": {
    "en": "Unknown",
--- a/frontend/src/assets/arrow-send.svg
+++ b/frontend/src/assets/arrow-send.svg
--- a/frontend/src/assets/build-it.svg
+++ b/frontend/src/assets/build-it.svg
--- a/frontend/src/assets/clip.svg
+++ b/frontend/src/assets/clip.svg
--- a/frontend/src/assets/clipboard.svg
+++ b/frontend/src/assets/clipboard.svg
--- a/frontend/src/assets/close.svg
+++ b/frontend/src/assets/close.svg
--- a/frontend/src/assets/cloud-connection.svg
+++ b/frontend/src/assets/cloud-connection.svg
--- a/frontend/src/assets/code.svg
+++ b/frontend/src/assets/code.svg
--- a/frontend/src/assets/default-user.svg
+++ b/frontend/src/assets/default-user.svg
--- a/frontend/src/assets/docs.svg
+++ b/frontend/src/assets/docs.svg
--- a/frontend/src/assets/ellipsis-h.svg
+++ b/frontend/src/assets/ellipsis-h.svg
--- a/frontend/src/assets/external-link.svg
+++ b/frontend/src/assets/external-link.svg
--- a/frontend/src/assets/globe.svg
+++ b/frontend/src/assets/globe.svg
--- a/frontend/src/assets/lightbulb.svg
+++ b/frontend/src/assets/lightbulb.svg
--- a/frontend/src/assets/list-type-number.svg
+++ b/frontend/src/assets/list-type-number.svg
--- a/frontend/src/assets/loading-outer.svg
+++ b/frontend/src/assets/loading-outer.svg
--- a/frontend/src/assets/message.svg
+++ b/frontend/src/assets/message.svg
--- a/frontend/src/assets/new-project.svg
+++ b/frontend/src/assets/new-project.svg
--- a/Show More
+++ b/Show More
Author	SHA1	Message	Date
openhands	c088a08e51	style: Fix linting in test_listen.py	2024-11-16 14:24:11 +00:00
openhands	2ce806e411	fix: Convert ResolverOutput to dict in response	2024-11-16 14:19:59 +00:00
openhands	45a1486f24	style: Format imports in listen.py	2024-11-16 13:54:26 +00:00
openhands	1f53c930fe	refactor: Remove unused SendPullRequestDataModel	2024-11-16 13:51:38 +00:00
openhands	dbf560d21b	refactor: Improve resolver API endpoints 1. Delete send_pull_request endpoint in listen.py 2. Remove file writing dependency in resolve_issue endpoint 3. Rename process_single_issue to create_pull_request_from_resolver_output 4. Update all references and fix tests	2024-11-16 13:49:22 +00:00
openhands	845f1b25ea	style: Fix linting issues	2024-11-16 13:14:43 +00:00
openhands	031e20105e	feat: Combine resolve_issue and send_pull_request API calls into a single endpoint	2024-11-16 13:11:55 +00:00
Graham Neubig	f03748226a	Merge branch 'main' into add-resolver-api-endpoints	2024-11-16 08:02:18 -05:00
openhands	27592c504a	fix: Fix send-pr endpoint to use correct file path and update tests to use Pydantic models	2024-11-16 13:01:24 +00:00
Ryan H. Tran	97f3249205	Move linter and diff utils to openhands-aci (#5020 )	2024-11-16 06:58:26 +01:00
sp.wack	9d47ddba38	Reduce output from frontend tests (#5023 )	2024-11-16 06:57:41 +01:00
OpenHands	f7652bd558	Fix issue #5080 : [Bug]: lint-fix.yml github action doesn't work on a branch not from this repo (#5081 )	2024-11-16 06:55:41 +01:00
openhands	a4f577222a	Fix pr #5058 : Add API endpoints for resolver functionality	2024-11-16 04:08:08 +00:00
openhands	c2265e83c5	Fix pr #5058 : Add API endpoints for resolver functionality	2024-11-16 03:02:33 +00:00
Graham Neubig	87925dd876	Merge branch 'main' into add-resolver-api-endpoints	2024-11-15 21:54:20 -05:00
OpenHands	2b7932b46c	Fix issue #5070 : [Bug]: lint-fix workflow is failing (#5078 )	2024-11-16 01:43:49 +00:00
Graham Neubig	95884c1c74	Lint	2024-11-15 20:01:16 -05:00
OpenHands	7074e45ec3	Fix issue #5059 : [Bug]: Github resolver looking for wrong PR number (#5062 ) Co-authored-by: Graham Neubig <neubig@gmail.com>	2024-11-15 19:41:48 -05:00
Graham Neubig	cfd3911f2b	Refactor resolver endpoints to use data models (#5073 ) Co-authored-by: openhands <openhands@all-hands.dev>	2024-11-15 18:11:01 -05:00
Graham Neubig	abde56ff7e	Update	2024-11-15 17:28:21 -05:00
openhands	66b4e5d14b	Fix failing tests in test_listen.py	2024-11-15 21:56:11 +00:00
Graham Neubig	cb92518f1b	Merge branch 'main' into add-resolver-api-endpoints	2024-11-15 16:18:28 -05:00
openhands	486355bfd5	Improve process_single_issue return type and error handling	2024-11-15 21:18:08 +00:00
Raymond Xu	a679fcc3b5	[docs] add tips from Graham Neubig on how to make good contributions (#5012 ) Co-authored-by: Graham Neubig <neubig@gmail.com>	2024-11-15 21:15:11 +00:00
Raymond Xu	8b1d5f5a3b	Always push repo or make a PR, comment (#5063 )	2024-11-15 21:14:47 +00:00
mamoodi	9882b62777	Update some OpenHands repo documentation and the official document site (#5060 )	2024-11-15 20:48:02 +00:00
OpenHands	b49bdb9d85	Fix issue #5064 : lint-fix github action (#5065 )	2024-11-15 15:47:24 -05:00
openhands	fba35a4be8	Revert changes to send_pull_request.py Keep the original implementation to avoid modifying core functionality	2024-11-15 18:32:58 +00:00
openhands	73e190e0f1	Add API endpoints for resolver functionality - Add /api/resolver/resolve-issue endpoint to resolve GitHub issues - Add /api/resolver/send-pr endpoint to create PRs/branches - Add tests for new endpoints - Make resolver functions async for better API integration	2024-11-15 18:29:22 +00:00
mamoodi	00ffc33d1b	Release 0.14.0 (#5027 )	2024-11-15 16:02:02 +00:00
sp.wack	1acb66c2b3	feat(frontend): Create push to Github action button in chat interface (#4993 )	2024-11-15 15:12:13 +00:00
Xingyao Wang	5b3db1bd33	feat: make add_in_context_learning_example configurable in fn call converter (#5018 )	2024-11-15 23:05:05 +08:00
Xingyao Wang	bdc4513937	fix(swebench): handle error in eval_infer and run_infer (#5017 )	2024-11-15 23:04:56 +08:00
sp.wack	ffc4d32440	feat(frontend): Keep prompt after project upload or repo selection (#4925 )	2024-11-15 16:56:47 +02:00
sp.wack	9cd248d475	feat(frontend): Display runtime ID in the browser console if available (#4978 )	2024-11-15 16:38:31 +02:00
OpenHands	5f52eebb40	Fix issue #5021 : Add links to the resolver messages (#5022 )	2024-11-15 13:05:25 +00:00
Graham Neubig	b0c4580999	Update openhands-resolver.yml with correct package name (#5014 )	2024-11-15 06:48:18 -05:00
Robert Brennan	f3b35663e9	fix zip downloads (#5009 )	2024-11-14 17:17:36 -05:00
OpenHands	be92965209	Fix issue #4944 : [Bug]: Missing GitHub token link in account settings (#4946 ) Co-authored-by: amanape <83104063+amanape@users.noreply.github.com>	2024-11-14 22:21:02 +02:00
sp.wack	89b304ccb7	refactor(frontend): Improve chat input padding (#4928 )	2024-11-14 22:19:04 +02:00
sp.wack	01cacf7c33	feat(frontend): Wait for events before rendering messages (#4994 ) Co-authored-by: mamoodi <mamoodiha@gmail.com>	2024-11-14 22:09:29 +02:00
Engel Nyst	fac5237c69	Fix user commands in terminal with function calling (#4955 ) Co-authored-by: Xingyao Wang <xingyao6@illinois.edu> Co-authored-by: Xingyao Wang <xingyao@all-hands.dev>	2024-11-14 19:14:36 +00:00
Robert Brennan	c784151765	fix file descriptor leaks (#4988 ) Co-authored-by: openhands <openhands@all-hands.dev>	2024-11-14 14:06:33 -05:00
Graham Neubig	ce6f99d80e	Add GITHUB_USERNAME env var to resolver step (#4999 ) Co-authored-by: openhands <openhands@all-hands.dev>	2024-11-14 18:42:59 +00:00
Ketan Ramaneti	852c90f64a	[fix eval] Fix issues with miniwob remote runtime evaluation (#5001 )	2024-11-14 18:00:48 +00:00
Ketan Ramaneti	42b49e6c43	[fix eval] Fix issues with aider_bench remote runtime evaluation (#5000 )	2024-11-14 17:58:45 +00:00
Xingyao Wang	07f0d1ccb3	feat(llm): convert function call request for non-funcall OSS model (#4711 ) Co-authored-by: Calvin Smith <email@cjsmith.io>	2024-11-15 00:40:09 +08:00
Robert Brennan	52a428d74a	Fix markdown ordered list numbering (#4989 ) Co-authored-by: openhands <openhands@all-hands.dev>	2024-11-14 10:59:48 -05:00
OpenHands	27cd507cd2	Fix issue #4985 : [Bug]: Cannot exit the session when on Jupyter or Browser tab in the UI (#4986 )	2024-11-14 10:06:35 -05:00
Graham Neubig	a753babb7a	Integrate OpenHands resolver into main repository (#4964 ) Co-authored-by: openhands <openhands@all-hands.dev> Co-authored-by: Rohit Malhotra <rohitvinodmalhotra@gmail.com>	2024-11-14 09:45:46 -05:00
Rohit Malhotra	38dc41ca42	Fix: [Bug] Do not render editor action buttons (save/discard) when displaying non-code files (#4903 )	2024-11-14 09:09:28 +02:00
Engel Nyst	8dee334236	Context Window Exceeded fix (#4977 )	2024-11-14 02:42:39 +00:00
Engel Nyst	a93f1402de	Clean up file logs (#4979 )	2024-11-13 20:17:21 +00:00
Robert Brennan	bc3f0ac24a	fix imports (#4974 )	2024-11-13 17:04:16 +00:00
Robert Brennan	f55ddbed0e	fix docker leak (#4970 )	2024-11-14 00:23:07 +08:00
Xingyao Wang	fd81670ba8	feat: add VSCode to OpenHands runtime and UI (#4745 ) Co-authored-by: openhands <openhands@all-hands.dev> Co-authored-by: Robert Brennan <accounts@rbren.io>	2024-11-14 00:20:49 +08:00
sp.wack	79ed4e3567	fix(frontend): Recover full message history if exists (#4961 )	2024-11-13 15:38:30 +02:00
sp.wack	b3fbbbaa9d	feat(frontend): Move posthog key to config and upgrade posthog-js (#4940 )	2024-11-13 07:56:04 +00:00
tofarr	87c02177d7	Reconnecting websockets (#4954 )	2024-11-13 09:38:26 +02:00
OpenHands	207df9dd30	Fix issue #4912 : [Bug]: BedrockException: "The number of toolResult blocks at messages.2.content exceeds the number of toolUse blocks of previous turn.". (#4937 ) Co-authored-by: Xingyao Wang <xingyao@all-hands.dev> Co-authored-by: Graham Neubig <neubig@gmail.com> Co-authored-by: mamoodi <mamoodiha@gmail.com>	2024-11-12 17:23:11 -05:00
tofarr	59f7093428	Fix max iterations (#4949 )	2024-11-12 21:09:43 +00:00
sp.wack	123fb4b75d	feat(posthog): Add saas login event (#4948 )	2024-11-12 20:37:59 +00:00
mamoodi	40e2d28e87	Release 0.13.1 (#4947 )	2024-11-12 15:08:10 -05:00
OpenHands	c555611d58	Fix issue #4941 : [Bug]: Browser tab does not reset after starting a new session (#4945 ) Co-authored-by: amanape <83104063+amanape@users.noreply.github.com>	2024-11-12 19:40:12 +00:00
Calvin Smith	50e7da9c3d	fix(evaluation): SWE-bench evaluation script supports multiprocessing (#4943 )	2024-11-12 12:19:57 -07:00
sp.wack	0cfb132ab7	fix(frontend): Remove dotted outline on focus (#4926 )	2024-11-12 18:27:06 +02:00
Robert Brennan	17f4c6e1a9	Refactor sessions a bit, and fix issue where runtimes get killed (#4900 )	2024-11-12 16:20:36 +00:00
Xingyao Wang	910b283ac2	fix(llm): bedrock throw errors if content contains empty string (#4935 )	2024-11-12 15:53:22 +00:00
OpenHands	b54724ac3f	Fix issue #4931 : Make use of microagents configurable in `codeact_agent` (#4932 ) Co-authored-by: Graham Neubig <neubig@gmail.com>	2024-11-12 15:42:13 +00:00
Robert Brennan	0633a99298	Fix resume runtime after a pause (#4904 )	2024-11-12 09:03:02 -05:00
Ryan H. Tran	d9c5f11046	Replace file editor with openhands-aci (#4782 )	2024-11-12 21:26:33 +08:00
Engel Nyst	32fdcd58e5	Update litellm (#4927 )	2024-11-12 11:24:19 +00:00
sp.wack	de71b7cdb8	test(frontend): Fix failing e2e test due to mock delay (#4923 )	2024-11-12 10:50:38 +00:00
sp.wack	04aeccfb69	fix(frontend): Remove quotes from suggestion (#4921 )	2024-11-12 12:30:43 +02:00
Faraz Shamim	4eea1286d4	Issue #4399 : Replaced all occurences (#4878 ) Co-authored-by: sp.wack <83104063+amanape@users.noreply.github.com>	2024-11-12 10:58:09 +01:00
Robert Brennan	488a320ffd	update to use github client lib (#4909 )	2024-11-12 00:56:50 +00:00
Robert Brennan	377fadc2eb	fix remote runtimes (#4902 )	2024-11-12 00:02:34 +00:00
Robert Brennan	7df7f43e3c	Revert "Add rate limiting to server endpoints" (#4910 )	2024-11-11 23:26:49 +00:00
Engel Nyst	a45aba512a	Tweak log levels (#4729 )	2024-11-11 22:51:56 +00:00
tofarr	a1a9d2f175	Refactor websocket (#4879 ) Co-authored-by: sp.wack <83104063+amanape@users.noreply.github.com>	2024-11-11 22:36:07 +00:00
Robert Brennan	79492b6551	Add rate limiting to server endpoints (#4867 ) Co-authored-by: openhands <openhands@all-hands.dev>	2024-11-11 16:54:22 -05:00
sp.wack	80fdb9a2f4	feat(posthog): Emit user activated event (#4886 )	2024-11-11 23:31:41 +02:00
Nafis Reza	975e75531d	Move assets/icons to dedicated folder (#4850 )	2024-11-11 20:17:04 +00:00
Robert Brennan	1b5f5bcdad	fixes for upcoming changes to remote API (#4834 )	2024-11-11 14:51:14 -05:00
Rohit Malhotra	8c00d96024	Support displaying images/videos/pdfs in the workspace (#4898 )	2024-11-11 20:22:17 +02:00
Robert Brennan	bf8ccc8fc3	fix infinite loop (#4873 ) Co-authored-by: amanape <83104063+amanape@users.noreply.github.com>	2024-11-11 10:59:43 +00:00
OpenHands	037d770f66	Fix issue #4884 : (chore) add missing FE translations (#4885 ) Co-authored-by: tobitege <10787084+tobitege@users.noreply.github.com>	2024-11-11 10:09:46 +00:00
sp.wack	dd50246672	test(frontend): Pass failing tests (#4887 )	2024-11-11 09:49:56 +00:00
Graham Neubig	090771674c	Update llms.md w/ more recent results (#4874 )	2024-11-10 03:12:09 +00:00